diff --git a/.gitignore b/.gitignore index eeac20e1..14187069 100644 --- a/.gitignore +++ b/.gitignore @@ -7,6 +7,7 @@ tasks.py .bash_history .ruffus_history.sqlite .idea/ +.pytest_cache/ # Package building *.egg-info/ diff --git a/docs/introduction.rst b/docs/introduction.rst index 988454fc..1e36c5a2 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -83,6 +83,7 @@ OCRmyPDF is limited by the Tesseract OCR engine. As such it experiences these l OCRmyPDF is also limited by the PDF specification: * PDF encodes the position of text glyphs but does not encode document structure. There is no markup that divides a document in sections, paragraphs, sentences, or even words (since blank spaces are not represented). As such all elements of document structure including the spaces between words must be derived heuristically. Some PDF viewers do a better job of this than others. +* Because some popular open source PDF viewers have a particularly hard time with spaces betweem words, OCRmyPDF appends a space to each text element as a workaround. While this mixes document structure with graphical information that ideally should be left to the PDF viewer to interpret, it improves compatibility with some viewers and does not cause problems for better ones. Ghostscript also imposes some limitations: diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 92d3b385..b548f7d5 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -5,6 +5,10 @@ OCRmyPDF uses `semantic versioning `_ for its command line i The OCRmyPDF package itself does not contain a public API, although it is fairly stable and breaking changes are usually timed with a major release. A future release will clearly define the stable public API. +v5.7.0 +------ + + v5.6.3 ------ diff --git a/ocrmypdf/hocrtransform.py b/ocrmypdf/hocrtransform.py index 75faf8d3..4c8bbe90 100755 --- a/ocrmypdf/hocrtransform.py +++ b/ocrmypdf/hocrtransform.py @@ -33,6 +33,7 @@ from reportlab.lib.units import inch from xml.etree import ElementTree from PIL import Image from collections import namedtuple +from math import atan, sin, cos import re import argparse @@ -49,13 +50,24 @@ class HocrTransform(): """ A class for converting documents from the hOCR format. For details of the hOCR format, see: - http://docs.google.com/View?docid=dfxcv4vc_67g844kf + http://kba.cloud/hocr-spec/ """ + box_pattern = re.compile(r'bbox((\s+\d+){4})') + baseline_pattern = re.compile(r''' + baseline \s+ + ([\-\+]?\d*\.?\d*) \s+ # +/- decimal float + ([\-\+]?\d+) # +/- int''', re.VERBOSE) + ligatures = str.maketrans({ + 'ff': 'ff', + 'ffi': 'f‌f‌i', + 'ffl': 'f‌f‌l', + 'fi': 'fi', + 'fl': 'fl', + }) + def __init__(self, hocrFileName, dpi): self.dpi = dpi - self.boxPattern = re.compile(r'bbox((\s+\d+){4})') - self.hocr = ElementTree.parse(hocrFileName) # if the hOCR file has a namespace, ElementTree requires its use to @@ -104,19 +116,31 @@ class HocrTransform(): text += element.tail return text - def element_coordinates(self, element): + @classmethod + def element_coordinates(cls, element): """ Returns a tuple containing the coordinates of the bounding box around an element """ out = (0, 0, 0, 0) if 'title' in element.attrib: - matches = self.boxPattern.search(element.attrib['title']) + matches = cls.box_pattern.search(element.attrib['title']) if matches: coords = matches.group(1).split() out = Rect._make(int(coords[n]) for n in range(4)) return out + @classmethod + def baseline(cls, element): + """ + Returns a tuple containing the baseline slope and intercept. + """ + if 'title' in element.attrib: + matches = cls.baseline_pattern.search(element.attrib['title']) + if matches: + return float(matches.group(1)), int(matches.group(2)) + return (0, 0) + def pt_from_pixel(self, pxl): """ Returns the quantity in PDF units (pt) given quantity in pixels @@ -124,20 +148,17 @@ class HocrTransform(): return Rect._make( (c / self.dpi * inch) for c in pxl) - def replace_unsupported_chars(self, s): + @classmethod + def replace_unsupported_chars(cls, s): """ Given an input string, returns the corresponding string that: - is available in the helvetica facetype - does not contain any ligature (to allow easy search in the PDF file) """ - # The 'u' before the character to replace indicates that it is a - # unicode character - s = s.replace(u"fl", "fl") - s = s.replace(u"fi", "fi") - return s + return s.translate(cls.ligatures) def to_pdf(self, outFileName, imageFileName=None, showBoundingboxes=False, - fontname="Helvetica", invisibleText=False): + fontname="Helvetica", invisibleText=False, interwordSpaces=False): """ Creates a PDF file with an image superimposed on top of the text. Text is positioned according to the bounding box of the lines in @@ -172,57 +193,19 @@ class HocrTransform(): pdf.rect( pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, fill=1) + + found_lines = False + for line in self.hocr.findall( + ".//%sspan[@class='%s']" % (self.xmlns, "ocr_line")): + found_lines = True + self._do_line(pdf, line, "ocrx_word", fontname, invisibleText, + interwordSpaces, showBoundingboxes) - # check if element with class 'ocrx_word' are available - # otherwise use 'ocr_line' as fallback - elemclass = "ocr_line" - if self.hocr.find( - ".//%sspan[@class='ocrx_word']" % (self.xmlns)) is not None: - elemclass = "ocrx_word" - - # itterate all text elements - # light green for bounding box of word/line - pdf.setStrokeColorRGB(1, 0, 0) - pdf.setLineWidth(0.5) # bounding box line width - pdf.setDash(6, 3) # bounding box is dashed - pdf.setFillColorRGB(0, 0, 0) # text in black - for elem in self.hocr.findall( - ".//%sspan[@class='%s']" % (self.xmlns, elemclass)): - - elemtxt = self._get_element_text(elem).rstrip() - - elemtxt = self.replace_unsupported_chars(elemtxt) - - if len(elemtxt) == 0: - continue - - pxl_coords = self.element_coordinates(elem) - pt = self.pt_from_pixel(pxl_coords) - - # draw the bbox border - if showBoundingboxes: - pdf.rect( - pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, - fill=0) - - text = pdf.beginText() - fontsize = pt.y2 - pt.y1 - text.setFont(fontname, fontsize) - if invisibleText: - text.setTextRenderMode(3) # Invisible (indicates OCR text) - - # set cursor to bottom left corner of bbox (adjust for dpi) - text.setTextOrigin(pt.x1, self.height - pt.y2) - - # scale the width of the text to fill the width of the bbox - text.setHorizScale( - 100 * (pt.x2 - pt.x1) / pdf.stringWidth( - elemtxt, fontname, fontsize)) - - # write the text to the page - text.textLine(elemtxt) - pdf.drawText(text) - + if not found_lines: + # Tesseract did not report any lines (just words) + root = self.hocr.find(".//%sdiv[@class='%s']" % (self.xmlns, "ocr_page")) + self._do_line(pdf, root, "ocrx_word", fontname, invisibleText, + interwordSpaces, showBoundingboxes) # put the image on the page, scaled to fill the page if imageFileName is not None: pdf.drawImage(imageFileName, 0, 0, @@ -233,6 +216,117 @@ class HocrTransform(): pdf.save() + @classmethod + def polyval(cls, poly, x): + return x * poly[0] + poly[1] + + + def _do_line(self, pdf, line, elemclass, fontname, invisibleText, + interwordSpaces, showBoundingboxes): + pxl_line_coords = self.element_coordinates(line) + line_box = self.pt_from_pixel(pxl_line_coords) + line_height = line_box.y2 - line_box.y1 + + slope, pxl_intercept = self.baseline(line) + if abs(slope) < 0.005: + slope = 0.0 + angle = atan(slope) + cos_a, sin_a = cos(angle), sin(angle) + + text = pdf.beginText() + intercept = pxl_intercept / self.dpi * inch + + # Don't allow the font to break out of the bounding box. Division by + # cos_a accounts for extra clearance between the glyph's vertical axis + # on a sloped baseline and the edge of the bounding box. + fontsize = (line_height - abs(intercept)) / cos_a + text.setFont(fontname, fontsize) + if invisibleText: + text.setTextRenderMode(3) # Invisible (indicates OCR text) + + # Intercept is normally negative, so this places it above the bottom + # of the line box + baseline_y2 = self.height - (line_box.y2 + intercept) + + if showBoundingboxes: + # draw the baseline in magenta, dashed + pdf.setDash() + pdf.setStrokeColorRGB(0.95, 0.65, 0.95) + pdf.setLineWidth(0.5) + # negate slope because it is defined as a rise/run in pixel + # coordinates and page coordinates have the y axis flipped + pdf.line(line_box.x1, + baseline_y2, + line_box.x2, + self.polyval((-slope, baseline_y2), + line_box.x2 - line_box.x1)) + # light green for bounding box of word/line + pdf.setDash(6, 3) + pdf.setStrokeColorRGB(1, 0, 0) + + text.setTextTransform( + cos_a, -sin_a, sin_a, cos_a, + line_box.x1, baseline_y2 + ) + pdf.setFillColorRGB(0, 0, 0) # text in black + + elements = line.findall( + ".//%sspan[@class='%s']" % (self.xmlns, elemclass)) + for elem in elements: + elemtxt = self._get_element_text(elem).strip() + elemtxt = self.replace_unsupported_chars(elemtxt) + if elemtxt == '': + continue + + pxl_coords = self.element_coordinates(elem) + box = self.pt_from_pixel(pxl_coords) + if interwordSpaces: + # if `--interword-spaces` is true, append a space + # to the end of each text element to allow simpler PDF viewers + # such as PDF.js to better recognize words in search and copy + # and paste. Do not remove space from last word in line, even + # though it would look better, because it will interfere with + # naive text extraction. \n does not work either. + elemtxt += ' ' + box = Rect._make(( + box.x1, + line_box.y1, + box.x2 + pdf.stringWidth(' ', fontname, line_height), + line_box.y2)) + box_width = box.x2 - box.x1 + font_width = pdf.stringWidth(elemtxt, fontname, fontsize) + + # draw the bbox border + if showBoundingboxes: + pdf.rect( + box.x1, + self.height - line_box.y2, + box_width, + line_height, + fill=0) + + # Adjust relative position of cursor + # This is equivalent to: + # text.setTextOrigin(pt.x1, self.height - line_box.y2) + # but the former generates a full text reposition matrix (Tm) in the + # content stream while this issues a "offset" (Td) command. + # .moveCursor() is relative to start of the text line, where the + # "text line" means whatever reportlab defines it as. Do not use + # use .getCursor(), since moveCursor() rather unintuitively plans + # its moves relative to .getStartOfLine(). + # For skewed lines, in the text transform we set up a rotated + # coordinate system, so we don't have to account for the + # incremental offset. Surprisingly most PDF viewers can handle this. + cursor = text.getStartOfLine() + dx = box.x1 - cursor[0] + dy = baseline_y2 - cursor[1] + text.moveCursor(dx, dy) + + text.setHorizScale(100 * box_width / font_width) + text.textOut(elemtxt) + pdf.drawText(text) + + if __name__ == "__main__": parser = argparse.ArgumentParser(description='Convert hocr file to PDF') parser.add_argument('-b', '--boundingboxes', action="store_true", @@ -242,10 +336,12 @@ if __name__ == "__main__": help='Resolution of the image that was OCRed') parser.add_argument('-i', '--image', default=None, help='Path to the image to be placed above the text') + parser.add_argument('--interword-spaces', action='store_true', + default=False, help='Add spaces between words') parser.add_argument('hocrfile', help='Path to the hocr file to be parsed') parser.add_argument( 'outputfile', help='Path to the PDF file to be generated') args = parser.parse_args() hocr = HocrTransform(args.hocrfile, args.resolution) - hocr.to_pdf(args.outputfile, args.image, args.boundingboxes) + hocr.to_pdf(args.outputfile, args.image, args.boundingboxes, interwordSpaces=args.interword_spaces) diff --git a/ocrmypdf/pipeline.py b/ocrmypdf/pipeline.py index 6202f01b..0bd48479 100644 --- a/ocrmypdf/pipeline.py +++ b/ocrmypdf/pipeline.py @@ -640,8 +640,8 @@ def render_hocr_page( hocrtransform = HocrTransform(hocr, dpi) hocrtransform.to_pdf(output_file, imageFileName=None, - showBoundingboxes=False, invisibleText=True) - + showBoundingboxes=False, invisibleText=True, + interwordSpaces=True) def flatten_groups(groups): for obj in groups: @@ -665,8 +665,8 @@ def render_hocr_debug_page( hocrtransform = HocrTransform(hocr, dpi) hocrtransform.to_pdf(output_file, imageFileName=None, - showBoundingboxes=True, invisibleText=False) - + showBoundingboxes=True, invisibleText=False, + interwordSpaces=True) def combine_layers( infiles,