diff --git a/docs/advanced.rst b/docs/advanced.rst index be4966f8..031be6b1 100644 --- a/docs/advanced.rst +++ b/docs/advanced.rst @@ -239,46 +239,46 @@ rendering OCRmyPDF has these PDF renderers: ``sandwich`` and ``hocr``. The renderer may be selected using ``--pdf-renderer``. The default is ``auto`` which lets OCRmyPDF select the renderer to use. Currently, -``auto`` always selects ``sandwich``. - -The ``sandwich`` renderer -------------------------- - -The ``sandwich`` renderer uses Tesseract's new text-only PDF feature, -which produces a PDF page that lays out the OCR in invisible text. This -page is then "sandwiched" onto the original PDF page, allowing lossless -application of OCR even to PDF pages that contain other vector objects. - -Currently this is the best renderer for most uses, however it is -implemented in Tesseract so OCRmyPDF cannot influence it. Currently some -problematic PDF viewers like Mozilla PDF.js and macOS Preview have -problems with segmenting its text output, and -mightrunseveralwordstogether. - -When image preprocessing features like ``--deskew`` are used, the -original PDF will be rendered as a full page and the OCR layer will be -placed on top. +``auto`` always selects ``hocr``. The ``hocr`` renderer --------------------- -The ``hocr`` renderer works with older versions of Tesseract. The image -layer is copied from the original PDF page if possible, avoiding -potentially lossy transcoding or loss of other PDF information. If -preprocessing is specified, then the image layer is a new PDF. (You may -need to disable PDF/A conversion nad optimization to eliminate all -lossy transformations.) +.. versionchanged:: 16.0.0 -Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone -looking to customize how OCR is presented should look here. A major -disadvantage of this renderer is it not capable of correctly handling -text outside the Latin alphabet (specifically, it supports the ISO 8859-1 -character set). Pull requests to improve the situation are welcome. +In both renderers, a text-only layer is rendered and sandwiched (overlaid) +on to either the original PDF page, or newly rasterized version of the +original PDF page (when ``--force-ocr`` is used). In this way, loss +of PDF information is generally avoided. (You may need to disable PDF/A +conversion and optimization to eliminate all lossy transformations.) -Currently, this renderer has the best compatibility with Mozilla's -PDF.js viewer. +The current approach used by the new hOCR renderer is a re-implementation +of Tesseract's PDF renderer, using the same Glyphless font and general +ideas, but fixing many technical issues that impeded it. The new hocr +provides better text placement accuracy, avoids issues with word +segmentation, and provides better positioning of skewed text. -This works in all versions of Tesseract. +Using the experimental API, it is also possible to edit the OCR output +from Tesseract, using any tool that is capable of editing hOCR files. + +Older versions of this renderer did not support non-Latin languages, but +it is now universal. + +The ``sandwich`` renderer +------------------------- + +The ``sandwich`` renderer uses Tesseract's text-only PDF feature, +which produces a PDF page that lays out the OCR in invisible text. + +Currently some problematic PDF viewers like Mozilla PDF.js and macOS +Preview have problems with segmenting its text output, and +mightrunseveralwordstogether. It also does not implement right to left +fonts (Arabic, Hebrew, Persian). The output of this renderer cannot +be edited. The sandwich renderer is retained for testing. + +When image preprocessing features like ``--deskew`` are used, the +original PDF will be rendered as a full page and the OCR layer will be +placed on top. Rendering and rasterizing options ================================= diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index a20da3df..c0e22302 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -146,7 +146,7 @@ def check_options(options): # Decide on what renderer to use if options.pdf_renderer == 'auto': - options.pdf_renderer = 'sandwich' + options.pdf_renderer = 'hocr' if not tesseract.has_thresholding() and options.tesseract_thresholding != 0: log.warning( @@ -216,7 +216,7 @@ class TesseractOcrEngine(OcrEngine): @staticmethod def creator_tag(options): - tag = '-PDF' if options.pdf_renderer == 'sandwich' else '' + tag = '-PDF' if options.pdf_renderer == 'sandwich' else 'hOCR' return f"Tesseract OCR{tag} {TesseractOcrEngine.version()}" def __str__(self): diff --git a/tests/plugins/tesseract_debug_rotate.py b/tests/plugins/tesseract_debug_rotate.py index 305ced83..47118101 100644 --- a/tests/plugins/tesseract_debug_rotate.py +++ b/tests/plugins/tesseract_debug_rotate.py @@ -52,7 +52,7 @@ class FixedRotateNoopOcrEngine(OcrEngine): @staticmethod def creator_tag(options): - tag = '-PDF' if options.pdf_renderer == 'sandwich' else '' + tag = '-PDF' if options.pdf_renderer == 'sandwich' else '-hOCR' return f"NO-OP {tag} {FixedRotateNoopOcrEngine.version()}" def __str__(self): diff --git a/tests/plugins/tesseract_noop.py b/tests/plugins/tesseract_noop.py index 92ac8500..2573ecfe 100644 --- a/tests/plugins/tesseract_noop.py +++ b/tests/plugins/tesseract_noop.py @@ -50,7 +50,7 @@ class NoopOcrEngine(OcrEngine): @staticmethod def creator_tag(options): - tag = '-PDF' if options.pdf_renderer == 'sandwich' else '' + tag = '-PDF' if options.pdf_renderer == 'sandwich' else '-hOCR' return f"NO-OP {tag} {NoopOcrEngine.version()}" def __str__(self): diff --git a/tests/test_rotation.py b/tests/test_rotation.py index cd643ec6..c610a205 100644 --- a/tests/test_rotation.py +++ b/tests/test_rotation.py @@ -195,7 +195,7 @@ def test_rotate_deskew_ocr_timeout(resources, outdir): '--tesseract-timeout', '0', '--pdf-renderer', - 'sandwich', + 'hocr', ) cmp = compare_images_monochrome( diff --git a/tests/test_tesseract.py b/tests/test_tesseract.py index ceaf8d35..74552db9 100644 --- a/tests/test_tesseract.py +++ b/tests/test_tesseract.py @@ -49,9 +49,7 @@ def test_skip_pages_does_not_replicate(resources, basename, outdir): def test_content_preservation(resources, outpdf): infile = resources / 'masks.pdf' - check_ocrmypdf( - infile, outpdf, '--pdf-renderer', 'sandwich', '--tesseract-timeout', '0' - ) + check_ocrmypdf(infile, outpdf, '--pdf-renderer', 'hocr', '--tesseract-timeout', '0') info = pdfinfo.PdfInfo(outpdf) page = info[0]