diff --git a/docs/advanced.rst b/docs/advanced.rst index 963ff797..6e59b3fc 100644 --- a/docs/advanced.rst +++ b/docs/advanced.rst @@ -114,6 +114,41 @@ exceed a certain number of megapixels with ``--skip-big``. (A 300 DPI, # Allow 300 seconds for OCR; skip any page larger than 50 megapixels ocrmypdf --tesseract-timeout 300 --skip-big 50 bigfile.pdf output.pdf +OCR for huge images +------------------- + +Separate from these settings, Tesseract has internal limits on the size +of images it will process. If you issue +``--tesseract-downsample-large-images``, OCRmyPDF will downsample images +to fit Tesseract limits. (The limits are usually entered only for scanned +images of oversized media, such as large maps or blueprints exceeding +110 cm or 43 inches in either dimension, and at high DPI.) + +``--tesseract-downsample-above`` adjusts the threshold at which images +will be downsampled. By default, only images that exceed any of Tesseract's +internal limits are downsampled. + +You will also need to set ``--tesseract-timeout`` high enough to allow +for processing. + +Only the image sent for OCR is downsampled. The original image is +preserved. + +.. code-block:: bash + + # Allow 600 seconds for OCR on huge images + ocrmypdf --tesseract-timeout 600 \ + --tesseract-downsample-large-images \ + bigfile.pdf output.pdf + + # Downsample images above 5000 pixels on the longest dimension to + # 5000 pixels + ocrmypdf --tesseract-timeout 120 \ + --tesseract-downsample-large-images \ + --tesseract-downsample-above 5000 \ + bigfile.pdf output_downsampled_ocr.pdf + + Overriding default tesseract ---------------------------- diff --git a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py index 6c530eca..35bba075 100644 --- a/src/ocrmypdf/builtin_plugins/tesseract_ocr.py +++ b/src/ocrmypdf/builtin_plugins/tesseract_ocr.py @@ -30,7 +30,7 @@ def add_options(parser): action='append', metavar='CFG', default=[], - help="Additional Tesseract configuration files -- see documentation", + help="Additional Tesseract configuration files -- see documentation.", ) tess.add_argument( '--tesseract-pagesegmode', @@ -38,7 +38,7 @@ def add_options(parser): type=int, metavar='PSM', choices=range(0, 14), - help="Set Tesseract page segmentation mode (see tesseract --help)", + help="Set Tesseract page segmentation mode (see tesseract --help).", ) tess.add_argument( '--tesseract-oem', @@ -75,7 +75,10 @@ def add_options(parser): metavar='SECONDS', help=( "Give up on OCR after the timeout, but copy the preprocessed page " - "into the final output." + "into the final output. This timeout is only used when using Tesseract " + "for OCR. When Tesseract is used for other operations such as " + "deskewing and orientation, the timeout is controlled by " + "--tesseract-non-ocr-timeout." ), ) tess.add_argument( @@ -175,6 +178,15 @@ def validate(pdfinfo, options): tess_threads = int(os.environ['OMP_THREAD_LIMIT']) log.debug("Using Tesseract OpenMP thread limit %d", tess_threads) + if ( + options.tesseract_downsample_above != 32767 + and not options.tesseract_downsample_large_images + ): + log.warning( + "The --tesseract-downsample-above argument will have no effect unless " + "--tesseract-downsample-large-images is also given." + ) + @hookimpl def filter_ocr_image(page: PageContext, image: Image.Image) -> Image.Image: diff --git a/src/ocrmypdf/cli.py b/src/ocrmypdf/cli.py index 1473b846..3d658d0e 100644 --- a/src/ocrmypdf/cli.py +++ b/src/ocrmypdf/cli.py @@ -177,7 +177,9 @@ Online documentation is located at: '--image-dpi', metavar='DPI', type=int, - help="For input image instead of PDF, use this DPI instead of file's.", + help="When the input file is an image, not a PDF, use this DPI instead " + "of the DPI claimed by the input file. If the input does not claim a " + "sensible DPI, this option will be required.", ) parser.add_argument( '--output-type',