diff --git a/.gitignore b/.gitignore index 93954ec3..b2881fb4 100644 --- a/.gitignore +++ b/.gitignore @@ -13,6 +13,7 @@ build/ dist/ wheelhouse/ +pip-wheel-metadata/ # Automatically generated files docs/_build/ diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index b9a75aff..b268b628 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -3,4 +3,4 @@ repos: rev: stable hooks: - id: black - language_version: python3.6 + language_version: python3.7 diff --git a/README.md b/README.md index 75415401..3509fb71 100644 --- a/README.md +++ b/README.md @@ -38,9 +38,9 @@ Main features - If requested deskews and/or cleans the image before performing OCR - Validates input and output files - Distributes work across all available CPU cores -- Uses [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) engine -- Supports more than [100 languages](https://github.com/tesseract-ocr/tessdata) recognized by Tesseract -- Battle-tested on thousands of PDFs, a test suite and continuous integration +- Uses [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) engine to recognize more than [100 languages](https://github.com/tesseract-ocr/tessdata) +- Scales properly to handle files with thousands of pages +- Battle-tested on millions of PDFs For details: please consult the [documentation](https://ocrmypdf.readthedocs.io/en/latest/). @@ -131,6 +131,11 @@ Press & Media - [c't 1-2014, page 59](http://heise.de/-2279695): Detailed presentation of OCRmyPDF v1.0 in the leading German IT magazine c't - [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](http://heise.de/-2356670) +Business enquiries +------------------ + +OCRmyPDF would not be the software that it is today is without companies and users choosing to provide support for feature development and consulting enquiries. We are happy to discuss all enquiries, whether for extending the existing feature set, or integrating OCRmyPDF into a larger system. + License ------- @@ -138,7 +143,7 @@ The OCRmyPDF software is licensed under the GNU GPLv3. Certain files are covered The license for each test file varies, and is noted in tests/resources/README.rst. The documentation is licensed under Creative Commons Attribution-ShareAlike 4.0 (CC-BY-SA 4.0). -OCRmyPDF versions prior to 6.0 were licensed under the MIT License. +OCRmyPDF versions prior to 6.0 were distributed under the MIT License. Disclaimer ---------- diff --git a/docs/installation.rst b/docs/installation.rst index 555ad0d5..73380839 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -364,6 +364,16 @@ The command line program should now be available: ocrmypdf --help +Installing on FreeBSD +--------------------- + +FreeBSD 11.2 is known to work. Other versions likely work but have not been tested. + +In general it should work to: + +#. `Install and build pikepdf `_. +#. Install the equivalent list of dependencies for Linux. + Installing the Docker image --------------------------- @@ -490,3 +500,17 @@ To install all of the development and test requirements: pip install -r requirements/dev.txt -r requirements/test.txt To add JBIG2 encoding, see :ref:`jbig2`. + +Shell completions +----------------- + +Completions for ``bash`` and ``fish`` are available in the project's +``misc/completion`` folder. The ``bash`` completions are likely ``zsh`` +compatible but this has not been confirmed. Package maintainers, please install +these at the appropriate locations for your system. + +To manually install the ``bash`` completion, copy ``misc/completion/ocrmypdf.bash`` to +``/etc/bash_completion.d/ocrmypdf`` (rename the file). + +To manually install the ``fish`` completion, copy ``misc/completion/ocrmypdf.fish`` to +``~/.config/fish/completions/ocrmypdf.fish``. diff --git a/docs/release_notes.rst b/docs/release_notes.rst index 70f4c434..6af0f263 100644 --- a/docs/release_notes.rst +++ b/docs/release_notes.rst @@ -13,6 +13,41 @@ Note that it is licensed under GPLv3, so scripts that ``import ocrmypdf`` and ar find: [^`]\#([0-9]{1,3})[^0-9] replace: `#$1 `_ +v8.3.0 +------ + +- Improved the strategy for updating pages when a new image of the page was produced. We know attempt to preserve more content from the original file, for annotations in particular. + +- For PDFs with more than 100 pages and a sequence where one PDF page was replaced and one or more subsequent ones were skipped, an intermediate file would be corrupted while grafting OCR text, causing processing to fail. + +- Previously, we resized the images produced by Ghostscript by a small number of pixels to ensure the output image size was an exactly what we wanted. Having discovered a way to get Ghostscript to produce the exact image sizes we require, we eliminated the resizing step. + +- Command line completions for ``bash`` are now available, in addition to ``fish``, both in ``misc/completion``. Package maintainers, please install these so users can take advantage. + +- Updated requirements. + +- pikepdf 1.3.0 is now required. + +v8.2.4 +------ + +- Fixed a false positive while checking for a certain type of PDF that only Acrobat can read. We now more accurately detect Acrobat-only PDFs. + +- OCRmyPDF holds fewer open file handles and is more prompt about releasing those it no longer needs. + +- Minor optimization: we no longer traverse the table of contents to ensure all references in it are resolved, as changes to libqpdf have made this unnecessary. + +- pikepdf 1.2.0 is now required. + +v8.2.3 +------ + +- Fixed that ``--mask-barcodes`` would occasionally leave a unwanted temporary file named ``junkpixt`` in the current working folder. + +- Fixed (hopefully) handling of Leptonica errors in an environment where a non-standard ``sys.stderr`` is present. + +- Improved help text for ``--verbose``. + v8.2.2 ------ diff --git a/misc/completion/ocrmypdf.bash b/misc/completion/ocrmypdf.bash new file mode 100644 index 00000000..50d4a25e --- /dev/null +++ b/misc/completion/ocrmypdf.bash @@ -0,0 +1,87 @@ +# ocrmypdf completion -*- shell-script -*- + +_ocrmypdf() +{ + local cur prev cword words split + _init_completion -s || return + + case $prev in + --version|-h|--help) + return + ;; + --user-words|--user-patterns|--tesseract-config) + _filedir + return + ;; + --output-type) + COMPREPLY=( $( compgen -W 'pdfa pdf pdfa-1 pdfa-2 pdfa-3' -- \ + "$cur" ) ) + return + ;; + --pdf-renderer) + COMPREPLY=( $( compgen -W 'auto hocr sandwich' -- "$cur" ) ) + return + ;; + --pdfa-image-compression) + COMPREPLY=( $( compgen -W 'auto jpeg lossless' -- "$cur" ) ) + return + ;; + -O|--optimize|--tesseract-oem) + COMPREPLY=( $( compgen -W '{0..3}' -- "$cur" ) ) + return + ;; + --jpeg-quality|--png-quality) + COMPREPLY=( $( compgen -W '{0..100}' -- "$cur" ) ) + return + ;; + -l|--language) + COMPREPLY=$( command tesseract --list-langs 2>/dev/null ) + COMPREPLY=( $( compgen -W '${COMPREPLY[@]##*:}' -- "$cur" ) ) + return + ;; + --image-dpi|--oversample|--skip-big|--max-image-mpixels|\ + --tesseract-timeout|--rotate-pages-threshold) + COMPREPLY=( $( compgen -P "$cur" -W '{0..9}' ) ) + return + ;; + -j|--jobs) + COMPREPLY=( $( compgen -W '{1..'$( _ncpus )'}' -- "$cur" ) ) + return + ;; + -v|--verbose) + COMPREPLY=( $( compgen -W '{1..9}' -- "$cur" ) ) # max level ? + return + ;; + --tesseract-pagesegmode) + COMPREPLY=( $( compgen -W '{1..13}' -- "$cur" ) ) + return + ;; + --sidecar|--title|--author|--subject|--keywords|--unpaper-args) + # argument required but no completions available + return + ;; + esac + + $split && return + + if [[ $cur == -* ]]; then + COMPREPLY=( $( compgen -W '--language --image-dpi --output-type + --sidecar --version --jobs --quiet --verbose --title --author + --subject --keywords --rotate-pages --remove-background --deskew + --clean --clean-final --unpaper-args --oversample --remove-vectors + --mask-barcodes --threshold --force-ocr --skip-text --redo-ocr + --skip-big --jpeg-quality --png-quality --jbig2-lossy + --max-image-mpixels --tesseract-config --tesseract-pagesegmode + --help --tesseract-oem --pdf-renderer --tesseract-timeout + --rotate-pages-threshold --pdfa-image-compression --user-words + --user-patterns --keep-temporary-files --flowchart --output-type' \ + -- "$cur" ) ) + return + else + _filedir + return + fi +} && +complete -F _ocrmypdf ocrmypdf + +# ex: filetype=sh diff --git a/misc/completions.fish b/misc/completion/ocrmypdf.fish similarity index 100% rename from misc/completions.fish rename to misc/completion/ocrmypdf.fish diff --git a/requirements/main.txt b/requirements/main.txt index e2429787..3d2e9ad1 100644 --- a/requirements/main.txt +++ b/requirements/main.txt @@ -5,7 +5,7 @@ chardet == 3.0.4 cffi == 1.12.2 img2pdf == 0.3.3 pdfminer.six == 20181108 -pikepdf == 1.1.0 +pikepdf == 1.3.0 Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin" pycparser == 2.19 python-xmp-toolkit == 2.0.1 diff --git a/requirements/test.txt b/requirements/test.txt index 7071ff9e..ad5ec593 100644 --- a/requirements/test.txt +++ b/requirements/test.txt @@ -1,6 +1,6 @@ -pytest == 4.4.0 +pytest >= 4.4.1, < 5 pytest-helpers-namespace >= 2019.1.8 -pytest-xdist +pytest-xdist == 1.28.0 pytest-cov >= 2.6.1 python-xmp-toolkit # requires apt-get install libexempi3 # or brew install exempi diff --git a/setup.py b/setup.py index fcccc173..ca4afb83 100644 --- a/setup.py +++ b/setup.py @@ -99,7 +99,7 @@ setup( 'cffi >= 1.9.1', # must be a setup and install requirement 'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely 'pdfminer.six == 20181108 ; sys_platform != "darwin"', - 'pikepdf >= 1.1.0, < 2', + 'pikepdf >= 1.3.0, < 2', 'Pillow >= 4.0.0, != 5.1.0 ; sys_platform == "darwin"', # Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3 # block 5.1.0, broken wheels diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 31b5abd1..624f69dd 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -23,10 +23,6 @@ import sys from . import PROGRAM_NAME, VERSION from ._sync import run_pipeline -# Hack to help debugger context find /usr/local/bin -if 'IDE_PROJECT_ROOTS' in os.environ: - os.environ['PATH'] = '/usr/local/bin:' + os.environ['PATH'] - # ------------- # Parser @@ -182,17 +178,9 @@ jobcontrol.add_argument( type=int, default=0, nargs='?', - metavar='LEVEL', - choices=range(0, 4), - help=( - "Print more verbose messages for each additional verbose level. Use " - "`-v 2` typically for much more detailed logging. Higher numbers " - "are probably only useful in debugging. " - "0 - Only errors (default); " - "1 - Error and warnings; " - "2 - Info, errors and warnings; " - "3 - All messages including debug messages" - ), + help="Print more verbose messages for each additional verbose level. Use " + "`-v 1` typically for much more detailed logging. Higher numbers " + "are probably only useful in debugging.", ) metadata = parser.add_argument_group( diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index 4ee576d2..9a0a1eb2 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -59,10 +59,14 @@ class PageContext: self.options = pdf_context.options self.pageno = pageno self.pageinfo = pdf_context.pdfinfo[pageno] - self.log = get_logger(pdf_context.options, '%s Page %d: ' % (pdf_context.name, pageno + 1)) + self.log = get_logger( + pdf_context.options, '%s Page %d: ' % (pdf_context.name, pageno + 1) + ) def get_path(self, name): - return os.path.join(self.pdf_context.work_folder, "%06d_%s" % (self.pageno + 1, name)) + return os.path.join( + self.pdf_context.work_folder, "%06d_%s" % (self.pageno + 1, name) + ) def cleanup_working_files(work_folder, options): @@ -97,21 +101,21 @@ class Logger: def debug(self, *args, **kwargs): if self.level <= DEBUG: - print('DEBUG', self.prefix, end='') - print(*args, **kwargs) + print('DEBUG', self.prefix, end='', file=sys.stderr) + print(*args, file=sys.stderr, **kwargs) def info(self, *args, **kwargs): if self.level <= INFO: - print('INFO', self.prefix, end='') - print(*args, **kwargs) + print('INFO', self.prefix, end='', file=sys.stderr) + print(*args, file=sys.stderr, **kwargs) def warning(self, *args, **kwargs): self.warn(*args, **kwargs) def warn(self, *args, **kwargs): if self.level <= WARN: - print('WARN', self.prefix, end='') - print(*args, **kwargs) + print('WARN', self.prefix, end='', file=sys.stderr) + print(*args, file=sys.stderr, **kwargs) def error(self, *args, **kwargs): if self.level <= ERROR: diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index aea5642c..71d5870e 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -36,9 +36,7 @@ from .exceptions import ( PriorOcrFoundError, ) from .exec import ghostscript, tesseract -from .helpers import ( - re_symlink -) +from .helpers import re_symlink from .hocrtransform import HocrTransform from .optimize import optimize from .pdfa import generate_pdfa_ps @@ -101,10 +99,7 @@ def triage_image_file(input_file, output_file, options, log): ) with open(output_file, 'wb') as outf: img2pdf.convert( - input_file, - layout_fun=layout_fun, - with_pdfrw=False, - outputstream=outf + input_file, layout_fun=layout_fun, with_pdfrw=False, outputstream=outf ) log.info("Successfully converted to PDF, processing...") except img2pdf.ImageOpenError as e: @@ -149,9 +144,7 @@ def triage(input_file, output_file, options, log): def get_pdfinfo(input_file, detailed_page_analysis=False): try: - return PdfInfo( - input_file, detailed_page_analysis=detailed_page_analysis - ) + return PdfInfo(input_file, detailed_page_analysis=detailed_page_analysis) except pikepdf.PasswordError: raise EncryptedPdfError() except pikepdf.PdfError: @@ -250,7 +243,9 @@ def is_ocr_required(page_context): if pageinfo.has_text: if not options.force_ocr and not (options.skip_text or options.redo_ocr): - log.error("page already has text! - aborting (use --force-ocr to force OCR)") + log.error( + "page already has text! - aborting (use --force-ocr to force OCR)" + ) raise PriorOcrFoundError() elif options.force_ocr: log.info("page already has text! - rasterizing text and running OCR anyway") @@ -259,7 +254,7 @@ def is_ocr_required(page_context): if pageinfo.has_corrupt_text: log.warn( "some text on this page cannot be mapped to characters: " - "consider using --force-ocr instead", + "consider using --force-ocr instead" ) else: log.info("redoing OCR") @@ -457,9 +452,16 @@ def preprocess_deskew(input_file, page_context): def preprocess_clean(input_file, page_context): from .exec import unpaper + output_file = page_context.get_path('pp_clean.png') dpi = get_page_square_dpi(page_context.pageinfo, page_context.options) - unpaper.clean(input_file, output_file, dpi, page_context.log, page_context.options.unpaper_args) + unpaper.clean( + input_file, + output_file, + dpi, + page_context.log, + page_context.options.unpaper_args, + ) return output_file @@ -479,7 +481,7 @@ def create_ocr_image(image, page_context): draw = ImageDraw.ImageDraw(im) xres, yres = im.info['dpi'] - page_context.log.info('resolution %r %r' % (xres, yres)) + page_context.log.debug('resolution %r %r' % (xres, yres)) if not options.force_ocr: # Do not mask text areas when forcing OCR, because we need to OCR @@ -488,7 +490,9 @@ def create_ocr_image(image, page_context): if options.redo_ocr: mask = True # Mask visible text, but not invisible text - for textarea in page_context.pageinfo.get_textareas(visible=mask, corrupt=None): + for textarea in page_context.pageinfo.get_textareas( + visible=mask, corrupt=None + ): # Calculate resolution based on the image size and page dimensions # without regard whatever resolution is in pageinfo (may differ or # be None) @@ -501,7 +505,7 @@ def create_ocr_image(image, page_context): im.height - bbox[1] * yscale, ] pixcoords = [int(round(c)) for c in pixcoords] - print('blanking %r', pixcoords) + page_context.log.debug('blanking %r', pixcoords) draw.rectangle(pixcoords, fill=white) # draw.rectangle(pixcoords, outline=pink) @@ -677,16 +681,15 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context): # NULs in DocumentInfo seem to be common since older Acrobats included them. # pikepdf can deal with this, but we make the world a better place by # stamping them out as soon as possible. - pdf_file = pikepdf.open(input_pdf) - if pdf_file.docinfo: - modified = False - for k, v in pdf_file.docinfo.items(): - if b'\x00' in bytes(v): - pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'') - modified = True - if modified: - pdf_file.save(input_pdf) - del pdf_file + with pikepdf.open(input_pdf) as pdf_file: + if pdf_file.docinfo: + modified = False + for k, v in pdf_file.docinfo.items(): + if b'\x00' in bytes(v): + pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'') + modified = True + if modified: + pdf_file.save(input_pdf) ghostscript.generate_pdfa( pdf_version=input_pdfinfo.min_version, @@ -740,6 +743,8 @@ def metadata_fixup(working_file, context): compress_streams=True, object_stream_mode=pikepdf.ObjectStreamMode.generate, ) + original.close() + pdf.close() return output_file diff --git a/src/ocrmypdf/_weave.py b/src/ocrmypdf/_weave.py index c6a00bd2..397b89cd 100644 --- a/src/ocrmypdf/_weave.py +++ b/src/ocrmypdf/_weave.py @@ -15,12 +15,14 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +from contextlib import suppress +from itertools import groupby from pathlib import Path import os import pikepdf from .exec import tesseract -MAX_OPEN_PAGE_PDFS = int(os.environ.get('_OCRMYPDF_MAX_OPEN_PAGE_PDFS', 100)) +MAX_REPLACE_PAGES = int(os.environ.get('_OCRMYPDF_MAX_REPLACE_PAGES', 100)) def _update_page_resources(*, page, font, font_key, procset): @@ -107,7 +109,7 @@ def _weave_layers_graft( stream = bytearray(pdf_text_contents) pattern = b'/Im1 Do' idx = stream.find(pattern) - stream[idx:(idx + len(pattern))] = b' ' * len(pattern) + stream[idx : (idx + len(pattern))] = b' ' * len(pattern) pdf_text_contents = bytes(stream) base_page = pdf_base.pages.p(page_num) @@ -156,6 +158,7 @@ def _weave_layers_graft( _update_page_resources( page=base_page, font=font, font_key=font_key, procset=procset ) + pdf_text.close() def _find_font(text, pdf_base): @@ -164,113 +167,23 @@ def _find_font(text, pdf_base): font, font_key = None, None possible_font_names = ('/f-0-0', '/F1') try: - pdf_text = pikepdf.open(text) - pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {}) - except Exception: + with pikepdf.open(text) as pdf_text: + try: + pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {}) + except (AttributeError, IndexError, KeyError): + return None, None + for f in possible_font_names: + pdf_text_font = pdf_text_fonts.get(f, None) + if pdf_text_font is not None: + font_key = f + break + if pdf_text_font: + font = pdf_base.copy_foreign(pdf_text_font) + return font, font_key + except (FileNotFoundError, pikepdf.PdfError): + # PdfError occurs if a 0-length file is written e.g. due to OCR timeout return None, None - for f in possible_font_names: - pdf_text_font = pdf_text_fonts.get(f, None) - if pdf_text_font is not None: - font_key = f - break - if pdf_text_font: - font = pdf_base.copy_foreign(pdf_text_font) - return font, font_key - - -def _traverse_toc(pdf_base, visitor_fn, log): - """ - Walk the table of contents, calling visitor_fn() at each node - - The /Outlines data structure is a messy data structure, but rather than - navigating hierarchically we just track unique nodes. Enqueue nodes when - we find them, and never visit them again. set() is awesome. We look for - the two types of object in the table of contents that can be page bookmarks - and update the page entry. - - """ - - visited = set() - queue = set() - link_keys = ('/Parent', '/First', '/Last', '/Prev', '/Next') - - if '/Outlines' not in pdf_base.root: - return - - queue.add(pdf_base.root.Outlines.objgen) - while queue: - objgen = queue.pop() - visited.add(objgen) - node = pdf_base.get_object(objgen) - log.debug('fix toc: exploring outline entries at %r', objgen) - - # Enumerate other nodes we could visit from here - for key in link_keys: - if key not in node: - continue - item = node[key] - if not item.is_indirect: - # Direct references are not allowed here, but it's not clear - # what we should do if we find any. Removing them is an option: - # node[key] = pdf_base.make_indirect(None) - continue - objgen = item.objgen - if objgen not in visited: - queue.add(objgen) - - if visitor_fn: - visitor_fn(pdf_base, node, log) - - -def _fix_toc(pdf_base, pageref_remap, log): - """Repair the table of contents - - Whenever we replace a page wholesale, it gets assigned a new objgen number - and other references to it within the PDF become invalid, most notably in - the table of contents (/Outlines in PDF-speak). In weave_layers we collect - pageref_remap, a mapping that describes the new objgen number given an old - one. (objgen is a tuple, and the gen is almost always zero.) - - It may ultimately be better to find a way to rebuild a page in place. - - """ - - if not pageref_remap: - return - - def remap_dest(dest_node): - """ - Inner helper function: change the objgen for any page from the old we - invalidated to its new one. - """ - try: - pageref = dest_node[0] - if pageref['/Type'] == '/Page' and pageref.objgen in pageref_remap: - new_objgen = pageref_remap[pageref.objgen] - dest_node[0] = pdf_base.get_object(new_objgen) - except (IndexError, TypeError) as e: - log.warning("This file may contain invalid table of contents entries") - log.debug(e) - - def visit_remap_dest(pdf_base, node, log): - """ - Visitor function to fix ToC entries - - Test for the two types of references to pages that can occur in ToCs. - Both types have the same final format (an indirect reference to the - target page). - """ - if '/Dest' in node: - # /Dest reference to another page (old method) - remap_dest(node['/Dest']) - elif '/A' in node: - # /A (action) command set to "GoTo" (newer method) - if '/S' in node['/A'] and node['/A']['/S'] == '/GoTo': - remap_dest(node['/A']['/D']) - - _traverse_toc(pdf_base, visit_remap_dest, log) - def weave_layers(layers, context): """Apply text layer and/or image layer changes to baseline file @@ -302,51 +215,40 @@ def weave_layers(layers, context): path_base = Path(context.origin).resolve() pdf_base = pikepdf.open(path_base) - keep_open = [] font, font_key, procset = None, None, None - pdfinfo = context.pdfinfo - interim_output_file = context.get_path('weave_layers_interim.pdf') - output_file = context.get_path('weave_layers.pdf') - pagerefs = {} - # Walk the table of contents first, to trigger pikepdf/qpdf to resolve all - # page references in the table of contents. Some PDF generators put invalid - # references in the ToC, so we want to resolve them to null before we - # create any references, or the ToC will be corrupted - _traverse_toc(pdf_base, None, log) + pdfinfo = context.pdfinfo + output_file = context.get_path('weave_layers.pdf') procset = pdf_base.make_indirect( pikepdf.Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]') ) + emplacements = 1 + interim_count = 0 + # Iterate rest for (pageno, image, text, sidecar, autorotate_correction) in layers: if text and not font: font, font_key = _find_font(text, pdf_base) - replacing = False + emplaced_page = False content_rotation = pdfinfo[pageno].rotation - path_image = Path(image).resolve() if image else None if path_image is not None and path_image != path_base: - # We are replacing the old page with a rasterized PDF of the new - # page - log.debug("Replace") - old_objgen = pdf_base.pages[pageno].objgen + # We are updating the old page with a rasterized PDF of the new + # page (without changing objgen, to preserve references) + log.debug("Emplacement update") + with pikepdf.open(image) as pdf_image: + emplacements += 1 + foreign_image_page = pdf_image.pages[0] + pdf_base.pages.append(foreign_image_page) + local_image_page = pdf_base.pages[-1] + pdf_base.pages[pageno].emplace(local_image_page) + del pdf_base.pages[-1] + emplaced_page = True - pdf_image = pikepdf.open(image) - keep_open.append(pdf_image) - image_page = pdf_image.pages[0] - pdf_base.pages[pageno] = image_page - - # We're adding a new page, which will get a new objgen number pair, - # so we need to update any references to it. qpdf did not like - # my attempt to update the old object in place, but that is an - # option to consider - pagerefs[old_objgen] = pdf_base.pages[pageno].objgen - replacing = True - - if replacing: + if emplaced_page: content_rotation = autorotate_correction text_rotation = autorotate_correction text_misaligned = (text_rotation - content_rotation) % 360 @@ -371,28 +273,38 @@ def weave_layers(layers, context): ) # Correct the rotation if applicable - pdf_base.pages[pageno].Rotate = ( - content_rotation - autorotate_correction - ) % 360 + pdf_base.pages[pageno].Rotate = (content_rotation - autorotate_correction) % 360 - if len(keep_open) > MAX_OPEN_PAGE_PDFS: - # qpdf limitations require us to keep files open when we intend - # to copy content from them before saving. However, we want to keep - # a lid on file handles and memory usage, so for big files we're - # going to stop and save periodically. Attach the font to page 1 - # even if page 1 doesn't use it, so we have a way to get it back. + if emplacements % MAX_REPLACE_PAGES == 0: + # Periodically save and reload the Pdf object. This will keep a + # lid on our memory usage for very large files. Attach the font to + # page 1 even if page 1 doesn't use it, so we have a way to get it + # back. + # TODO refactor this to outside the loop page0 = pdf_base.pages[0] _update_page_resources( page=page0, font=font, font_key=font_key, procset=procset ) - pdf_base.save(interim_output_file) - del pdf_base - keep_open = [] - pdf_base = pikepdf.open(interim_output_file) + # We cannot read and write the same file, that will corrupt it + # but we don't to keep more copies than we need to. Delete intermediates. + # {interim_count} is the opened file we were updateing + # {interim_count - 1} can be deleted + # {interim_count + 1} is the new file will produce and open + old_file = output_file + f'_working{interim_count - 1}.pdf' + if not context.options.keep_temporary_files: + with suppress(FileNotFoundError): + os.unlink(old_file) + + next_file = output_file + f'_working{interim_count + 1}.pdf' + pdf_base.save(next_file) + pdf_base.close() + + pdf_base = pikepdf.open(next_file) procset = pdf_base.pages[0].Resources.ProcSet - font, font_key = None, None # Reacquire this information + font, font_key = None, None # Ensure we reacquire this information + interim_count += 1 - _fix_toc(pdf_base, pagerefs, log) pdf_base.save(output_file) + pdf_base.close() return output_file diff --git a/src/ocrmypdf/exec/ghostscript.py b/src/ocrmypdf/exec/ghostscript.py index af91960b..70497092 100644 --- a/src/ocrmypdf/exec/ghostscript.py +++ b/src/ocrmypdf/exec/ghostscript.py @@ -129,8 +129,7 @@ def rasterize_pdf( :param filter_vector: if True, remove vector graphics objects :return: """ - res = xres, yres - int_res = round(xres), round(yres) + res = round(xres, 6), round(yres, 6) if not page_dpi: page_dpi = res @@ -145,7 +144,7 @@ def rasterize_pdf( f'-sDEVICE={raster_device}', f'-dFirstPage={pageno}', f'-dLastPage={pageno}', - f'-r{str(int_res[0])}x{str(int_res[1])}', + f'-r{res[0]:f}x{res[1]:f}', ] + (['-dFILTERVECTOR'] if filter_vector else []) + [ @@ -168,23 +167,8 @@ def rasterize_pdf( log.error('Ghostscript rasterizing failed') raise SubprocessOutputError() - # Ghostscript only accepts integers for output resolution - # if the resolution happens to be fractional, then the discrepancy - # would change the size of the output page, especially if the DPI - # is quite low. Resize the image to the expected size - tmp.seek(0) with Image.open(tmp) as im: - expected_size = ( - round(im.size[0] / int_res[0] * res[0]), - round(im.size[1] / int_res[1] * res[1]), - ) - if expected_size != im.size or page_dpi != (xres, yres): - log.debug( - f"Ghostscript: resize output image {im.size} -> {expected_size}" - ) - im = im.resize(expected_size) - if rotation is not None: log.debug("Rotating output by %i", rotation) # rotation is a clockwise angle and Image.ROTATE_* is @@ -269,7 +253,6 @@ def generate_pdfa( "-dBATCH", "-dNOPAUSE", "-dCompatibilityLevel=" + str(pdf_version), - "-dNumRenderingThreads=" + str(threads), "-sDEVICE=pdfwrite", "-dAutoRotatePages=/None", "-sColorConversionStrategy=" + strategy, diff --git a/src/ocrmypdf/leptonica.py b/src/ocrmypdf/leptonica.py index 0cc8b643..1cdec312 100644 --- a/src/ocrmypdf/leptonica.py +++ b/src/ocrmypdf/leptonica.py @@ -43,11 +43,6 @@ lept = ffi.dlopen(find_library('lept')) lept.setMsgSeverity(lept.L_SEVERITY_WARNING) -def stderr(*objs): - """Shorthand print to stderr.""" - print("leptonica.py:", *objs, file=sys.stderr) - - class _LeptonicaErrorTrap: """ Context manager to trap errors reported by Leptonica. @@ -66,6 +61,7 @@ class _LeptonicaErrorTrap: def __init__(self): self.tmpfile = None self.copy_of_stderr = -1 + self.no_stderr = False def __enter__(self): from io import UnsupportedOperation @@ -73,33 +69,40 @@ class _LeptonicaErrorTrap: self.tmpfile = TemporaryFile() # Save the old stderr, and redirect stderr to temporary file - sys.stderr.flush() + with suppress(AttributeError): + sys.stderr.flush() try: self.copy_of_stderr = os.dup(sys.stderr.fileno()) os.dup2(self.tmpfile.fileno(), sys.stderr.fileno(), inheritable=False) + except AttributeError: + # We are in some unusual context where our Python process does not + # have a sys.stderr. Leptonica still expects to write to file + # descriptor 2, so we are going to ensure it is redirected. + self.copy_of_stderr = None + self.no_stderr = True + os.dup2(self.tmpfile.fileno(), 2, inheritable=False) except UnsupportedOperation: self.copy_of_stderr = None return def __exit__(self, exc_type, exc_value, traceback): # Restore old stderr - sys.stderr.flush() + with suppress(AttributeError): + sys.stderr.flush() + if self.copy_of_stderr is not None: os.dup2(self.copy_of_stderr, sys.stderr.fileno()) os.close(self.copy_of_stderr) + if self.no_stderr: + os.close(2) - # Get data from tmpfile (in with block to ensure it is closed) - with self.tmpfile as tmpfile: - tmpfile.seek(0) # Cursor will be at end, so move back to beginning - leptonica_output = tmpfile.read().decode(errors='replace') - - assert self.tmpfile.closed - assert not sys.stderr.closed - - # If there are Python errors, let them bubble up + # Get data from tmpfile + self.tmpfile.seek(0) # Cursor will be at end, so move back to beginning + leptonica_output = self.tmpfile.read().decode(errors='replace') + self.tmpfile.close() + # If there are Python errors, record them if exc_type: logger.warning(leptonica_output) - return False # If there are Leptonica errors, wrap them in Python excpetions if 'Error' in leptonica_output: @@ -614,9 +617,10 @@ class Pix(LeptonicaObject): except (LeptonicaError, ValueError, IndexError): return finally: - with suppress(FileNotFoundError): - os.unlink('junkpixt.png') # leptonica may produce this - os.unlink('junkpixt') + leptonica_junk = ('junkpixt.png', 'junkpixt') + for junk in leptonica_junk: + with suppress(FileNotFoundError): + os.unlink(junk) # leptonica may produce this for n, s in enumerate(sarray): decoded = s.decode() diff --git a/src/ocrmypdf/pdfa.py b/src/ocrmypdf/pdfa.py index 9bc3b729..684254e7 100644 --- a/src/ocrmypdf/pdfa.py +++ b/src/ocrmypdf/pdfa.py @@ -131,19 +131,19 @@ def file_claims_pdfa(filename): do full PDF/A validation. """ - pdf = pikepdf.open(filename) - pdfmeta = pdf.open_metadata() - if not pdfmeta.pdfa_status: - return { - 'pass': False, - 'output': 'pdf', - 'conformance': 'No PDF/A metadata in XMP', - } - valid_part_conforms = {'1A', '1B', '2A', '2B', '2U', '3A', '3B', '3U'} - conformance = f'PDF/A-{pdfmeta.pdfa_status}' - pdfa_dict = {} - if pdfmeta.pdfa_status in valid_part_conforms: - pdfa_dict['pass'] = True - pdfa_dict['output'] = 'pdfa' - pdfa_dict['conformance'] = conformance + with pikepdf.open(filename) as pdf: + pdfmeta = pdf.open_metadata() + if not pdfmeta.pdfa_status: + return { + 'pass': False, + 'output': 'pdf', + 'conformance': 'No PDF/A metadata in XMP', + } + valid_part_conforms = {'1A', '1B', '2A', '2B', '2U', '3A', '3B', '3U'} + conformance = f'PDF/A-{pdfmeta.pdfa_status}' + pdfa_dict = {} + if pdfmeta.pdfa_status in valid_part_conforms: + pdfa_dict['pass'] = True + pdfa_dict['output'] = 'pdfa' + pdfa_dict['conformance'] = conformance return pdfa_dict diff --git a/src/ocrmypdf/pdfinfo/__init__.py b/src/ocrmypdf/pdfinfo/__init__.py index 1dd94e60..aaad8ebe 100644 --- a/src/ocrmypdf/pdfinfo/__init__.py +++ b/src/ocrmypdf/pdfinfo/__init__.py @@ -618,8 +618,9 @@ def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None): if not log: log = Mock() - pdf = pikepdf.open(infile) + pdf = pikepdf.open(infile) # Do not close in this function if pdf.is_encrypted: + pdf.close() raise EncryptedPdfError() # Triggered by encryption with empty passwd if detailed_analysis: pages_xml = None @@ -755,7 +756,13 @@ class PdfInfo: infile, detailed_page_analysis, log=log ) self._needs_rendering = pdf.root.get('/NeedsRendering', False) - self._has_acroform = '/AcroForm' in pdf.root + self._has_acroform = False + if '/AcroForm' in pdf.root: + if len(pdf.root.AcroForm.get('/Fields', [])) > 0: + self._has_acroform = True + elif '/XFA' in pdf.root.AcroForm: + self._has_acroform = True + pdf.close() @property def pages(self): diff --git a/tests/resources/link.pdf b/tests/resources/link.pdf new file mode 100644 index 00000000..2633d975 Binary files /dev/null and b/tests/resources/link.pdf differ diff --git a/tests/test_ghostscript.py b/tests/test_ghostscript.py new file mode 100644 index 00000000..8756e8a8 --- /dev/null +++ b/tests/test_ghostscript.py @@ -0,0 +1,81 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import logging +from decimal import Decimal + +import pikepdf +import pytest +from PIL import Image + +from ocrmypdf.exec.ghostscript import rasterize_pdf + + +@pytest.fixture +def linn(resources): + path = resources / 'linn.pdf' + return path, pikepdf.open(path) + + +def test_rasterize_size(linn, outdir, caplog): + path, pdf = linn + page_size_pts = (pdf.pages[0].MediaBox[2], pdf.pages[0].MediaBox[3]) + assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0 + page_size = (page_size_pts[0] / Decimal(72), page_size_pts[1] / Decimal(72)) + target_size = Decimal('200.0'), Decimal('150.0') + target_dpi = 42.0, 4242.0 + + log = logging.getLogger() + rasterize_pdf( + path, + outdir / 'out.png', + target_size[0] / page_size[0], + target_size[1] / page_size[1], + raster_device='pngmono', + log=log, + page_dpi=target_dpi, + ) + + with Image.open(outdir / 'out.png') as im: + assert im.size == target_size + assert im.info['dpi'] == target_dpi + + +def test_rasterize_rotated(linn, outdir, caplog): + path, pdf = linn + page_size_pts = (pdf.pages[0].MediaBox[2], pdf.pages[0].MediaBox[3]) + assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0 + page_size = (page_size_pts[0] / Decimal(72), page_size_pts[1] / Decimal(72)) + target_size = Decimal('200.0'), Decimal('150.0') + target_dpi = 42.0, 4242.0 + + log = logging.getLogger() + caplog.set_level(logging.DEBUG) + rasterize_pdf( + path, + outdir / 'out.png', + target_size[0] / page_size[0], + target_size[1] / page_size[1], + raster_device='pngmono', + log=log, + page_dpi=target_dpi, + rotation=90, + ) + + with Image.open(outdir / 'out.png') as im: + assert im.size == (target_size[1], target_size[0]) + assert im.info['dpi'] == (target_dpi[1], target_dpi[0]) diff --git a/tests/test_lept.py b/tests/test_lept.py index 142fc702..5fe68109 100644 --- a/tests/test_lept.py +++ b/tests/test_lept.py @@ -17,7 +17,9 @@ from os import fspath +import os from pickle import dumps, loads +from unittest.mock import patch import pytest from PIL import Image, ImageChops @@ -85,3 +87,18 @@ def test_leptonica_compile(tmpdir): # existing compiled library. Also compile in API mode so that we test # the interfaces, even though we use it ABI mode. ffibuilder.compile(tmpdir=fspath(tmpdir), target=fspath(tmpdir / 'lepttest.*')) + + +def test_with_stderr(capsys): + # pytest redirects stderr too; we must disable this for the test to be valid + with capsys.disabled(): + with pytest.raises(FileNotFoundError): + lept.Pix.open("does_not_exist1") + + +def test_without_stderr(capsys): + # pytest redirects stderr too; we must disable this for the test to be valid + with capsys.disabled(): + with patch('sys.stderr', new=None): + with pytest.raises(FileNotFoundError): + lept.Pix.open("does_not_exist2") diff --git a/tests/test_metadata.py b/tests/test_metadata.py index 3ec07fa5..5b7e235e 100644 --- a/tests/test_metadata.py +++ b/tests/test_metadata.py @@ -18,6 +18,7 @@ import datetime from datetime import timezone +import logging import mmap from os import fspath from pathlib import Path @@ -284,7 +285,7 @@ def test_kodak_toc(resources, outpdf, spoof_tesseract_noop): assert isinstance(p.root.Outlines.First, pikepdf.Dictionary) -def test_metadata_fixup_warning(resources, outdir): +def test_metadata_fixup_warning(resources, outdir, caplog): from ocrmypdf.__main__ import parser from ocrmypdf._pipeline import metadata_fixup @@ -295,13 +296,10 @@ def test_metadata_fixup_warning(resources, outdir): copyfile(resources / 'graph.pdf', outdir / 'graph.pdf') context = PDFContext(options, outdir, outdir / 'graph.pdf', None) - context.log = MagicMock() - metadata_fixup( - working_file=outdir / 'graph.pdf', - context=context, - ) - context.log.warn.assert_not_called() - context.log.error.assert_not_called() + context.log = logging.getLogger() + metadata_fixup(working_file=outdir / 'graph.pdf', context=context) + for record in caplog.records: + assert record.levelname != 'WARNING' # Now add some metadata that will not be copyable graph = pikepdf.open(outdir / 'graph.pdf') @@ -310,12 +308,9 @@ def test_metadata_fixup_warning(resources, outdir): graph.save(outdir / 'graph_mod.pdf') context = PDFContext(options, outdir, outdir / 'graph_mod.pdf', None) - context.log = MagicMock() - metadata_fixup( - working_file=outdir / 'graph.pdf', - context=context, - ) - context.log.warn.assert_called_once() + context.log = logging.getLogger() + metadata_fixup(working_file=outdir / 'graph.pdf', context=context) + assert any(record.levelname == 'WARNING' for record in caplog.records) def test_prevent_gs_invalid_xml(resources, outdir): @@ -333,7 +328,9 @@ def test_prevent_gs_invalid_xml(resources, outdir): pdfinfo = PdfInfo(resources / 'enron1.pdf') context = PDFContext(options, outdir, resources / 'enron1.pdf', pdfinfo) - convert_to_pdfa(str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context) + convert_to_pdfa( + str(outdir / 'layers.rendered.pdf'), str(outdir / 'pdfa.ps'), context + ) with open(outdir / 'pdfa.pdf', 'rb') as f: with mmap.mmap( diff --git a/tests/test_unpaper.py b/tests/test_unpaper.py index 213ef7d0..5b55e819 100644 --- a/tests/test_unpaper.py +++ b/tests/test_unpaper.py @@ -15,8 +15,11 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . +import argparse +import logging from os import fspath -from unittest.mock import MagicMock, patch +from pathlib import Path +from unittest.mock import patch import pytest @@ -56,7 +59,7 @@ def test_no_unpaper(resources, no_outpdf): with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version: mock_unpaper_version.side_effect = FileNotFoundError("unpaper") with pytest.raises(SystemExit): - check_options(options, log=MagicMock()) + check_options(options, log=logging.getLogger()) def test_old_unpaper(spoof_unpaper_oldversion, resources, no_outpdf): diff --git a/tests/test_weave.py b/tests/test_weave.py index 34452215..06fa2a0a 100644 --- a/tests/test_weave.py +++ b/tests/test_weave.py @@ -15,35 +15,15 @@ # You should have received a copy of the GNU General Public License # along with OCRmyPDF. If not, see . -from unittest.mock import MagicMock -import logging import os import pytest import pikepdf -from ocrmypdf._weave import _fix_toc, _update_page_resources check_ocrmypdf = pytest.helpers.check_ocrmypdf -def test_invalid_toc(resources, outdir, caplog): - pdf = pikepdf.open(resources / 'toc.pdf') - - # Corrupt a TOC entry - pdf.Root.Outlines.Last.Dest = pikepdf.Array([None, 0.0, 0.1, 0.2]) - pdf.save(outdir / 'test.pdf') - - pdf = pikepdf.open(outdir / 'test.pdf') - remap = {} - remap[pdf.pages[0].objgen] = pdf.pages[0].objgen # Dummy remap - - # Confirm we complain about the TOC and don't throw an exception - log = logging.getLogger() - _fix_toc(pdf, remap, log) - assert 'invalid table of contents entries' in caplog.text - - def test_no_glyphless_weave(resources, outdir): pdf = pikepdf.open(resources / 'francais.pdf') pdf_aspect = pikepdf.open(resources / 'aspect.pdf') @@ -53,7 +33,7 @@ def test_no_glyphless_weave(resources, outdir): pdf.save(outdir / 'test.pdf') env = os.environ.copy() - env['_OCRMYPDF_MAX_OPEN_PAGE_PDFS'] = '2' + env['_OCRMYPDF_MAX_REPLACE_PAGES'] = '2' check_ocrmypdf( outdir / 'test.pdf', outdir / 'out.pdf', @@ -62,3 +42,21 @@ def test_no_glyphless_weave(resources, outdir): '0', env=env, ) + + +@pytest.helpers.needs_pdfminer +def test_links(resources, outpdf): + check_ocrmypdf( + resources / 'link.pdf', + outpdf, + '--redo-ocr', + '--oversample', + '200', + '--output-type', + 'pdf', + ) + pdf = pikepdf.open(outpdf) + p1 = pdf.pages[0] + p2 = pdf.pages[1] + assert p1.Annots[0].A.D[0].objgen == p2.objgen + assert p2.Annots[0].A.D[0].objgen == p1.objgen