diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index 3e69effd..94235d1e 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -250,13 +250,17 @@ ocrsettings = parser.add_argument_group( "Control how OCR is applied") ocrsettings.add_argument( '-f', '--force-ocr', action='store_true', - help="Rasterize any fonts or vector objects on each page, apply OCR, and " + help="Rasterize any text or vector objects on each page, apply OCR, and " "save the rastered output (this rewrites the PDF)") ocrsettings.add_argument( '-s', '--skip-text', action='store_true', help="Skip OCR on any pages that already contain text, but include the " "page in final output; useful for PDFs that contain a mix of " "images, text pages, and/or previously OCRed pages") +ocrsettings.add_argument( + '--redo-ocr', action='store_true', + help="Remove any invisible text, and apply OCR") + ocrsettings.add_argument( '--skip-big', type=numeric(float, 0, 5000), metavar='MPixels', help="Skip OCR on pages larger than the specified amount of megapixels, " diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index e114d6b6..b72a8603 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -243,7 +243,7 @@ def is_ocr_required(pageinfo, log, options): if pageinfo.has_text: msg = "{0:4d}: page already has text! – {1}" - if not options.force_ocr and not options.skip_text: + if not options.force_ocr and not (options.skip_text or options.redo_ocr): log.error(msg.format(page, "aborting (use --force-ocr to force OCR)")) raise PriorOcrFoundError() @@ -251,6 +251,10 @@ def is_ocr_required(pageinfo, log, options): log.info(msg.format(page, "rasterizing text and running OCR anyway")) ocr_required = True + elif options.redo_ocr and pageinfo.only_ocr_text: + log.info(msg.format(page, + "redoing OCR")) + ocr_required = True elif options.skip_text: log.info(msg.format(page, "skipping all processing on this page")) @@ -559,12 +563,13 @@ def select_ocr_image( user.""" image = infiles[0] - if context.get_options().force_ocr: + options = context.get_options() + pageinfo = get_pageinfo(image, context) + + if options.force_ocr or (options.redo_ocr and pageinfo.only_ocr_text): re_symlink(image, output_file, log) return - pageinfo = get_pageinfo(image, context) - with Image.open(image) as im: from PIL import ImageColor from PIL import ImageDraw diff --git a/src/ocrmypdf/_weave.py b/src/ocrmypdf/_weave.py index 7d08ffd6..7d97a6af 100644 --- a/src/ocrmypdf/_weave.py +++ b/src/ocrmypdf/_weave.py @@ -43,8 +43,39 @@ def _update_page_resources(*, page, font, font_key, procset): resources['/ProcSet'] = procset +def _strip_old_text(pdf, page): + stream = [] + in_text_obj = False + + page.page_contents_coalesce() + for operands, operator in pikepdf.parse_content_stream(page, ''): + if not in_text_obj: + if operator == pikepdf.Operator('BT'): + in_text_obj = True + else: + stream.append((operands, operator)) + else: + if operator == pikepdf.Operator('ET'): + in_text_obj = False + + def convert(op): + try: + return op.unparse() + except AttributeError: + return str(op).encode('ascii') + + lines = [] + for operands, operator in stream: + line = b' '.join(convert(op) for op in operands) + b' ' + operator.unparse() + lines.append(line) + + content_stream = b'\n'.join(lines) + page.Contents = pikepdf.Stream(pdf, content_stream) + + def _weave_layers_graft( - *, pdf_base, page_num, text, font, font_key, procset, rotation, log): + *, pdf_base, page_num, text, font, font_key, procset, rotation, + strip_old_text, log): """Insert the text layer from text page 0 on to pdf_base at page_num""" log.debug("Grafting") @@ -109,6 +140,9 @@ def _weave_layers_graft( new_text_layer = pikepdf.Stream(pdf_base, pdf_text_contents) + if strip_old_text: + _strip_old_text(pdf_base, base_page) + base_page.page_contents_add(new_text_layer, prepend=True) _update_page_resources( @@ -336,10 +370,12 @@ def weave_layers( if text and font: # Graft the text layer onto this page, whether new or old + strip_old = (context.get_options().redo_ocr + and pdfinfo[page_num - 1].only_ocr_text) _weave_layers_graft( pdf_base=pdf_base, page_num=page_num, text=text, font=font, font_key=font_key, rotation=text_misaligned, procset=procset, - log=log + strip_old_text=strip_old, log=log ) # Correct the rotation if applicable diff --git a/src/ocrmypdf/pdfinfo.py b/src/ocrmypdf/pdfinfo.py index a2748c23..fd54a729 100644 --- a/src/ocrmypdf/pdfinfo.py +++ b/src/ocrmypdf/pdfinfo.py @@ -107,6 +107,20 @@ ContentsInfo = namedtuple('ContentsInfo', ['xobject_settings', 'inline_images', 'found_text', 'found_vector']) +class VectorInfo: + def __init__(self): + pass + + +class TextInfo: + def __init__(self, invisible, visible): + self.invisible = invisible + self.visible = visible + + def __bool__(self): + return self.invisible or self.visible + + def _normalize_stack(graphobjs): """Convert runs of qQ's in the stack into single graphobjs""" for operands, operator in graphobjs: @@ -143,9 +157,14 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): xobject_settings = [] inline_images = [] found_text, found_vector = False, False - text_operators = set("""Tj " ' TJ""".split()) - vector_operators = set('S s f F f* B B* b b*'.split()) - operator_whitelist = """q Q Do cm TJ Tj " ' BI ID EI S s f F f* B B* b b*""" + found_invisible_text, found_visible_text = False, False + text_mode_ops = set("""BT ET Tr""".split()) + text_showing_ops = set("""Tj " ' TJ""".split()) + vector_ops = set('S s f F f* B B* b b*'.split()) + image_ops = set('BI ID EI q Q Do cm'.split()) + text_render_mode = 0 + operator_whitelist = ' '.join( + text_mode_ops | text_showing_ops | vector_ops | image_ops) for n, graphobj in enumerate(_normalize_stack( pikepdf.parse_content_stream(contentstream, operator_whitelist))): @@ -176,15 +195,25 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): iimage=iimage, shorthand=ctm.shorthand, stack_depth=len(stack)) inline_images.append(inline) - elif operator in text_operators: + elif operator in text_mode_ops: + if operator == 'BT': + text_render_mode = 0 + elif operator == 'Tr': + text_render_mode = operands[0] + elif operator in text_showing_ops: found_text = True - elif operator in vector_operators: + if text_render_mode == 3: + found_invisible_text = True + else: + found_visible_text = True + elif operator in vector_ops: found_vector = True return ContentsInfo( xobject_settings=xobject_settings, inline_images=inline_images, - found_text=found_text, + found_text=TextInfo(invisible=found_invisible_text, + visible=found_visible_text), found_vector=found_vector) @@ -252,11 +281,6 @@ def _get_dpi(ctm_shorthand, image_size): return dpi_w, dpi_h -class VectorInfo: - def __init__(self): - pass - - class ImageInfo: DPI_PREC = Decimal('1.000') @@ -444,11 +468,11 @@ def _find_form_xobject_images(pdf, container, contentsinfo): # but in practice both Form XObjects and multiple drawing of the # same object are both very rare. ctm_shorthand = settings.shorthand - yield from _find_images( + yield from _process_content_streams( pdf=pdf, container=form_xobject, shorthand=ctm_shorthand) -def _find_images(*, pdf, container, shorthand=None): +def _process_content_streams(*, pdf, container, shorthand=None): """Find all individual instances of images drawn in the container Usually the container is a page, but it may also be a Form XObject. @@ -491,6 +515,8 @@ def _find_images(*, pdf, container, shorthand=None): if contentsinfo.found_vector: yield VectorInfo() + if contentsinfo.found_text: + yield contentsinfo.found_text yield from _find_inline_images(contentsinfo) yield from _find_regular_images(container, contentsinfo) yield from _find_form_xobject_images(pdf, container, contentsinfo) @@ -591,15 +617,20 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext): pageinfo['rotate'] = 0 userunit_shorthand = (userunit, 0, 0, userunit, 0, 0) - pageinfo['images'] = [im for im in - _find_images(pdf=pdf, container=page, - shorthand=userunit_shorthand)] + contentsinfo = [ci for ci in + _process_content_streams(pdf=pdf, container=page, + shorthand=userunit_shorthand)] - if any(isinstance(im, VectorInfo) for im in pageinfo['images']): + pageinfo['has_vector'] = False + if any(isinstance(ci, VectorInfo) for ci in contentsinfo): pageinfo['has_vector'] = True - pageinfo['images'] = [im for im in pageinfo['images'] - if not isinstance(im, VectorInfo)] + textinfos = [ti for ti in contentsinfo if isinstance(ti, TextInfo)] + all_invisible = all(ti.invisible for ti in textinfos) and len(textinfos) > 0 + pageinfo['only_ocr_text'] = all_invisible + + pageinfo['images'] = [im for im in contentsinfo + if isinstance(im, ImageInfo)] if pageinfo['images']: xres = Decimal(max(image.xres for image in pageinfo['images'])) yres = Decimal(max(image.yres for image in pageinfo['images'])) @@ -666,6 +697,10 @@ class PageInfo: def has_vector(self): return self._pageinfo['has_vector'] + @property + def only_ocr_text(self): + return self._pageinfo['only_ocr_text'] + @property def width_inches(self): return self._pageinfo['width_inches'] diff --git a/tests/test_pdfinfo.py b/tests/test_pdfinfo.py index c3ee0b6d..574f8832 100644 --- a/tests/test_pdfinfo.py +++ b/tests/test_pdfinfo.py @@ -175,3 +175,13 @@ def test_vector(resources): filename = resources / 'vector.pdf' pdf = pdfinfo.PdfInfo(filename) assert pdf[0].has_vector + assert not pdf[0].only_ocr_text + assert not pdf[0].has_text + + +def test_ocr_detection(resources): + filename = resources / 'graph_ocred.pdf' + pdf = pdfinfo.PdfInfo(filename) + assert not pdf[0].has_vector + assert pdf[0].only_ocr_text + assert pdf[0].has_text