From 2ab94b3151329ba0ad209df1565a59baf43721f9 Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Sat, 28 Apr 2018 01:57:41 -0700 Subject: [PATCH] unsplit: it's alive First successful file output. --- src/ocrmypdf/pipeline.py | 182 +++++++++++++++++++++------------------ 1 file changed, 96 insertions(+), 86 deletions(-) diff --git a/src/ocrmypdf/pipeline.py b/src/ocrmypdf/pipeline.py index efbcacc8..620345e8 100644 --- a/src/ocrmypdf/pipeline.py +++ b/src/ocrmypdf/pipeline.py @@ -366,7 +366,7 @@ def marker_pages( # Ruffus needs to see a file for any task it generates, so make very # file a symlink back to the source. for n in range(npages): - page = Path(work_folder) / '{0:06d}.page.pdf'.format(n + 1) + page = Path(work_folder) / '{0:06d}.marker.pdf'.format(n + 1) page.symlink_to(input_file) @@ -388,14 +388,14 @@ def ocr_or_skip( options = context.get_options() work_folder = context.get_work_folder() pdfinfo = context.get_pdfinfo() - + for input_file in input_files: pageno = page_number(input_file) - 1 pageinfo = pdfinfo[pageno] alt_suffix = \ '.ocr.page.pdf' if is_ocr_required(pageinfo, log, options) \ else '.skip.page.pdf' - + re_symlink( input_file, os.path.join( @@ -770,12 +770,19 @@ def combine_layers( pdf_base = None font = None - for prefix, layers in groupby(sorted(flatten_groups(infiles)), - key=page_number): + def input_sorter(key): + try: + return page_number(key) + except ValueError: + return -1 + flat_inputs = sorted(flatten_groups(infiles), key=input_sorter) + + for prefix, layers in groupby(flat_inputs, key=input_sorter): layers = list(layers) + log.info(prefix) log.info(layers) - if not path_base and layers[0].endswith('000000.pdf'): - path_base = Path(layers[0]) + if not path_base: + path_base = Path(layers[0]).resolve() pdf_base = pikepdf.open(path_base) continue @@ -797,15 +804,24 @@ def combine_layers( base_page.page_contents_add(new_text_layer) if not font: - font = pdf_base.make_indirect( - pdf_text.pages[0].Resources.Font['/f0-0-0']) - fonts = base_page.Resources['/Font'] - fonts['/f0-0-0'] = font - procset = base_page.ProcSet - if not '/Text' in procset: - list_procset = procset.as_list() - list_procset.append(pikepdf.name('/Text')) - base_page.ProcSet = pikepdf.Array(list_procset) + pdf_text_font = pdf_text.pages[0].Resources.Font['/f-0-0'] + font = pdf_base._copy_foreign(pdf_text_font) + + # Update page fonts with reference to Glyphless + try: + base_fonts = base_page['/Resources']['/Font'] + except KeyError: + base_fonts = pikepdf.Dictionary({}) + if not '/f-0-0' in base_fonts: + base_fonts['/f-0-0'] = font + base_page['/Resources']['/Font'] = base_fonts + + # Technical we should add /Text to /ProcSet, but /ProcSet is + # both optional and useless, so just remove it if it exists + try: + del base_page['/ProcSet'] + except KeyError: + pass else: # We are going to replace the old page @@ -814,78 +830,78 @@ def combine_layers( pdf_base.save(output_file) -def _old_combine_layers(): - text = next(ii for ii in flatten_groups(infiles) - if ii.endswith('.text.pdf')) # single page PDF from tesseract - image = next(ii for ii in flatten_groups(infiles) - if ii.endswith('.image-layer.pdf')) # single/multipage +# def _old_combine_layers(): +# text = next(ii for ii in flatten_groups(infiles) +# if ii.endswith('.text.pdf')) # single page PDF from tesseract +# image = next(ii for ii in flatten_groups(infiles) +# if ii.endswith('.image-layer.pdf')) # single/multipage - pdf_text = pypdf.PdfFileReader(text) - pdf_image = pypdf.PdfFileReader(image) +# pdf_text = pypdf.PdfFileReader(text) +# pdf_image = pypdf.PdfFileReader(image) - page_text = pdf_text.getPage(0) +# page_text = pdf_text.getPage(0) - # The text page always will be oriented up by this stage - # but if lossless_reconstruction, pdf_image may have a rotation applied - # We have to eliminate the /Rotate tag (because it applies to the whole - # page) and rotate the image layer to match the text page - # Also, pdf_image may not have its mediabox nailed to (0, 0), so may need - # translation - page_image = pdf_image.getPage(0) - try: - # pypdf DictionaryObject.get() does not resolve indirect objects but - # __getitem__ does - rotation = page_image['/Rotate'] - except KeyError: - rotation = 0 +# # The text page always will be oriented up by this stage +# # but if lossless_reconstruction, pdf_image may have a rotation applied +# # We have to eliminate the /Rotate tag (because it applies to the whole +# # page) and rotate the image layer to match the text page +# # Also, pdf_image may not have its mediabox nailed to (0, 0), so may need +# # translation +# page_image = pdf_image.getPage(0) +# try: +# # pypdf DictionaryObject.get() does not resolve indirect objects but +# # __getitem__ does +# rotation = page_image['/Rotate'] +# except KeyError: +# rotation = 0 - # /Rotate is a clockwise rotation: 90 means page facing "east" - # The negative of this value is the angle that eliminates that rotation - rotation = -rotation % 360 +# # /Rotate is a clockwise rotation: 90 means page facing "east" +# # The negative of this value is the angle that eliminates that rotation +# rotation = -rotation % 360 - x1 = page_image.mediaBox.getLowerLeft_x() - x2 = page_image.mediaBox.getUpperRight_x() - y1 = page_image.mediaBox.getLowerLeft_y() - y2 = page_image.mediaBox.getUpperRight_y() +# x1 = page_image.mediaBox.getLowerLeft_x() +# x2 = page_image.mediaBox.getUpperRight_x() +# y1 = page_image.mediaBox.getLowerLeft_y() +# y2 = page_image.mediaBox.getUpperRight_y() - # Rotation occurs about the page's (0, 0). Most pages will have the media - # box at (0, 0) with all content in the first quadrant but some cropped - # files may have an offset mediabox. We translate the page so that its - # bottom left corner after rotation is pinned to (0, 0) with the image - # in the first quadrant. - if rotation == 0: - tx, ty = -x1, -y1 - elif rotation == 90: - tx, ty = y2, -x1 - elif rotation == 180: - tx, ty = x2, y2 - elif rotation == 270: - tx, ty = -y1, x2 - else: - pass +# # Rotation occurs about the page's (0, 0). Most pages will have the media +# # box at (0, 0) with all content in the first quadrant but some cropped +# # files may have an offset mediabox. We translate the page so that its +# # bottom left corner after rotation is pinned to (0, 0) with the image +# # in the first quadrant. +# if rotation == 0: +# tx, ty = -x1, -y1 +# elif rotation == 90: +# tx, ty = y2, -x1 +# elif rotation == 180: +# tx, ty = x2, y2 +# elif rotation == 270: +# tx, ty = -y1, x2 +# else: +# pass - if rotation != 0: - log.info("{0:4d}: rotating image layer {1} degrees".format( - page_number(image), rotation)) +# if rotation != 0: +# log.info("{0:4d}: rotating image layer {1} degrees".format( +# page_number(image), rotation)) - try: - page_text.mergeRotatedScaledTranslatedPage( - page_image, rotation, 1.0, tx, ty, expand=False) - except (AttributeError, ValueError) as e: - if 'writeToStream' in str(e) or 'invalid literal' in str(e): - raise PdfMergeFailedError() from e +# try: +# page_text.mergeRotatedScaledTranslatedPage( +# page_image, rotation, 1.0, tx, ty, expand=False) +# except (AttributeError, ValueError) as e: +# if 'writeToStream' in str(e) or 'invalid literal' in str(e): +# raise PdfMergeFailedError() from e - pdf_output = pypdf.PdfFileWriter() - pdf_output.addPage(page_text) +# pdf_output = pypdf.PdfFileWriter() +# pdf_output.addPage(page_text) - # If the input was scaled, re-apply the scaling - pageinfo = get_pageinfo(text, context) - if pageinfo.userunit != 1: - page_text[pypdf.generic.NameObject('/UserUnit')] = pageinfo.userunit - pdf_output._header = b'%PDF-1.6' # Hack header to correct version +# # If the input was scaled, re-apply the scaling +# pageinfo = get_pageinfo(text, context) +# if pageinfo.userunit != 1: +# page_text[pypdf.generic.NameObject('/UserUnit')] = pageinfo.userunit +# pdf_output._header = b'%PDF-1.6' # Hack header to correct version - with open(output_file, "wb") as out: - pdf_output.write(out) +# with open(output_file, "wb") as out: +# pdf_output.write(out) def ocr_tesseract_and_render_pdf( @@ -1363,20 +1379,14 @@ def build_pipeline(options, work_folder, log, context): task_ocr_tesseract_textonly_pdf.graphviz(fillcolor='"#ff69b4"') task_ocr_tesseract_textonly_pdf.active_if(options.pdf_renderer == 'sandwich') - task_copy_repaired = main_pipeline.transform( - task_func=shutil.copy, - input=task_repair_and_parse_pdf, - filter=formatter(r'\.repaired\.pdf'), - output=os.path.join(work_folder, '000000.pdf') - ) - task_combine_layers = main_pipeline.collate( task_func=combine_layers, - input=[task_copy_repaired, + input=[task_repair_and_parse_pdf, task_render_hocr_page, task_ocr_tesseract_textonly_pdf, task_select_image_layer], - filter=regex(r".*/(\d{6})(?:\.text\.pdf|\.image-layer\.pdf|\.pdf)"), + filter=regex( + r".*/((?:\d{6}(?:\.text\.pdf|\.image-layer\.pdf))|(?:origin\.repaired\.pdf))"), output=os.path.join(work_folder, r'layers.rendered.pdf'), extras=[log, context]) task_combine_layers.graphviz(fillcolor='"#00cc66"')