diff --git a/docs/pipeline.svg b/docs/pipeline.svg index a8ac187e..bb461c94 100644 --- a/docs/pipeline.svg +++ b/docs/pipeline.svg @@ -4,550 +4,470 @@ - - + + Pipeline: - + clustertasks - -Pipeline: + +Pipeline: t0 - - - - -ocrmypdf.pipeline.triage + + + + +ocrmypdf.pipeline.triage t1 - - - - -ocrmypdf.pipeline.repair_and_parse_pdf + + + + +ocrmypdf.pipeline.repair_and_parse_pdf t0->t1 - - + + t2 - - -ocrmypdf.pipeline.pre_split_pages + + +ocrmypdf.pipeline.pre_split_pages t1->t2 - - + + t20 - - - - -ocrmypdf.pipeline.generate_postscript_stub + + + + +ocrmypdf.pipeline.generate_postscript_stub t1->t20 - - - - - -t24 - - -ocrmypdf.pipeline.merge_pages_mupdf - - - -t1->t24 - - - - - -t23 - - -ocrmypdf.pipeline.merge_pages_qpdf - - - -t1->t23 - - - - - -t3 - - - - -ocrmypdf.pipeline.split_page - - - -t2->t3 - - - - - -t4 - - -ocrmypdf.pipeline.ocr_or_skip - - - -t3->t4 - - - - - -t5 - - - - -ocrmypdf.pipeline.rasterize_preview - - - -t4->t5 - - - - - -t6 - - - - -ocrmypdf.pipeline.orient_page - - - -t4->t6 - - - - - -t5->t6 - - - - - -t7 - - - - -ocrmypdf.pipeline.rasterize_with_ghostscript - - - -t6->t7 - - - - - -t17 - - - - -ocrmypdf.pipeline.ocr_tesseract_textonly_pdf - - - -t6->t17 - - - - - -t14 - -ocrmypdf.pipeline.select_image_layer - - - -t6->t14 - - - - - -t21 - - - - -ocrmypdf.pipeline.skip_page - - - -t6->t21 - - - - - -t19 - - - - -ocrmypdf.pipeline.ocr_tesseract_and_render_pdf - - - -t6->t19 - - - - - -t8 - - - - -ocrmypdf.pipeline.preprocess_remove_background - - - -t7->t8 - - - - - -t13 - -ocrmypdf.pipeline.select_visible_page_image - - - -t7->t13 - - - - - -t9 - - - - -ocrmypdf.pipeline.preprocess_deskew - - - -t8->t9 - - - - - -t8->t13 - - - - - -t10 - - - - -ocrmypdf.pipeline.preprocess_clean - - - -t9->t10 - - - - - -t9->t13 - - - - - -t11 - - - - -ocrmypdf.pipeline.select_ocr_image - - - -t10->t11 - - - - - -t10->t13 - - - - - -t12 - - - - -ocrmypdf.pipeline.ocr_tesseract_hocr - - - -t11->t12 - - - - - -t11->t17 - - - - - -t15 - - - - -ocrmypdf.pipeline.render_hocr_page - - - -t12->t15 - - - - - -t16 - - - - -ocrmypdf.pipeline.render_hocr_debug_page - - - -t12->t16 - - - - - -t25 - - -ocrmypdf.pipeline.merge_sidecars - - - -t12->t25 - - - - - -t18 - - - - -ocrmypdf.pipeline.combine_layers - - - -t15->t18 - - - - - -t17->t18 - - - - - -t17->t25 - - - - - -t13->t14 - - - - - -t13->t16 - - - - - -t13->t19 - - - - - -t14->t18 - - + + t22 - - -ocrmypdf.pipeline.merge_pages_ghostscript + + +ocrmypdf.pipeline.merge_pages + + + +t1->t22 + + + + + +t3 + + + + +ocrmypdf.pipeline.split_page + + + +t2->t3 + + + + + +t4 + + +ocrmypdf.pipeline.ocr_or_skip + + + +t3->t4 + + + + + +t5 + + + + +ocrmypdf.pipeline.rasterize_preview + + + +t4->t5 + + + + + +t6 + + + + +ocrmypdf.pipeline.orient_page + + + +t4->t6 + + + + + +t5->t6 + + + + + +t7 + + + + +ocrmypdf.pipeline.rasterize_with_ghostscript + + + +t6->t7 + + + + + +t17 + + + + +ocrmypdf.pipeline.ocr_tesseract_textonly_pdf + + + +t6->t17 + + + + + +t14 + +ocrmypdf.pipeline.select_image_layer + + + +t6->t14 + + + + + +t21 + + + + +ocrmypdf.pipeline.skip_page + + + +t6->t21 + + + + + +t19 + + + + +ocrmypdf.pipeline.ocr_tesseract_and_render_pdf + + + +t6->t19 + + + + + +t8 + + + + +ocrmypdf.pipeline.preprocess_remove_background + + + +t7->t8 + + + + + +t13 + +ocrmypdf.pipeline.select_visible_page_image + + + +t7->t13 + + + + + +t9 + + + + +ocrmypdf.pipeline.preprocess_deskew + + + +t8->t9 + + + + + +t8->t13 + + + + + +t10 + + + + +ocrmypdf.pipeline.preprocess_clean + + + +t9->t10 + + + + + +t9->t13 + + + + + +t11 + + + + +ocrmypdf.pipeline.select_ocr_image + + + +t10->t11 + + + + + +t10->t13 + + + + + +t12 + + + + +ocrmypdf.pipeline.ocr_tesseract_hocr + + + +t11->t12 + + + + + +t11->t17 + + + + + +t15 + + + + +ocrmypdf.pipeline.render_hocr_page + + + +t12->t15 + + + + + +t16 + + + + +ocrmypdf.pipeline.render_hocr_debug_page + + + +t12->t16 + + + + + +t23 + + +ocrmypdf.pipeline.merge_sidecars + + + +t12->t23 + + + + + +t18 + + + + +ocrmypdf.pipeline.combine_layers + + + +t15->t18 + + + + + +t17->t18 + + + + + +t17->t23 + + + + + +t13->t14 + + + + + +t13->t16 + + + + + +t13->t19 + + + + + +t14->t18 + + t18->t22 - - - - - -t18->t24 - - - - - -t18->t23 - - + + t16->t22 - - - - - -t16->t24 - - - - - -t16->t23 - - + + t21->t22 - - - - - -t21->t24 - - - - - -t21->t23 - - + + t19->t22 - - - - - -t19->t24 - - + + - + t19->t23 - - - - - -t19->t25 - - + + t20->t22 - - + + - - -t26 - - -ocrmypdf.pipeline.copy_final + + +t24 + + +ocrmypdf.pipeline.copy_final - - -t22->t26 - - - - - -t24->t26 - - - - - -t23->t26 - - + + +t22->t24 + + diff --git a/src/ocrmypdf/pipeline.py b/src/ocrmypdf/pipeline.py index c11c06bd..8872a7e3 100644 --- a/src/ocrmypdf/pipeline.py +++ b/src/ocrmypdf/pipeline.py @@ -927,13 +927,14 @@ def skip_page( re_symlink(input_file, output_file, log) -def _merge_pages_common( +def merge_pages( input_files_groups, output_file, log, context): """Determine ordered list of PDF pages to merge. Returns PDF from which metadata should be drawn if present (for qpdf).""" + options = context.get_options() input_files = list(f for f in flatten_groups(input_files_groups) if not f.endswith('.txt')) @@ -958,17 +959,22 @@ def _merge_pages_common( pdf_pages = sorted(input_files, key=input_file_order) log.debug("Final pages: " + "\n".join(pdf_pages)) - return pdf_pages, metadata_file + args = (pdf_pages, metadata_file, output_file, log, context) + if options.output_type.startswith('pdfa'): + _do_merge_ghostscript(*args) + elif fitz: + _do_merge_mupdf(*args) + else: + _do_merge_qpdf(*args) -def merge_pages_ghostscript( - input_files_groups, +def _do_merge_ghostscript( + pdf_pages, + metadata_file, output_file, log, context): options = context.get_options() - pdf_pages, _ = _merge_pages_common( - input_files_groups, output_file, log, context) input_pdfinfo = context.get_pdfinfo() ghostscript.generate_pdfa( pdf_version=input_pdfinfo.min_version, @@ -986,14 +992,13 @@ def merge_pages_ghostscript( os.replace(output_file + '_toc.pdf', output_file) -def merge_pages_qpdf( - input_files_groups, +def _do_merge_qpdf( + pdf_pages, + metadata_file, output_file, log, context): options = context.get_options() - pdf_pages, metadata_file = _merge_pages_common( - input_files_groups, output_file, log, context) reader_metadata = pypdf.PdfFileReader(metadata_file) pdfmark = get_pdfmark(reader_metadata, options) @@ -1017,18 +1022,15 @@ def merge_pages_qpdf( log=log) -def merge_pages_mupdf( - input_files_groups, +def _do_merge_mupdf( + pdf_pages, + metadata_file, output_file, log, context): assert fitz options = context.get_options() - - pdf_pages, metadata_file = _merge_pages_common( - input_files_groups, output_file, log, context) - doc = fitz.Document() reader_metadata = pypdf.PdfFileReader(metadata_file) @@ -1300,7 +1302,6 @@ def build_pipeline(options, work_folder, log, context): extras=[log, context]) task_generate_postscript_stub.active_if(options.output_type.startswith('pdfa')) - # Bypass valve task_skip_page = main_pipeline.transform( task_func=skip_page, @@ -1311,41 +1312,16 @@ def build_pipeline(options, work_folder, log, context): extras=[log, context]) # Merge pages - task_merge_pages_ghostscript = main_pipeline.merge( - task_func=merge_pages_ghostscript, + task_merge_pages = main_pipeline.merge( + task_func=merge_pages, input=[task_combine_layers, + task_repair_and_parse_pdf, task_render_hocr_debug_page, task_skip_page, task_ocr_tesseract_and_render_pdf, task_generate_postscript_stub], output=os.path.join(work_folder, 'merged.pdf'), extras=[log, context]) - task_merge_pages_ghostscript.active_if( - options.output_type.startswith('pdfa')) - - task_merge_pages_qpdf = main_pipeline.merge( - task_func=merge_pages_qpdf, - input=[task_combine_layers, - task_render_hocr_debug_page, - task_skip_page, - task_ocr_tesseract_and_render_pdf, - task_repair_and_parse_pdf], - output=os.path.join(work_folder, 'merged.pdf'), - extras=[log, context]) - task_merge_pages_qpdf.active_if( - options.output_type == 'pdf' and not fitz) - - task_merge_pages_mupdf = main_pipeline.merge( - task_func=merge_pages_mupdf, - input=[task_combine_layers, - task_render_hocr_debug_page, - task_skip_page, - task_ocr_tesseract_and_render_pdf, - task_repair_and_parse_pdf], - output=os.path.join(work_folder, 'merged.pdf'), - extras=[log, context]) - task_merge_pages_mupdf.active_if( - options.output_type == 'pdf' and fitz) task_merge_sidecars = main_pipeline.merge( task_func=merge_sidecars, @@ -1359,8 +1335,6 @@ def build_pipeline(options, work_folder, log, context): # Finalize main_pipeline.merge( task_func=copy_final, - input=[task_merge_pages_ghostscript, - task_merge_pages_mupdf, - task_merge_pages_qpdf], + input=[task_merge_pages], output=options.output_file, extras=[log, context])