Refactor weave_layers, introduce progress bar
Fixes a bug in this branch where --sidecar would fail by trying to iterator the executor futures twice.
This commit is contained in:
@@ -10,3 +10,4 @@ Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin"
|
||||
pycparser == 2.19
|
||||
python-xmp-toolkit == 2.0.1
|
||||
reportlab == 3.5.13
|
||||
tqdm == 4.32.1
|
||||
|
||||
@@ -104,6 +104,7 @@ setup(
|
||||
# Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3
|
||||
# block 5.1.0, broken wheels
|
||||
'reportlab >= 3.3.0', # oldest released version with sane image handling
|
||||
'tqdm >= 4',
|
||||
],
|
||||
extras_require={'pdfminer': ['pdfminer.six == 20181108']},
|
||||
tests_require=tests_require,
|
||||
|
||||
@@ -21,6 +21,8 @@ import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
from tqdm import tqdm
|
||||
|
||||
from . import PROGRAM_NAME, VERSION
|
||||
from .exceptions import ExitCode
|
||||
from ._sync import run_pipeline
|
||||
@@ -464,6 +466,21 @@ debugging.add_argument(
|
||||
)
|
||||
|
||||
|
||||
class TqdmConsole:
|
||||
"""Wrapper to log messages in a way that is compatible with the progress bar"""
|
||||
|
||||
def __init__(self, file):
|
||||
self.file = file
|
||||
|
||||
def write(self, msg):
|
||||
# When no progress bar is active, tqdm.write() routes to print()
|
||||
tqdm.write(msg.rstrip(), file=self.file)
|
||||
|
||||
def flush(self):
|
||||
if hasattr(self.file, "flush"):
|
||||
self.file.flush()
|
||||
|
||||
|
||||
def setup_app_logging(options):
|
||||
"""Set up logging"""
|
||||
|
||||
|
||||
+37
-12
@@ -18,9 +18,10 @@
|
||||
import os
|
||||
import atexit
|
||||
import concurrent.futures
|
||||
from collections import namedtuple
|
||||
from tempfile import mkdtemp
|
||||
from ._jobcontext import PDFContext, get_logger, cleanup_working_files
|
||||
from ._weave import weave_layers
|
||||
from ._weave import OcrGrafter
|
||||
from ._pipeline import (
|
||||
triage,
|
||||
get_pdfinfo,
|
||||
@@ -60,6 +61,12 @@ from ._validation import (
|
||||
from .pdfa import file_claims_pdfa
|
||||
from .exec import qpdf
|
||||
|
||||
from tqdm import tqdm
|
||||
|
||||
PageResult = namedtuple(
|
||||
'PageResult', 'pageno, pdf_page_from_image, ocr, text, orientation_correction'
|
||||
)
|
||||
|
||||
|
||||
def exec_page_sync(page_context):
|
||||
options = page_context.options
|
||||
@@ -115,12 +122,12 @@ def exec_page_sync(page_context):
|
||||
ocr_image_out, page_context
|
||||
)
|
||||
|
||||
return (
|
||||
page_context.pageno,
|
||||
pdf_page_from_image_out,
|
||||
ocr_out,
|
||||
text_out,
|
||||
orientation_correction,
|
||||
return PageResult(
|
||||
pageno=page_context.pageno,
|
||||
pdf_page_from_image=pdf_page_from_image_out,
|
||||
ocr=ocr_out,
|
||||
text=text_out,
|
||||
orientation_correction=orientation_correction,
|
||||
)
|
||||
|
||||
|
||||
@@ -141,18 +148,36 @@ def exec_concurrent(context):
|
||||
max_workers = min(len(context.pdfinfo), context.options.jobs)
|
||||
if max_workers > 1:
|
||||
context.log.info("Start processing %d pages concurrent" % max_workers)
|
||||
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||
layers = executor.map(exec_page_sync, context.get_page_contexts())
|
||||
|
||||
sidecars = {}
|
||||
layers = []
|
||||
ocrgraft = OcrGrafter(context)
|
||||
with tqdm(
|
||||
total=(2 * len(context.pdfinfo)), desc='OCR', unit='page', unit_scale=0.5
|
||||
) as pbar, concurrent.futures.ProcessPoolExecutor(
|
||||
max_workers=max_workers
|
||||
) as executor:
|
||||
# layers = executor.map(exec_page_sync, context.get_page_contexts())
|
||||
futures = [
|
||||
executor.submit(exec_page_sync, ctx) for ctx in context.get_page_contexts()
|
||||
]
|
||||
for future in concurrent.futures.as_completed(futures):
|
||||
page_result = future.result()
|
||||
sidecars[page_result.pageno] = page_result.text
|
||||
pbar.update()
|
||||
ocrgraft.graft_page(page_result)
|
||||
pbar.update()
|
||||
|
||||
# Output sidecar text
|
||||
if context.options.sidecar:
|
||||
sidecars = [layer[3] for layer in layers]
|
||||
text = merge_sidecars(sidecars, context)
|
||||
ordered_sidecars = [sidecars[pageno] for pageno in sorted(sidecars)]
|
||||
text = merge_sidecars(ordered_sidecars, context)
|
||||
# Copy text file to destination
|
||||
copy_final(text, context.options.sidecar, context)
|
||||
|
||||
# Merge layers to one single pdf
|
||||
pdf = weave_layers(layers, context)
|
||||
# pdf = weave_layers(layers, context)
|
||||
pdf = ocrgraft.finalize()
|
||||
|
||||
# PDF/A and metadata
|
||||
pdf = post_process(pdf, context)
|
||||
|
||||
+70
-87
@@ -185,126 +185,109 @@ def _find_font(text, pdf_base):
|
||||
return None, None
|
||||
|
||||
|
||||
def weave_layers(layers, context):
|
||||
"""Apply text layer and/or image layer changes to baseline file
|
||||
class OcrGrafter:
|
||||
def __init__(self, context):
|
||||
self.context = context
|
||||
self.log = context.log
|
||||
self.path_base = Path(context.origin).resolve()
|
||||
|
||||
This is where the magic happens. infiles will be the main PDF to modify,
|
||||
and optional .text.pdf and .image-layer.pdf files, organized however ruffus
|
||||
organizes them.
|
||||
self.pdf_base = pikepdf.open(self.path_base)
|
||||
self.font, self.font_key = None, None
|
||||
|
||||
From .text.pdf, we copy the content stream (which contains the Tesseract
|
||||
OCR results), and rotate it into place. The first time we do this, we also
|
||||
copy the GlyphlessFont, and then reference that font again.
|
||||
self.pdfinfo = context.pdfinfo
|
||||
self.output_file = context.get_path('weave_layers.pdf')
|
||||
|
||||
For .image-layer.pdf, we check if this is a "pointer" to the original file,
|
||||
or a new file. If a new file, we replace the page and remember that we
|
||||
replaced this page.
|
||||
self.procset = self.pdf_base.make_indirect(
|
||||
pikepdf.Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]')
|
||||
)
|
||||
|
||||
Every 100 open files, we save intermediate results, to avoid any resource
|
||||
limits, since pikepdf/qpdf need to keep a lot of open file handles in the
|
||||
background. When objects are copied from one file to another qpdf, qpdf
|
||||
doesn't actually copy the data until asked to write, so all the resources
|
||||
it may need to remain available.
|
||||
self.emplacements = 1
|
||||
self.interim_count = 0
|
||||
|
||||
For completeness, we set up a /ProcSet on every page, although it's
|
||||
unlikely any PDF viewer cares about this anymore.
|
||||
|
||||
"""
|
||||
|
||||
log = context.log
|
||||
|
||||
path_base = Path(context.origin).resolve()
|
||||
pdf_base = pikepdf.open(path_base)
|
||||
font, font_key, procset = None, None, None
|
||||
|
||||
pdfinfo = context.pdfinfo
|
||||
output_file = context.get_path('weave_layers.pdf')
|
||||
|
||||
procset = pdf_base.make_indirect(
|
||||
pikepdf.Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]')
|
||||
)
|
||||
|
||||
emplacements = 1
|
||||
interim_count = 0
|
||||
|
||||
# Iterate rest
|
||||
for (pageno, image, text, sidecar, autorotate_correction) in layers:
|
||||
if text and not font:
|
||||
font, font_key = _find_font(text, pdf_base)
|
||||
def graft_page(self, page_result):
|
||||
pageno, image, text, sidecar, autorotate_correction = page_result
|
||||
if text and not self.font:
|
||||
self.font, self.font_key = _find_font(text, self.pdf_base)
|
||||
|
||||
emplaced_page = False
|
||||
content_rotation = pdfinfo[pageno].rotation
|
||||
content_rotation = self.pdfinfo[pageno].rotation
|
||||
path_image = Path(image).resolve() if image else None
|
||||
if path_image is not None and path_image != path_base:
|
||||
if path_image is not None and path_image != self.path_base:
|
||||
# We are updating the old page with a rasterized PDF of the new
|
||||
# page (without changing objgen, to preserve references)
|
||||
log.debug("Emplacement update")
|
||||
self.log.debug("Emplacement update")
|
||||
with pikepdf.open(image) as pdf_image:
|
||||
emplacements += 1
|
||||
self.emplacements += 1
|
||||
foreign_image_page = pdf_image.pages[0]
|
||||
pdf_base.pages.append(foreign_image_page)
|
||||
local_image_page = pdf_base.pages[-1]
|
||||
pdf_base.pages[pageno].emplace(local_image_page)
|
||||
del pdf_base.pages[-1]
|
||||
self.pdf_base.pages.append(foreign_image_page)
|
||||
local_image_page = self.pdf_base.pages[-1]
|
||||
self.pdf_base.pages[pageno].emplace(local_image_page)
|
||||
del self.pdf_base.pages[-1]
|
||||
emplaced_page = True
|
||||
|
||||
if emplaced_page:
|
||||
content_rotation = autorotate_correction
|
||||
text_rotation = autorotate_correction
|
||||
text_misaligned = (text_rotation - content_rotation) % 360
|
||||
log.debug(
|
||||
self.log.debug(
|
||||
'%r',
|
||||
[text_rotation, autorotate_correction, text_misaligned, content_rotation],
|
||||
)
|
||||
|
||||
if text and font:
|
||||
if text and self.font:
|
||||
# Graft the text layer onto this page, whether new or old
|
||||
strip_old = context.options.redo_ocr
|
||||
strip_old = self.context.options.redo_ocr
|
||||
_weave_layers_graft(
|
||||
pdf_base=pdf_base,
|
||||
pdf_base=self.pdf_base,
|
||||
page_num=pageno + 1,
|
||||
text=text,
|
||||
font=font,
|
||||
font_key=font_key,
|
||||
font=self.font,
|
||||
font_key=self.font_key,
|
||||
rotation=text_misaligned,
|
||||
procset=procset,
|
||||
procset=self.procset,
|
||||
strip_old_text=strip_old,
|
||||
log=log,
|
||||
log=self.log,
|
||||
)
|
||||
|
||||
# Correct the rotation if applicable
|
||||
pdf_base.pages[pageno].Rotate = (content_rotation - autorotate_correction) % 360
|
||||
self.pdf_base.pages[pageno].Rotate = (
|
||||
content_rotation - autorotate_correction
|
||||
) % 360
|
||||
|
||||
if emplacements % MAX_REPLACE_PAGES == 0:
|
||||
# Periodically save and reload the Pdf object. This will keep a
|
||||
# lid on our memory usage for very large files. Attach the font to
|
||||
# page 1 even if page 1 doesn't use it, so we have a way to get it
|
||||
# back.
|
||||
# TODO refactor this to outside the loop
|
||||
page0 = pdf_base.pages[0]
|
||||
_update_page_resources(
|
||||
page=page0, font=font, font_key=font_key, procset=procset
|
||||
)
|
||||
if self.emplacements % MAX_REPLACE_PAGES == 0:
|
||||
self.save_and_reload()
|
||||
|
||||
# We cannot read and write the same file, that will corrupt it
|
||||
# but we don't to keep more copies than we need to. Delete intermediates.
|
||||
# {interim_count} is the opened file we were updateing
|
||||
# {interim_count - 1} can be deleted
|
||||
# {interim_count + 1} is the new file will produce and open
|
||||
old_file = output_file + f'_working{interim_count - 1}.pdf'
|
||||
if not context.options.keep_temporary_files:
|
||||
with suppress(FileNotFoundError):
|
||||
os.unlink(old_file)
|
||||
def save_and_reload(self):
|
||||
# Periodically save and reload the Pdf object. This will keep a
|
||||
# lid on our memory usage for very large files. Attach the font to
|
||||
# page 1 even if page 1 doesn't use it, so we have a way to get it
|
||||
# back.
|
||||
# TODO refactor this to outside the loop
|
||||
page0 = self.pdf_base.pages[0]
|
||||
_update_page_resources(
|
||||
page=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
||||
)
|
||||
|
||||
next_file = output_file + f'_working{interim_count + 1}.pdf'
|
||||
pdf_base.save(next_file)
|
||||
pdf_base.close()
|
||||
# We cannot read and write the same file, that will corrupt it
|
||||
# but we don't to keep more copies than we need to. Delete intermediates.
|
||||
# {interim_count} is the opened file we were updateing
|
||||
# {interim_count - 1} can be deleted
|
||||
# {interim_count + 1} is the new file will produce and open
|
||||
old_file = self.output_file + f'_working{self.interim_count - 1}.pdf'
|
||||
if not self.context.options.keep_temporary_files:
|
||||
with suppress(FileNotFoundError):
|
||||
os.unlink(old_file)
|
||||
|
||||
pdf_base = pikepdf.open(next_file)
|
||||
procset = pdf_base.pages[0].Resources.ProcSet
|
||||
font, font_key = None, None # Ensure we reacquire this information
|
||||
interim_count += 1
|
||||
next_file = self.output_file + f'_working{self.interim_count + 1}.pdf'
|
||||
self.pdf_base.save(next_file)
|
||||
self.pdf_base.close()
|
||||
|
||||
pdf_base.save(output_file)
|
||||
pdf_base.close()
|
||||
return output_file
|
||||
self.pdf_base = pikepdf.open(next_file)
|
||||
self.procset = self.pdf_base.pages[0].Resources.ProcSet
|
||||
self.font, self.font_key = None, None # Ensure we reacquire this information
|
||||
self.interim_count += 1
|
||||
|
||||
def finalize(self):
|
||||
self.pdf_base.save(self.output_file)
|
||||
self.pdf_base.close()
|
||||
return self.output_file
|
||||
|
||||
@@ -59,7 +59,7 @@ def test_no_unpaper(resources, no_outpdf):
|
||||
with patch("ocrmypdf.exec.unpaper.version") as mock_unpaper_version:
|
||||
mock_unpaper_version.side_effect = FileNotFoundError("unpaper")
|
||||
with pytest.raises(SystemExit):
|
||||
check_options(options, log=logging.getLogger())
|
||||
check_options(options)
|
||||
|
||||
|
||||
def test_old_unpaper(spoof_unpaper_oldversion, resources, no_outpdf):
|
||||
|
||||
Reference in New Issue
Block a user