Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2261c51eff | ||
|
|
5c470778a3 | ||
|
|
4124889f36 | ||
|
|
a23c22b0e8 | ||
|
|
dd1f5f7215 | ||
|
|
5e2206bae7 | ||
|
|
079ee86d43 |
+1
-1
@@ -32,7 +32,7 @@ stages:
|
||||
choco install --yes --no-progress --pre tesseract
|
||||
choco install --yes --no-progress python3
|
||||
choco install --yes --no-progress ghostscript
|
||||
# choco install --yes --no-progress pngquant
|
||||
choco install --yes --no-progress pngquant
|
||||
displayName: "Install system packages"
|
||||
- pwsh: |
|
||||
refreshenv
|
||||
|
||||
@@ -12,6 +12,16 @@ may be unreliable. Use the API to depend on precise behavior.
|
||||
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||
wish to use some of its features for working with PDFs.
|
||||
|
||||
v11.7.0
|
||||
=======
|
||||
|
||||
- We now support using ``--sidecar`` in conjunction with ``--pages``; these arguments
|
||||
used to be mutually exclusive. (#735)
|
||||
- Fixed a possible issue with PDF/A-1b generation. Acrobat complained that our PDFs use
|
||||
object streams. More robust PDF/A validators like veraPDF don't consider this a
|
||||
problem, but we'll honor Acrobat's objection from here on. This may increase file
|
||||
size of PDF/A-1b files. PDF/A-2b files will not be affected.
|
||||
|
||||
v11.6.2
|
||||
=======
|
||||
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@ build-backend = "setuptools.build_meta"
|
||||
|
||||
[tool.black]
|
||||
line-length = 88
|
||||
target-version = ["py36", "py37", "py38", "py39"]
|
||||
target-version = ["py36", "py37", "py38"]
|
||||
skip-string-normalization = true
|
||||
include = '\.pyi?$'
|
||||
exclude = '''
|
||||
|
||||
@@ -739,6 +739,24 @@ def should_linearize(working_file: Path, context: PdfContext):
|
||||
return False
|
||||
|
||||
|
||||
def get_pdf_save_settings(output_type: str):
|
||||
if output_type == 'pdfa-1':
|
||||
# Trigger recompression to ensure object streams are removed, because
|
||||
# Acrobat complains about them in PDF/A-1b validation.
|
||||
return dict(
|
||||
preserve_pdfa=True,
|
||||
compress_streams=True,
|
||||
stream_decode_level=pikepdf.StreamDecodeLevel.generalized,
|
||||
object_stream_mode=pikepdf.ObjectStreamMode.disable,
|
||||
)
|
||||
else:
|
||||
return dict(
|
||||
preserve_pdfa=True,
|
||||
compress_streams=True,
|
||||
object_stream_mode=(pikepdf.ObjectStreamMode.generate),
|
||||
)
|
||||
|
||||
|
||||
def metadata_fixup(working_file: Path, context: PdfContext):
|
||||
output_file = context.get_path('metafix.pdf')
|
||||
options = context.options
|
||||
@@ -783,9 +801,7 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
||||
|
||||
pdf.save(
|
||||
output_file,
|
||||
compress_streams=True,
|
||||
preserve_pdfa=True,
|
||||
object_stream_mode=pikepdf.ObjectStreamMode.generate,
|
||||
**get_pdf_save_settings(options.output_type),
|
||||
linearize=( # Don't linearize if optimize() will be linearizing too
|
||||
should_linearize(working_file, context)
|
||||
if options.optimize == 0
|
||||
@@ -799,20 +815,34 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
||||
def optimize_pdf(input_file: Path, context: PdfContext):
|
||||
output_file = context.get_path('optimize.pdf')
|
||||
save_settings = dict(
|
||||
compress_streams=True,
|
||||
preserve_pdfa=True,
|
||||
object_stream_mode=pikepdf.ObjectStreamMode.generate,
|
||||
linearize=should_linearize(input_file, context),
|
||||
**get_pdf_save_settings(context.options.output_type),
|
||||
)
|
||||
optimize(input_file, output_file, context, save_settings)
|
||||
return output_file
|
||||
|
||||
|
||||
def enumerate_compress_ranges(iterable):
|
||||
skipped_from = None
|
||||
for index, txt_file in enumerate(iterable):
|
||||
index += 1
|
||||
if txt_file:
|
||||
if skipped_from is not None:
|
||||
yield (skipped_from, index - 1), None
|
||||
skipped_from = None
|
||||
yield (index, index), txt_file
|
||||
else:
|
||||
if skipped_from is None:
|
||||
skipped_from = index
|
||||
if skipped_from is not None:
|
||||
yield (skipped_from, index), None
|
||||
|
||||
|
||||
def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
||||
output_file = context.get_path('sidecar.txt')
|
||||
with open(output_file, 'w', encoding="utf-8") as stream:
|
||||
for page_num, txt_file in enumerate(txt_files):
|
||||
if page_num != 0:
|
||||
for (frm, to), txt_file in enumerate_compress_ranges(txt_files):
|
||||
if frm != 1:
|
||||
stream.write('\f') # Form feed between pages
|
||||
if txt_file:
|
||||
with open(txt_file, 'r', encoding="utf-8") as in_:
|
||||
@@ -825,7 +855,11 @@ def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
||||
else:
|
||||
stream.write(txt)
|
||||
else:
|
||||
stream.write(f'[OCR skipped on page {(page_num + 1)}]')
|
||||
if frm != to:
|
||||
pages = f'{frm}-{to}'
|
||||
else:
|
||||
pages = f'{frm}'
|
||||
stream.write(f'[OCR skipped on page(s) {pages}]')
|
||||
return output_file
|
||||
|
||||
|
||||
|
||||
@@ -184,8 +184,6 @@ def check_options_ocr_behavior(options):
|
||||
)
|
||||
if exclusive_options >= 2:
|
||||
raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.")
|
||||
if options.pages and options.sidecar:
|
||||
raise BadArgsError("--pages and --sidecar are mutually exclusive")
|
||||
if options.pages:
|
||||
options.pages = _pages_from_ranges(options.pages)
|
||||
|
||||
|
||||
+28
-31
@@ -181,45 +181,42 @@ def check_pdf(input_file: Path) -> bool:
|
||||
Checks for proper formatting and proper linearization. Uses pikepdf (which in
|
||||
turn, uses QPDF) to perform the checks.
|
||||
"""
|
||||
pdf = None
|
||||
try:
|
||||
pdf = pikepdf.open(input_file)
|
||||
except pikepdf.PdfError as e:
|
||||
log.error(e)
|
||||
return False
|
||||
else:
|
||||
messages = pdf.check()
|
||||
for msg in messages:
|
||||
if 'error' in msg.lower():
|
||||
log.error(msg)
|
||||
with pdf:
|
||||
messages = pdf.check()
|
||||
for msg in messages:
|
||||
if 'error' in msg.lower():
|
||||
log.error(msg)
|
||||
else:
|
||||
log.warning(msg)
|
||||
|
||||
sio = StringIO()
|
||||
linearize_msgs = ''
|
||||
try:
|
||||
# If linearization is missing entirely, we do not complain. We do
|
||||
# complain if linearization is present but incorrect.
|
||||
pdf.check_linearization(sio)
|
||||
except RuntimeError:
|
||||
pass
|
||||
except ( # Workaround for a problematic pikepdf version
|
||||
getattr(pikepdf, 'ForeignObjectError')
|
||||
if pikepdf.__version__ == '2.1.0'
|
||||
else NeverRaise
|
||||
):
|
||||
pass
|
||||
else:
|
||||
log.warning(msg)
|
||||
linearize_msgs = sio.getvalue()
|
||||
if linearize_msgs:
|
||||
log.warning(linearize_msgs)
|
||||
|
||||
sio = StringIO()
|
||||
linearize_msgs = ''
|
||||
try:
|
||||
# If linearization is missing entirely, we do not complain. We do
|
||||
# complain if linearization is present but incorrect.
|
||||
pdf.check_linearization(sio)
|
||||
except RuntimeError:
|
||||
pass
|
||||
except (
|
||||
getattr(pikepdf, 'ForeignObjectError')
|
||||
if pikepdf.__version__ == '2.1.0' # This version may throw wrong exception
|
||||
else NeverRaise
|
||||
):
|
||||
pass
|
||||
else:
|
||||
linearize_msgs = sio.getvalue()
|
||||
if linearize_msgs:
|
||||
log.warning(linearize_msgs)
|
||||
|
||||
if not messages and not linearize_msgs:
|
||||
return True
|
||||
return False
|
||||
finally:
|
||||
if pdf:
|
||||
pdf.close()
|
||||
if not messages and not linearize_msgs:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def clamp(n, smallest, largest): # mypy doesn't understand types for this
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
# © 2021 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
|
||||
check_ocrmypdf = pytest.helpers.check_ocrmypdf
|
||||
|
||||
|
||||
@pytest.mark.parametrize('optimize', (0, 3))
|
||||
@pytest.mark.parametrize('pdfa_level', (1, 2, 3))
|
||||
def test_pdfa(resources, outpdf, optimize, pdfa_level):
|
||||
check_ocrmypdf(
|
||||
resources / 'francais.pdf',
|
||||
outpdf,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
f'--output-type=pdfa-{pdfa_level}',
|
||||
f'--optimize={optimize}',
|
||||
)
|
||||
if pdfa_level in (2, 3):
|
||||
# PDF/A-2 allows ObjStm
|
||||
assert b'/ObjStm' in outpdf.read_bytes()
|
||||
elif pdfa_level == 1:
|
||||
# PDF/A-1 might allow ObjStm, but Acrobat does not approve it, so
|
||||
# we don't use it
|
||||
assert b'/ObjStm' not in outpdf.read_bytes()
|
||||
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
with pdf.open_metadata() as m:
|
||||
assert m.pdfa_status == f'{pdfa_level}B'
|
||||
@@ -61,3 +61,90 @@ def test_dpi_needed(image, text, vector, result, rgb_image, outdir):
|
||||
|
||||
assert _pipeline.get_canvas_square_dpi(pi[0], mock) == result
|
||||
assert _pipeline.get_page_square_dpi(pi[0], mock) == result
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
# Name for nicer -v output
|
||||
'name,input,output',
|
||||
(
|
||||
(
|
||||
'empty_input',
|
||||
# Input:
|
||||
(),
|
||||
# Output:
|
||||
(),
|
||||
),
|
||||
(
|
||||
'no_values',
|
||||
# Input:
|
||||
('', '', '', '', ''),
|
||||
# Output:
|
||||
(
|
||||
((1, 5), None),
|
||||
),
|
||||
),
|
||||
(
|
||||
'no_empty_values',
|
||||
# Input:
|
||||
('v', 'w', 'x', 'y', 'z'),
|
||||
# Output:
|
||||
(
|
||||
((1, 1), 'v'),
|
||||
((2, 2), 'w'),
|
||||
((3, 3), 'x'),
|
||||
((4, 4), 'y'),
|
||||
((5, 5), 'z'),
|
||||
),
|
||||
),
|
||||
(
|
||||
'skip_head',
|
||||
# Input:
|
||||
('', '', 'x', 'y', 'z'),
|
||||
# Output:
|
||||
(
|
||||
((1, 2), None),
|
||||
((3, 3), 'x'),
|
||||
((4, 4), 'y'),
|
||||
((5, 5), 'z'),
|
||||
),
|
||||
),
|
||||
(
|
||||
'skip_tail',
|
||||
# Input:
|
||||
('x', 'y', 'z', '', ''),
|
||||
# Output:
|
||||
(
|
||||
((1, 1), 'x'),
|
||||
((2, 2), 'y'),
|
||||
((3, 3), 'z'),
|
||||
((4, 5), None),
|
||||
),
|
||||
),
|
||||
(
|
||||
'range_in_middle',
|
||||
# Input:
|
||||
('x', '', '', '', 'y'),
|
||||
# Output:
|
||||
(
|
||||
((1, 1), 'x'),
|
||||
((2, 4), None),
|
||||
((5, 5), 'y'),
|
||||
),
|
||||
),
|
||||
(
|
||||
'range_in_middle_2',
|
||||
# Input:
|
||||
('x', '', '', 'y', '', '', '', 'z'),
|
||||
# Output:
|
||||
(
|
||||
((1, 1), 'x'),
|
||||
((2, 3), None),
|
||||
((4, 4), 'y'),
|
||||
((5, 7), None),
|
||||
((8, 8), 'z'),
|
||||
),
|
||||
),
|
||||
),
|
||||
)
|
||||
def test_enumerate_compress_ranges(name, input, output):
|
||||
assert output == tuple(_pipeline.enumerate_compress_ranges(input))
|
||||
@@ -90,8 +90,6 @@ def test_mutex_options():
|
||||
vd.check_options_ocr_behavior(make_opts(redo_ocr=True, skip_text=True))
|
||||
with pytest.raises(BadArgsError):
|
||||
vd.check_options_ocr_behavior(make_opts(redo_ocr=True, force_ocr=True))
|
||||
with pytest.raises(BadArgsError):
|
||||
vd.check_options_ocr_behavior(make_opts(pages='1-3', sidecar='file.txt'))
|
||||
|
||||
|
||||
def test_optimizing(caplog):
|
||||
|
||||
Reference in New Issue
Block a user