Compare commits

...
11 Commits
6 changed files with 94 additions and 39 deletions
+2 -2
View File
@@ -31,7 +31,7 @@ jobs:
- os: ubuntu-latest
python: "3.9"
- os: ubuntu-latest
python: "pypy-3.7"
python: "pypy-3.8"
- os: ubuntu-latest
python: "3.9"
tesseract5: true
@@ -196,7 +196,7 @@ jobs:
- name: Install system packages
run: |
choco install --yes --no-progress --pre tesseract
choco install --yes --no-progress --ignore-checksums ghostscript pngquant
choco install --yes --no-progress --ignore-checksums ghostscript
- name: Install Python packages
run: |
+3 -3
View File
@@ -1,6 +1,6 @@
repos:
- repo: https://github.com/pre-commit/pre-commit-hooks
rev: v4.1.0
rev: v4.2.0
hooks:
- id: check-case-conflict
- id: check-merge-conflict
@@ -22,12 +22,12 @@ repos:
hooks:
- id: setup-cfg-fmt
- repo: https://github.com/asottile/pyupgrade
rev: v2.31.1
rev: v2.32.1
hooks:
- id: pyupgrade
args: ["--py37-plus"]
- repo: https://github.com/pre-commit/mirrors-mypy
rev: v0.942
rev: v0.950
hooks:
- id: mypy
additional_dependencies:
+5
View File
@@ -18,6 +18,11 @@ tagged yet.
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
v13.4.6
=======
- Convert error on corrupt ICC profiles into a warning. Thanks to @oscherler.
v13.4.5
=======
+18 -15
View File
@@ -37,6 +37,7 @@ from pikepdf import (
PdfImage,
PdfInlineImage,
PdfMatrix,
UnsupportedImageTypeError,
parse_content_stream,
)
@@ -350,13 +351,20 @@ class ImageInfo:
if self._color == Colorspace.icc:
# Check the ICC profile to determine actual colorspace
pim_icc = pim.icc
if pim_icc.profile.xcolor_space == 'GRAY':
self._comp = 1
elif pim_icc.profile.xcolor_space == 'CMYK':
self._comp = 4
else:
self._comp = 3
try:
pim_icc = pim.icc
if pim_icc.profile.xcolor_space == 'GRAY':
self._comp = 1
elif pim_icc.profile.xcolor_space == 'CMYK':
self._comp = 4
else:
self._comp = 3
except UnsupportedImageTypeError as ex:
self._comp = None
logger.warning(
f"An image with a corrupt or unreadable ICC profile was found. "
f"The output PDF may not match the input PDF visually: {ex}. {self}"
)
else:
if isinstance(self._color, Colorspace):
self._comp = FRIENDLY_COMP.get(self._color)
@@ -409,15 +417,10 @@ class ImageInfo:
return _get_dpi(self._shorthand, (self._width, self._height))
def __repr__(self):
class_locals = {
attr: getattr(self, attr, None)
for attr in dir(self)
if not attr.startswith('_')
}
return (
"<ImageInfo '{name}' {type_} {width}x{height} {color} "
"{comp} {bpc} {enc} {dpi}>"
).format(**class_locals)
f"<ImageInfo '{self.name}' {self.type_} {self.width}x{self.height} "
f"{self.color} {self.comp} {self.bpc} {self.enc} {self.dpi}>"
)
def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
+50 -11
View File
@@ -465,12 +465,18 @@ def test_overlay(resources, outpdf):
)
def test_destination_not_writable(resources, outdir):
if os.name != 'nt' and (os.getuid() == 0 or os.geteuid() == 0):
pytest.xfail(reason="root can write to anything")
@pytest.fixture
def protected_file(outdir):
protected_file = outdir / 'protected.pdf'
protected_file.touch()
protected_file.chmod(0o400) # Read-only
yield protected_file
@pytest.mark.skipif(
os.name == 'nt' or os.geteuid() == 0, reason="root can write to anything"
)
def test_destination_not_writable(resources, protected_file):
p = run_ocrmypdf(
resources / 'jbig2.pdf',
protected_file,
@@ -480,7 +486,8 @@ def test_destination_not_writable(resources, outdir):
assert p.returncode == ExitCode.file_access_error, "Expected error"
def test_tesseract_config_valid(resources, outdir):
@pytest.fixture
def valid_tess_config(outdir):
cfg_file = outdir / 'test.cfg'
with cfg_file.open('w') as f:
f.write(
@@ -490,20 +497,22 @@ language_model_penalty_non_dict_word 0
language_model_penalty_non_freq_dict_word 0
'''
)
yield cfg_file
def test_tesseract_config_valid(resources, valid_tess_config, outpdf):
check_ocrmypdf(
resources / '3small.pdf',
outdir / 'out.pdf',
outpdf,
'--tesseract-config',
cfg_file,
valid_tess_config,
'--pages',
'1',
)
@pytest.mark.slow # This test sometimes times out in CI
@pytest.mark.parametrize('renderer', RENDERERS)
def test_tesseract_config_invalid(renderer, resources, outdir):
@pytest.fixture
def invalid_tess_config(outdir):
cfg_file = outdir / 'test.cfg'
with cfg_file.open('w') as f:
f.write(
@@ -511,14 +520,19 @@ def test_tesseract_config_invalid(renderer, resources, outdir):
THIS FILE IS INVALID
'''
)
yield cfg_file
@pytest.mark.slow # This test sometimes times out in CI
@pytest.mark.parametrize('renderer', RENDERERS)
def test_tesseract_config_invalid(renderer, resources, invalid_tess_config, outpdf):
p = run_ocrmypdf(
resources / 'ccitt.pdf',
outdir / 'out.pdf',
outpdf,
'--pdf-renderer',
renderer,
'--tesseract-config',
cfg_file,
invalid_tess_config,
)
assert (
"parameter not found" in p.stderr.lower()
@@ -801,6 +815,9 @@ def test_text_curves(resources, outpdf):
info = PdfInfo(outpdf)
assert len(info.pages[0].images) == 0, "added images to the vector PDF"
def test_text_curves_force(resources, outpdf):
with patch('ocrmypdf._pipeline.VECTOR_PAGE_DPI', 100):
check_ocrmypdf(
resources / 'vector.pdf',
outpdf,
@@ -922,3 +939,25 @@ def test_outputtype_none(resources, outtxt):
'tests/plugins/tesseract_noop.py',
)
assert p.returncode == ExitCode.ok
@pytest.fixture
def graph_bad_icc(resources, outdir):
synth_input_file = outdir / 'graph-bad-icc.pdf'
with pikepdf.open(resources / 'graph.pdf') as pdf:
icc = pdf.make_stream(
b'invalid icc profile', N=3, Alternate=pikepdf.Name.DeviceRGB
)
pdf.pages[0].Resources.XObject['/Im0'].ColorSpace = pikepdf.Array(
[pikepdf.Name.ICCBased, icc]
)
pdf.save(synth_input_file)
yield synth_input_file
def test_corrupt_icc(graph_bad_icc, outpdf, caplog):
result = run_ocrmypdf_api(graph_bad_icc, outpdf)
assert result == ExitCode.ok
assert any(
'corrupt or unreadable ICC profile' in rec.message for rec in caplog.records
)
+16 -8
View File
@@ -4,23 +4,31 @@
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
import os
import pikepdf
import pytest
from ocrmypdf.exceptions import MissingDependencyError
from .conftest import check_ocrmypdf
@pytest.mark.parametrize('optimize', (0, 3))
@pytest.mark.parametrize('pdfa_level', (1, 2, 3))
def test_pdfa(resources, outpdf, optimize, pdfa_level):
check_ocrmypdf(
resources / 'francais.pdf',
outpdf,
'--plugin',
'tests/plugins/tesseract_noop.py',
f'--output-type=pdfa-{pdfa_level}',
f'--optimize={optimize}',
)
try:
check_ocrmypdf(
resources / 'francais.pdf',
outpdf,
'--plugin',
'tests/plugins/tesseract_noop.py',
f'--output-type=pdfa-{pdfa_level}',
f'--optimize={optimize}',
)
except MissingDependencyError as e:
if 'pngquant' in str(e) and optimize in (2, 3) and os.name == 'nt':
pytest.xfail("pngquant currently not available on Windows")
if pdfa_level in (2, 3):
# PDF/A-2 allows ObjStm
assert b'/ObjStm' in outpdf.read_bytes()