Compare commits

...
18 Commits
Author SHA1 Message Date
James R. Barlow 21cacad93b unpaper: super syntax 2022-06-02 17:05:59 -07:00
James R. Barlow 3589f4e7d1 unpaper: fix unsaved change 2022-06-02 02:22:48 -07:00
James R. Barlow 1cdc2591e5 v13.4.7 release notes 2022-06-02 02:00:16 -07:00
James R. Barlow e05f9575a8 Merge remote-tracking branch 'origin/master' 2022-06-02 01:57:39 -07:00
James R. Barlow 10c703e119 unpaper: use TemporaryDirectory(ignore_cleanup_errors=True) where available
Fixes #974 when used in conjunction with Python 3.10.

Reviewed other uses of TemporaryDirectory in ocrmypdf and decided it
was not worth fixing them since neither are exactly production code.
2022-06-02 01:56:41 -07:00
James R. Barlow 0ac15dd0b2 Suppress libxmp DeprecationWarning during test 2022-06-01 00:46:16 -07:00
Robert SchützandGitHub 808b24d59f ignore PermissionError when calling os.nice() (#973) 2022-05-28 18:07:54 -07:00
James R. Barlow c082526dea Test pypy-3.8 instead of 3.7 2022-05-26 13:52:56 -07:00
James R. Barlow 33cdabaf65 tests: account for test that expected pngquant for windows 2022-05-26 13:52:22 -07:00
James R. Barlow 94f8e36601 ci: don't install pngquant for windows anymore 2022-05-26 13:01:58 -07:00
James R. Barlow 865002c7be v13.4.6 release notes 2022-05-26 00:59:14 -07:00
James R. Barlow 5d0cc0a092 tests: Extract some test fixtures for better clarity 2022-05-26 00:57:31 -07:00
James R. Barlow 6c427f82ea Add test case for corrupt ICC profiles 2022-05-26 00:41:19 -07:00
James R. Barlow e7a44ba87a info: adjust ICC warning message 2022-05-26 00:21:31 -07:00
James R. Barlow c311768452 Merge branch 'corrupt-icc' of https://github.com/oscherler/OCRmyPDF into oscherler-corrupt-icc 2022-05-26 00:14:47 -07:00
James R. Barlow f53fedee63 pre-commit: autoupdate 2022-05-25 15:21:55 -07:00
James R. Barlow 87838127b0 info: replace introspection with explicit f-string 2022-05-25 13:16:40 -07:00
Olivier Scherler 4db4df5c72 Log a warning instead of failing on images with a corrupt ICC profile. 2022-05-25 12:37:21 +02:00
9 changed files with 141 additions and 51 deletions
+2 -2
View File
@@ -31,7 +31,7 @@ jobs:
- os: ubuntu-latest
python: "3.9"
- os: ubuntu-latest
python: "pypy-3.7"
python: "pypy-3.8"
- os: ubuntu-latest
python: "3.9"
tesseract5: true
@@ -196,7 +196,7 @@ jobs:
- name: Install system packages
run: |
choco install --yes --no-progress --pre tesseract
choco install --yes --no-progress --ignore-checksums ghostscript pngquant
choco install --yes --no-progress --ignore-checksums ghostscript
- name: Install Python packages
run: |
+3 -3
View File
@@ -1,6 +1,6 @@
repos:
- repo: https://github.com/pre-commit/pre-commit-hooks
rev: v4.1.0
rev: v4.2.0
hooks:
- id: check-case-conflict
- id: check-merge-conflict
@@ -22,12 +22,12 @@ repos:
hooks:
- id: setup-cfg-fmt
- repo: https://github.com/asottile/pyupgrade
rev: v2.31.1
rev: v2.32.1
hooks:
- id: pyupgrade
args: ["--py37-plus"]
- repo: https://github.com/pre-commit/mirrors-mypy
rev: v0.942
rev: v0.950
hooks:
- id: mypy
additional_dependencies:
+12
View File
@@ -18,6 +18,18 @@ tagged yet.
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
v13.4.7
=======
- Fixed PermissionError when cleaning up temporary files in rare cases. :issue:`974`
- Fixed PermissionError when calling ``os.nice`` on platforms that lack it. :issue:`973`
- Suppressed some warnings from libxmp during tests.
v13.4.6
=======
- Convert error on corrupt ICC profiles into a warning. Thanks to @oscherler.
v13.4.5
=======
+2 -1
View File
@@ -10,6 +10,7 @@ import logging
import os
import signal
import sys
from contextlib import suppress
from multiprocessing import set_start_method
from ocrmypdf import __version__
@@ -34,7 +35,7 @@ def sigbus(*args):
def run(args=None):
_parser, options, plugin_manager = get_parser_options_plugins(args=args)
if hasattr(os, 'nice'):
with suppress(AttributeError, PermissionError):
os.nice(5)
verbosity = options.verbose
+19 -2
View File
@@ -13,11 +13,11 @@
import logging
import os
import shlex
import sys
from contextlib import contextmanager
from decimal import Decimal
from pathlib import Path
from subprocess import PIPE, STDOUT
from tempfile import TemporaryDirectory
from typing import Iterator, List, Optional, Tuple, Union
from PIL import Image
@@ -25,6 +25,23 @@ from PIL import Image
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
from ocrmypdf.subprocess import get_version, run
if sys.version_info >= (3, 10):
from tempfile import TemporaryDirectory
else:
from tempfile import TemporaryDirectory as _TemporaryDirectory
# Consume the ignore_cleanup_errors kwarg in Python 3.9 and older, without acting
# on this keyword. Users who need this issue full resolved should upgrade to Python
# 3.10.
# See: https://github.com/python/cpython/pull/24793
class TemporaryDirectory(_TemporaryDirectory):
def __init__(self, ignore_cleanup_errors=False, **kwargs):
super().__init__(**kwargs)
del _TemporaryDirectory
UNPAPER_IMAGE_PIXEL_LIMIT = 256 * 1024 * 1024
DecFloat = Union[Decimal, float]
@@ -82,7 +99,7 @@ def _setup_unpaper_io(input_file: Path) -> Iterator[Tuple[Path, Path, Path]]:
raise UnpaperImageTooLargeError(w=im.width, h=im.height)
im, im_modified, suffix = _convert_image(im)
with TemporaryDirectory() as tmpdir:
with TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
tmppath = Path(tmpdir)
if im_modified or input_file.suffix != '.pnm':
input_pnm = tmppath / 'input.pnm'
+18 -15
View File
@@ -37,6 +37,7 @@ from pikepdf import (
PdfImage,
PdfInlineImage,
PdfMatrix,
UnsupportedImageTypeError,
parse_content_stream,
)
@@ -350,13 +351,20 @@ class ImageInfo:
if self._color == Colorspace.icc:
# Check the ICC profile to determine actual colorspace
pim_icc = pim.icc
if pim_icc.profile.xcolor_space == 'GRAY':
self._comp = 1
elif pim_icc.profile.xcolor_space == 'CMYK':
self._comp = 4
else:
self._comp = 3
try:
pim_icc = pim.icc
if pim_icc.profile.xcolor_space == 'GRAY':
self._comp = 1
elif pim_icc.profile.xcolor_space == 'CMYK':
self._comp = 4
else:
self._comp = 3
except UnsupportedImageTypeError as ex:
self._comp = None
logger.warning(
f"An image with a corrupt or unreadable ICC profile was found. "
f"The output PDF may not match the input PDF visually: {ex}. {self}"
)
else:
if isinstance(self._color, Colorspace):
self._comp = FRIENDLY_COMP.get(self._color)
@@ -409,15 +417,10 @@ class ImageInfo:
return _get_dpi(self._shorthand, (self._width, self._height))
def __repr__(self):
class_locals = {
attr: getattr(self, attr, None)
for attr in dir(self)
if not attr.startswith('_')
}
return (
"<ImageInfo '{name}' {type_} {width}x{height} {color} "
"{comp} {bpc} {enc} {dpi}>"
).format(**class_locals)
f"<ImageInfo '{self.name}' {self.type_} {self.width}x{self.height} "
f"{self.color} {self.comp} {self.bpc} {self.enc} {self.dpi}>"
)
def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
+50 -11
View File
@@ -465,12 +465,18 @@ def test_overlay(resources, outpdf):
)
def test_destination_not_writable(resources, outdir):
if os.name != 'nt' and (os.getuid() == 0 or os.geteuid() == 0):
pytest.xfail(reason="root can write to anything")
@pytest.fixture
def protected_file(outdir):
protected_file = outdir / 'protected.pdf'
protected_file.touch()
protected_file.chmod(0o400) # Read-only
yield protected_file
@pytest.mark.skipif(
os.name == 'nt' or os.geteuid() == 0, reason="root can write to anything"
)
def test_destination_not_writable(resources, protected_file):
p = run_ocrmypdf(
resources / 'jbig2.pdf',
protected_file,
@@ -480,7 +486,8 @@ def test_destination_not_writable(resources, outdir):
assert p.returncode == ExitCode.file_access_error, "Expected error"
def test_tesseract_config_valid(resources, outdir):
@pytest.fixture
def valid_tess_config(outdir):
cfg_file = outdir / 'test.cfg'
with cfg_file.open('w') as f:
f.write(
@@ -490,20 +497,22 @@ language_model_penalty_non_dict_word 0
language_model_penalty_non_freq_dict_word 0
'''
)
yield cfg_file
def test_tesseract_config_valid(resources, valid_tess_config, outpdf):
check_ocrmypdf(
resources / '3small.pdf',
outdir / 'out.pdf',
outpdf,
'--tesseract-config',
cfg_file,
valid_tess_config,
'--pages',
'1',
)
@pytest.mark.slow # This test sometimes times out in CI
@pytest.mark.parametrize('renderer', RENDERERS)
def test_tesseract_config_invalid(renderer, resources, outdir):
@pytest.fixture
def invalid_tess_config(outdir):
cfg_file = outdir / 'test.cfg'
with cfg_file.open('w') as f:
f.write(
@@ -511,14 +520,19 @@ def test_tesseract_config_invalid(renderer, resources, outdir):
THIS FILE IS INVALID
'''
)
yield cfg_file
@pytest.mark.slow # This test sometimes times out in CI
@pytest.mark.parametrize('renderer', RENDERERS)
def test_tesseract_config_invalid(renderer, resources, invalid_tess_config, outpdf):
p = run_ocrmypdf(
resources / 'ccitt.pdf',
outdir / 'out.pdf',
outpdf,
'--pdf-renderer',
renderer,
'--tesseract-config',
cfg_file,
invalid_tess_config,
)
assert (
"parameter not found" in p.stderr.lower()
@@ -801,6 +815,9 @@ def test_text_curves(resources, outpdf):
info = PdfInfo(outpdf)
assert len(info.pages[0].images) == 0, "added images to the vector PDF"
def test_text_curves_force(resources, outpdf):
with patch('ocrmypdf._pipeline.VECTOR_PAGE_DPI', 100):
check_ocrmypdf(
resources / 'vector.pdf',
outpdf,
@@ -922,3 +939,25 @@ def test_outputtype_none(resources, outtxt):
'tests/plugins/tesseract_noop.py',
)
assert p.returncode == ExitCode.ok
@pytest.fixture
def graph_bad_icc(resources, outdir):
synth_input_file = outdir / 'graph-bad-icc.pdf'
with pikepdf.open(resources / 'graph.pdf') as pdf:
icc = pdf.make_stream(
b'invalid icc profile', N=3, Alternate=pikepdf.Name.DeviceRGB
)
pdf.pages[0].Resources.XObject['/Im0'].ColorSpace = pikepdf.Array(
[pikepdf.Name.ICCBased, icc]
)
pdf.save(synth_input_file)
yield synth_input_file
def test_corrupt_icc(graph_bad_icc, outpdf, caplog):
result = run_ocrmypdf_api(graph_bad_icc, outpdf)
assert result == ExitCode.ok
assert any(
'corrupt or unreadable ICC profile' in rec.message for rec in caplog.records
)
+19 -9
View File
@@ -6,10 +6,10 @@
import datetime
import warnings
from datetime import timezone
from os import fspath
from shutil import copyfile
from unittest.mock import patch
import pikepdf
import pytest
@@ -173,6 +173,19 @@ def test_creation_date_preserved(output_type, resources, infile, outpdf):
assert seconds_between_dates(date_after, datetime.datetime.now(timezone.utc)) < 1000
@pytest.fixture
def libxmp_file_to_dict():
try:
with warnings.catch_warnings():
warnings.simplefilter("ignore", DeprecationWarning)
from libxmp.utils import (
file_to_dict, # pylint: disable=import-outside-toplevel
)
except Exception: # pylint: disable=broad-except
pytest.skip("libxmp not available or libexempi3 not installed")
return file_to_dict
@pytest.mark.parametrize(
'test_file,output_type',
[
@@ -182,15 +195,12 @@ def test_creation_date_preserved(output_type, resources, infile, outpdf):
('3small.pdf', 'pdfa'),
],
)
def test_xml_metadata_preserved(test_file, output_type, resources, outpdf):
def test_xml_metadata_preserved(
libxmp_file_to_dict, test_file, output_type, resources, outpdf
):
input_file = resources / test_file
try:
from libxmp.utils import file_to_dict # pylint: disable=import-outside-toplevel
except Exception: # pylint: disable=broad-except
pytest.skip("libxmp not available or libexempi3 not installed")
before = file_to_dict(str(input_file))
before = libxmp_file_to_dict(str(input_file))
check_ocrmypdf(
input_file,
@@ -202,7 +212,7 @@ def test_xml_metadata_preserved(test_file, output_type, resources, outpdf):
'tests/plugins/tesseract_noop.py',
)
after = file_to_dict(str(outpdf))
after = libxmp_file_to_dict(str(outpdf))
equal_properties = [
'dc:contributor',
+16 -8
View File
@@ -4,23 +4,31 @@
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
import os
import pikepdf
import pytest
from ocrmypdf.exceptions import MissingDependencyError
from .conftest import check_ocrmypdf
@pytest.mark.parametrize('optimize', (0, 3))
@pytest.mark.parametrize('pdfa_level', (1, 2, 3))
def test_pdfa(resources, outpdf, optimize, pdfa_level):
check_ocrmypdf(
resources / 'francais.pdf',
outpdf,
'--plugin',
'tests/plugins/tesseract_noop.py',
f'--output-type=pdfa-{pdfa_level}',
f'--optimize={optimize}',
)
try:
check_ocrmypdf(
resources / 'francais.pdf',
outpdf,
'--plugin',
'tests/plugins/tesseract_noop.py',
f'--output-type=pdfa-{pdfa_level}',
f'--optimize={optimize}',
)
except MissingDependencyError as e:
if 'pngquant' in str(e) and optimize in (2, 3) and os.name == 'nt':
pytest.xfail("pngquant currently not available on Windows")
if pdfa_level in (2, 3):
# PDF/A-2 allows ObjStm
assert b'/ObjStm' in outpdf.read_bytes()