Compare commits

...
17 Commits
Author SHA1 Message Date
James R. Barlow 177349cc84 v14.2.1 release notes 2023-05-23 13:21:46 -07:00
James R. Barlow 070c9772ce Add test to confirm we can find formx optimizable images 2023-05-23 00:33:54 -07:00
James R. Barlow 1bc09045a5 optimize: Add recursion into Form XObjects 2023-05-23 00:07:15 -07:00
James R. Barlow e46a18dd2f Refactor searching for optimizable images 2023-05-22 23:42:10 -07:00
James R. Barlow c64871c2ed pyproject: Make classifiers more generic 2023-05-22 23:37:42 -07:00
James R. Barlow de909fb99a Docker arm64 needs to build pycairo; add its dependencies 2023-05-10 01:31:31 -07:00
James R. Barlow 731b2fc477 Fix interaction with old Python and singledispatch and typing 2023-05-10 00:15:34 -07:00
James R. Barlow 214f6ec759 v14.2.0 release notes 2023-05-09 23:32:57 -07:00
James R. Barlow 080aa4dbd1 imageops: fix rare assert failure 2023-05-09 23:27:08 -07:00
James R. Barlow 7af5dcd4a4 Improve directions about install other languages 2023-05-05 13:35:27 -07:00
James R. Barlow fe9f52fbe7 Remove Ghostscript UserUnit warning
All supported Ghostscript versions have proper UserUnit support.
2023-04-29 00:42:38 -07:00
James R. Barlow fcbdeb8dbe Fix resampling attributeerror when Pillow < 9.1.0 2023-04-25 15:17:45 -07:00
James R. Barlow cb251a8d03 Merge branches 'feature/better-imageops' and 'feature/cleanup-fitz' 2023-04-16 01:41:46 -07:00
James R. Barlow 3731fdfd72 Add --tesseract-downsample-above X 2023-04-16 01:41:19 -07:00
James R. Barlow b2e6a6431e Tidy use of pikepdf.open without with 2023-04-15 20:17:44 -07:00
James R. Barlow 9ff1e56bf6 Tidy test_metadata - importorskip idiom 2023-04-15 17:56:13 -07:00
James R. Barlow 2b30f74fce Improve imageops
Fix issues and add better tests. Require hypothesis.
2023-04-15 17:48:33 -07:00
14 changed files with 297 additions and 152 deletions
+3 -1
View File
@@ -25,7 +25,9 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
libffi-dev \ libffi-dev \
ca-certificates \ ca-certificates \
curl \ curl \
git git \
libcairo2-dev \
pkg-config
# Get the latest pip (Ubuntu version doesn't support manylinux2010) # Get the latest pip (Ubuntu version doesn't support manylinux2010)
RUN \ RUN \
+1
View File
@@ -6,6 +6,7 @@
.venv*/ .venv*/
.tox/ .tox/
.vscode/ .vscode/
.hypothesis/
.ipynb_checkpoints/ .ipynb_checkpoints/
.mypy_cache/ .mypy_cache/
.pytest_cache/ .pytest_cache/
+19
View File
@@ -28,6 +28,25 @@ tagged yet.
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg .. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
v14.2.1
=======
- Fixed :issue:`977`, where images inside Form XObjects were always excluded
from image optimization.
v14.2.0
=======
- Added `--tesseract-downsample-above` to downsample larger images even when
they do not exceed Tesseract's internal limits. This can be used to speed
up OCR, possibly sacrificing accuracy.
- Fixed resampling AttributeError on older Pillow. :issue:`1096`
- Removed an error about using Ghostscript on PDFs with that have the /UserUnit
feature in use. Previously, Ghostscript would fail to process these PDFs,
but in all supported versions it is now supported, so the error is no longer
needed.
- Improved documentation around installing other language packs for Tesseract.
v14.1.0 v14.1.0
======= =======
+3 -6
View File
@@ -37,16 +37,12 @@ classifiers = [
"Intended Audience :: Science/Research", "Intended Audience :: Science/Research",
"Intended Audience :: System Administrators", "Intended Audience :: System Administrators",
"License :: OSI Approved :: Mozilla Public License 2.0 (MPL 2.0)", "License :: OSI Approved :: Mozilla Public License 2.0 (MPL 2.0)",
"Operating System :: MacOS :: MacOS X", "Operating System :: MacOS",
"Operating System :: Microsoft :: Windows :: Windows 10", "Operating System :: Microsoft :: Windows",
"Operating System :: POSIX", "Operating System :: POSIX",
"Operating System :: POSIX :: BSD", "Operating System :: POSIX :: BSD",
"Operating System :: POSIX :: Linux", "Operating System :: POSIX :: Linux",
"Programming Language :: Python :: 3", "Programming Language :: Python :: 3",
"Programming Language :: Python :: 3 :: Only",
"Programming Language :: Python :: 3.8",
"Programming Language :: Python :: 3.9",
"Programming Language :: Python :: 3.10",
"Topic :: Scientific/Engineering :: Image Recognition", "Topic :: Scientific/Engineering :: Image Recognition",
"Topic :: Text Processing :: Indexing", "Topic :: Text Processing :: Indexing",
"Topic :: Text Processing :: Linguistic", "Topic :: Text Processing :: Linguistic",
@@ -69,6 +65,7 @@ docs = ["sphinx", "sphinx-issues", "sphinx-rtd-theme"]
extended_test = ["PyMuPDF==1.19.1"] extended_test = ["PyMuPDF==1.19.1"]
test = [ test = [
"coverage[toml]>=5", "coverage[toml]>=5",
"hypothesis>=6.0.0",
"pytest>=6.0.0", "pytest>=6.0.0",
"pytest-cov>=2.11.1", "pytest-cov>=2.11.1",
"pytest-xdist>=2.2.0", "pytest-xdist>=2.2.0",
-9
View File
@@ -185,15 +185,6 @@ def validate_pdfinfo_options(context: PdfContext) -> None:
"Designer and can only be read by Adobe Acrobat or Adobe Reader." "Designer and can only be read by Adobe Acrobat or Adobe Reader."
) )
raise InputFileError() raise InputFileError()
if pdfinfo.has_userunit and options.output_type.startswith('pdfa'):
log.error(
"This input file uses a PDF feature that is not supported "
"by Ghostscript, so you cannot use --output-type=pdfa for this "
"file. (Specifically, it uses the PDF-1.6 /UserUnit feature to "
"support very large or small page sizes, and Ghostscript cannot "
"output these files.) Use --output-type=pdf instead."
)
raise InputFileError()
if pdfinfo.has_acroform: if pdfinfo.has_acroform:
if options.redo_ocr: if options.redo_ocr:
log.error( log.error(
+10 -2
View File
@@ -61,12 +61,20 @@ def check_options_languages(options: Namespace, ocr_engine_languages: set[str])
return return
missing_languages = options.languages - ocr_engine_languages missing_languages = options.languages - ocr_engine_languages
if missing_languages: if missing_languages:
lang_text = '\n'.join(lang for lang in missing_languages)
msg = ( msg = (
"OCR engine does not have language data for the following " "OCR engine does not have language data for the following "
"requested languages: \n" "requested languages: \n"
f"{lang_text}\n"
"Please install the appropriate language data for your OCR engine.\n"
"\n"
"See the online documentation for instructions:\n"
" https://ocrmypdf.readthedocs.io/en/latest/languages.html\n"
"\n"
"Note: most languages are identified by a 3-digit ISO 639-2 Code.\n"
"For example, English is 'eng', German is 'deu', and Spanish is 'spa'."
"\n"
) )
msg += '\n'.join(lang for lang in missing_languages)
msg += '\nNote: most languages are identified by a 3-digit ISO 639-2 Code'
raise MissingDependencyError(msg) raise MissingDependencyError(msg)
+17 -1
View File
@@ -102,6 +102,20 @@ def add_options(parser):
"of --tesseract-timeout to ensure Tesseract has enough to time." "of --tesseract-timeout to ensure Tesseract has enough to time."
), ),
) )
tess.add_argument(
'--tesseract-downsample-above',
action='store',
type=numeric(int, 100, 32767),
default=32767,
help=(
"Downsample images larger than this size pixel size in either dimension "
"before OCR. --tesseract-downsample-large-images downsamples only when "
"an image exceeds Tesseract's internal limits. This argument causes "
"downsampling to occur when an image exceeds the given size. This may "
"reduce OCR quality, but on large images the most desirable text is "
"usually larger."
),
)
tess.add_argument( tess.add_argument(
'--user-words', '--user-words',
metavar='FILE', metavar='FILE',
@@ -170,10 +184,12 @@ def filter_ocr_image(page: PageContext, image: Image.Image) -> Image.Image:
or more than 2**31 bytes. This function resizes the image to fit within or more than 2**31 bytes. This function resizes the image to fit within
those limits. those limits.
""" """
threshold = min(page.options.tesseract_downsample_above, 32767)
options = page.options options = page.options
if options.tesseract_downsample_large_images: if options.tesseract_downsample_large_images:
size = calculate_downsample( size = calculate_downsample(
image, max_size=(32767, 32767), max_bytes=(2**31) - 1 image, max_size=(threshold, threshold), max_bytes=(2**31) - 1
) )
image = downsample_image(image, size) image = downsample_image(image, size)
return image return image
+96 -37
View File
@@ -6,10 +6,27 @@
from __future__ import annotations from __future__ import annotations
import logging import logging
from math import ceil, floor, sqrt from functools import singledispatch
from math import floor, sqrt
from typing import Optional, Tuple
from PIL import Image from PIL import Image
# Remove this workaround when we require Pillow >= 9.1.0
try:
Resampling = Image.Resampling # type: ignore
except AttributeError:
# Pillow 9 shim
Resampling = Image # type: ignore
# While from __future__ import annotations, we use singledispatch here, which
# does not support annotations. Disable check about using old-style typing
# until Python 3.10, OR when drop singledispatch in ocrmypdf 15.
# ruff: noqa: UP006
# ruff: noqa: UP007
log = logging.getLogger(__name__) log = logging.getLogger(__name__)
@@ -26,14 +43,78 @@ def bytes_per_pixel(mode: str) -> int:
return 4 return 4
@singledispatch
def calculate_downsample( def calculate_downsample(
image: Image.Image, image_size: Tuple[int, int],
bytes_per_pixel: int,
*, *,
max_size: tuple[int, int] | None = None, max_size: Optional[Tuple[int, int]] = None,
max_pixels: int | None = None, max_pixels: Optional[int] = None,
max_bytes: int | None = None, max_bytes: Optional[int] = None,
) -> tuple[int, int]: ) -> Tuple[int, int]:
"""Calculate image size required to downsample an image to fit lmiits. """Calculate image size required to downsample an image to fit limits.
If no limit is exceeded, the input image's size is returned.
Args:
image_size: Dimensions of image.
bytes_per_pixel: Number of bytes per pixel.
max_size: The maximum width and height of the image.
max_pixels: The maximum number of pixels in the image. Some image consumers
limit the total number of pixels as some value other than width*height.
max_bytes: The maximum number of bytes in the image. RGB is counted as 4
bytes; all other modes are counted as 1 byte.
"""
size = image_size
if max_size is not None:
overage = max_size[0] / size[0], max_size[1] / size[1]
size_factor = min(overage)
if size_factor < 1.0:
log.debug("Resizing image to fit image dimensions limit")
size = floor(size[0] * size_factor), floor(size[1] * size_factor)
if size[0] == 0:
size = 1, min(size[1], max_size[1])
elif size[1] == 0:
size = min(size[0], max_size[0]), 1
if max_pixels is not None:
if size[0] * size[1] > max_pixels:
log.debug("Resizing image to fit image pixel limit")
pixels_factor = sqrt(max_pixels / (size[0] * size[1]))
size = floor(size[0] * pixels_factor), floor(size[1] * pixels_factor)
if max_bytes is not None:
bpp = bytes_per_pixel
# stride = bytes per line
stride = size[0] * bpp
height = size[1]
if stride * height > max_bytes:
log.debug("Resizing image to fit image byte size limit")
bytes_factor = sqrt(max_bytes / (stride * height))
scaled_stride = floor(stride * bytes_factor)
scaled_height = floor(height * bytes_factor)
if scaled_stride == 0:
scaled_stride = bpp
scaled_height = min(max_bytes // bpp, scaled_height)
if scaled_height == 0:
scaled_height = 1
scaled_stride = min(max_bytes // scaled_height, scaled_stride)
size = floor(scaled_stride / bpp), scaled_height
return size
@calculate_downsample.register
def _(
image: Image.Image,
arg: None = None,
*,
max_size: Optional[Tuple[int, int]] = None,
max_pixels: Optional[int] = None,
max_bytes: Optional[int] = None,
) -> Tuple[int, int]:
"""Calculate image size required to downsample an image to fit limits.
If no limit is exceeded, the input image's size is returned. If no limit is exceeded, the input image's size is returned.
@@ -45,42 +126,20 @@ def calculate_downsample(
max_bytes: The maximum number of bytes in the image. RGB is counted as 4 max_bytes: The maximum number of bytes in the image. RGB is counted as 4
bytes; all other modes are counted as 1 byte. bytes; all other modes are counted as 1 byte.
""" """
size = image.size return calculate_downsample(
image.size,
if max_size is not None: bytes_per_pixel(image.mode),
major_axis = max(image.size) max_size=max_size,
size_factor = max(max_size) / major_axis max_pixels=max_pixels,
if size_factor < 1.0: max_bytes=max_bytes,
log.debug("Resizing image to fit Tesseract image size limit") )
size = floor(size[0] * size_factor), floor(size[1] * size_factor)
if max_pixels is not None:
if size[0] * size[1] > max_pixels:
log.debug("Resizing image to fit image pixel limit")
pixels_factor = sqrt(max_pixels / (image.size[0] * image.size[1]))
size = floor(size[0] * pixels_factor), floor(size[1] * pixels_factor)
if max_bytes is not None:
bpp = bytes_per_pixel(image.mode)
# stride = bytes per line
stride = size[0] * bpp
height = size[1]
if stride * height > max_bytes:
log.debug("Resizing image to fit image byte size limit")
bytes_factor = sqrt((max_bytes) / (stride * height))
scaled_stride = floor(stride * bytes_factor)
scaled_height = floor(height * bytes_factor)
size = ceil(scaled_stride / bpp), scaled_height
assert (size[0] * bpp * size[1]) <= max_bytes
return size
def downsample_image( def downsample_image(
image: Image.Image, image: Image.Image,
new_size: tuple[int, int], new_size: tuple[int, int],
*, *,
resample_mode: Image.Resampling = Image.Resampling.BICUBIC, resample_mode: Image.Resampling = Resampling.BICUBIC,
reducing_gap: int = 3, reducing_gap: int = 3,
) -> Image.Image: ) -> Image.Image:
"""Downsample an image to fit within the given limits. """Downsample an image to fit within the given limits.
+60 -23
View File
@@ -230,6 +230,65 @@ def extract_image_generic(
return None return None
def _find_image_xrefs_container(
pdf: Pdf,
container: Object,
pageno: int,
include_xrefs: MutableSet[Xref],
exclude_xrefs: MutableSet[Xref],
pageno_for_xref: dict[Xref, int],
depth: int = 0,
):
"""Find all image XRefs in a page or Form XObject and add to the include/exclude sets."""
if depth > 10:
log.warning("Recursion depth exceeded in _find_image_xrefs_page")
return
try:
xobjs = container.Resources.XObject
except AttributeError:
return
for _imname, image in dict(xobjs).items():
if image.objgen[1] != 0:
continue # Ignore images in an incremental PDF
if Name.Subtype in image and image.Subtype == Name.Form:
# Recurse into Form XObjects
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
_find_image_xrefs_container(
pdf,
image,
pageno,
include_xrefs,
exclude_xrefs,
pageno_for_xref,
depth + 1,
)
continue
xref = Xref(image.objgen[0])
if Name.SMask in image:
# Ignore soft masks
smask_xref = Xref(image.SMask.objgen[0])
exclude_xrefs.add(smask_xref)
log.debug(f"xref {smask_xref}: skipping image because it is an SMask")
include_xrefs.add(xref)
log.debug(f"xref {xref}: treating as an optimization candidate")
if xref not in pageno_for_xref:
pageno_for_xref[xref] = pageno
def _find_image_xrefs(pdf: Pdf):
include_xrefs: MutableSet[Xref] = set()
exclude_xrefs: MutableSet[Xref] = set()
pageno_for_xref: dict[Xref, int] = {}
for pageno, page in enumerate(pdf.pages):
_find_image_xrefs_container(
pdf, page, pageno, include_xrefs, exclude_xrefs, pageno_for_xref
)
working_xrefs = include_xrefs - exclude_xrefs
return working_xrefs, pageno_for_xref
def extract_images( def extract_images(
pike: Pdf, pike: Pdf,
root: Path, root: Path,
@@ -250,30 +309,8 @@ def extract_images(
it does a tuple should be returned: (xref, ext) where .ext is the file it does a tuple should be returned: (xref, ext) where .ext is the file
extension. extract_fn must also extract the file it finds interesting. extension. extract_fn must also extract the file it finds interesting.
""" """
include_xrefs: MutableSet[Xref] = set()
exclude_xrefs: MutableSet[Xref] = set()
pageno_for_xref = {}
errors = 0 errors = 0
for pageno, page in enumerate(pike.pages): working_xrefs, pageno_for_xref = _find_image_xrefs(pike)
try:
xobjs = page.Resources.XObject
except AttributeError:
continue
for _imname, image in dict(xobjs).items():
if image.objgen[1] != 0:
continue # Ignore images in an incremental PDF
xref = Xref(image.objgen[0])
if Name.SMask in image:
# Ignore soft masks
smask_xref = Xref(image.SMask.objgen[0])
exclude_xrefs.add(smask_xref)
log.debug(f"xref {smask_xref}: skipping image because it is an SMask")
include_xrefs.add(xref)
log.debug(f"xref {xref}: treating as an optimization candidate")
if xref not in pageno_for_xref:
pageno_for_xref[xref] = pageno
working_xrefs = include_xrefs - exclude_xrefs
for xref in working_xrefs: for xref in working_xrefs:
image = pike.get_object((xref, 0)) image = pike.get_object((xref, 0))
try: try:
+22
View File
@@ -3,6 +3,8 @@
from __future__ import annotations from __future__ import annotations
import hypothesis.strategies as st
from hypothesis import given
from PIL import Image from PIL import Image
from ocrmypdf.imageops import bytes_per_pixel, calculate_downsample, downsample_image from ocrmypdf.imageops import bytes_per_pixel, calculate_downsample, downsample_image
@@ -23,6 +25,26 @@ def test_calculate_downsample():
assert calculate_downsample(im, max_bytes=100000) == (100, 100) assert calculate_downsample(im, max_bytes=100000) == (100, 100)
@given(
st.one_of(st.just("RGB"), st.just('L')),
st.integers(min_value=1, max_value=100000),
st.integers(min_value=1, max_value=100000),
st.integers(min_value=64, max_value=100000),
st.integers(min_value=64, max_value=100000),
st.integers(min_value=64 * 64, max_value=1000000),
)
def test_calculate_downsample_hypothesis(mode, im_w, im_h, max_x, max_y, max_bytes):
result = calculate_downsample(
(im_w, im_h),
bytes_per_pixel(mode),
max_size=(max_x, max_y),
max_bytes=max_bytes,
)
assert result[0] <= max_x
assert result[1] <= max_y
assert result[0] * result[1] * bytes_per_pixel(mode) <= max_bytes
def test_downsample_image(): def test_downsample_image():
im = Image.new('RGB', (100, 100)) im = Image.new('RGB', (100, 100))
im.info['dpi'] = (300, 300) im.info['dpi'] = (300, 300)
+40 -50
View File
@@ -22,16 +22,9 @@ from ocrmypdf.pdfinfo import PdfInfo
from .conftest import check_ocrmypdf, run_ocrmypdf from .conftest import check_ocrmypdf, run_ocrmypdf
try:
import fitz
except ImportError:
fitz = None
@pytest.mark.parametrize("output_type", ['pdfa', 'pdf']) @pytest.mark.parametrize("output_type", ['pdfa', 'pdf'])
def test_preserve_docinfo(output_type, resources, outpdf): def test_preserve_docinfo(output_type, resources, outpdf):
pdf_before = pikepdf.open(resources / 'graph.pdf')
output = check_ocrmypdf( output = check_ocrmypdf(
resources / 'graph.pdf', resources / 'graph.pdf',
outpdf, outpdf,
@@ -40,14 +33,13 @@ def test_preserve_docinfo(output_type, resources, outpdf):
'--plugin', '--plugin',
'tests/plugins/tesseract_noop.py', 'tests/plugins/tesseract_noop.py',
) )
with pikepdf.open(resources / 'graph.pdf') as pdf_before, pikepdf.open(
pdf_after = pikepdf.open(output) output
) as pdf_after:
for key in ('/Title', '/Author'): for key in ('/Title', '/Author'):
assert pdf_before.docinfo[key] == pdf_after.docinfo[key] assert pdf_before.docinfo[key] == pdf_after.docinfo[key]
pdfa_info = file_claims_pdfa(str(output))
pdfa_info = file_claims_pdfa(str(output)) assert pdfa_info['output'] == output_type
assert pdfa_info['output'] == output_type
@pytest.mark.parametrize("output_type", ['pdfa', 'pdf']) @pytest.mark.parametrize("output_type", ['pdfa', 'pdf'])
@@ -71,19 +63,17 @@ def test_override_metadata(output_type, resources, outpdf):
assert p.returncode == ExitCode.ok, p.stderr assert p.returncode == ExitCode.ok, p.stderr
before = pikepdf.open(input_file) with pikepdf.open(input_file) as before, pikepdf.open(outpdf) as after:
after = pikepdf.open(outpdf) assert after.docinfo.Title == german, after.docinfo
assert after.docinfo.Author == chinese, after.docinfo
assert after.docinfo.get('/Keywords', '') == ''
assert after.docinfo.Title == german, after.docinfo before_date = decode_pdf_date(str(before.docinfo.CreationDate))
assert after.docinfo.Author == chinese, after.docinfo after_date = decode_pdf_date(str(after.docinfo.CreationDate))
assert after.docinfo.get('/Keywords', '') == '' assert before_date == after_date
before_date = decode_pdf_date(str(before.docinfo.CreationDate)) pdfa_info = file_claims_pdfa(outpdf)
after_date = decode_pdf_date(str(after.docinfo.CreationDate)) assert pdfa_info['output'] == output_type
assert before_date == after_date
pdfa_info = file_claims_pdfa(outpdf)
assert pdfa_info['output'] == output_type
def test_high_unicode(resources, no_outpdf): def test_high_unicode(resources, no_outpdf):
@@ -106,10 +96,10 @@ def test_high_unicode(resources, no_outpdf):
assert p.returncode == ExitCode.bad_args, p.stderr assert p.returncode == ExitCode.bad_args, p.stderr
@pytest.mark.skipif(not fitz, reason="test uses fitz")
@pytest.mark.parametrize('ocr_option', ['--skip-text', '--force-ocr']) @pytest.mark.parametrize('ocr_option', ['--skip-text', '--force-ocr'])
@pytest.mark.parametrize('output_type', ['pdf', 'pdfa']) @pytest.mark.parametrize('output_type', ['pdf', 'pdfa'])
def test_bookmarks_preserved(output_type, ocr_option, resources, outpdf): def test_bookmarks_preserved(output_type, ocr_option, resources, outpdf):
fitz = pytest.importorskip('fitz')
input_file = resources / 'toc.pdf' input_file = resources / 'toc.pdf'
before_toc = fitz.Document(str(input_file)).get_toc() before_toc = fitz.Document(str(input_file)).get_toc()
@@ -147,23 +137,24 @@ def test_creation_date_preserved(output_type, resources, infile, outpdf):
'tests/plugins/tesseract_noop.py', 'tests/plugins/tesseract_noop.py',
) )
pdf_before = pikepdf.open(input_file) with pikepdf.open(input_file) as pdf_before, pikepdf.open(outpdf) as pdf_after:
pdf_after = pikepdf.open(outpdf) before = pdf_before.trailer.get('/Info', {})
after = pdf_after.trailer.get('/Info', {})
before = pdf_before.trailer.get('/Info', {}) if not before:
after = pdf_after.trailer.get('/Info', {}) assert after.get('/CreationDate', '') != ''
else:
# We expect that the creation date stayed the same
date_before = decode_pdf_date(str(before['/CreationDate']))
date_after = decode_pdf_date(str(after['/CreationDate']))
assert seconds_between_dates(date_before, date_after) < 1000
if not before: # We expect that the modified date is quite recent
assert after.get('/CreationDate', '') != '' date_after = decode_pdf_date(str(after['/ModDate']))
else: assert (
# We expect that the creation date stayed the same seconds_between_dates(date_after, datetime.datetime.now(timezone.utc))
date_before = decode_pdf_date(str(before['/CreationDate'])) < 1000
date_after = decode_pdf_date(str(after['/CreationDate'])) )
assert seconds_between_dates(date_before, date_after) < 1000
# We expect that the modified date is quite recent
date_after = decode_pdf_date(str(after['/ModDate']))
assert seconds_between_dates(date_after, datetime.datetime.now(timezone.utc)) < 1000
@pytest.fixture @pytest.fixture
@@ -280,10 +271,9 @@ def test_kodak_toc(resources, outpdf):
'tests/plugins/tesseract_noop.py', 'tests/plugins/tesseract_noop.py',
) )
p = pikepdf.open(outpdf) with pikepdf.open(outpdf) as p:
if pikepdf.Name.First in p.Root.Outlines:
if pikepdf.Name.First in p.Root.Outlines: assert isinstance(p.Root.Outlines.First, pikepdf.Dictionary)
assert isinstance(p.Root.Outlines.First, pikepdf.Dictionary)
def test_metadata_fixup_warning(resources, outdir, caplog): def test_metadata_fixup_warning(resources, outdir, caplog):
@@ -301,10 +291,10 @@ def test_metadata_fixup_warning(resources, outdir, caplog):
assert record.levelname != 'WARNING', "Unexpected warning" assert record.levelname != 'WARNING', "Unexpected warning"
# Now add some metadata that will not be copyable # Now add some metadata that will not be copyable
graph = pikepdf.open(outdir / 'graph.pdf') with pikepdf.open(outdir / 'graph.pdf') as graph:
with graph.open_metadata() as meta: with graph.open_metadata() as meta:
meta['prism2:publicationName'] = 'OCRmyPDF Test' meta['prism2:publicationName'] = 'OCRmyPDF Test'
graph.save(outdir / 'graph_mod.pdf') graph.save(outdir / 'graph_mod.pdf')
context = PdfContext( context = PdfContext(
options, outdir, outdir / 'graph_mod.pdf', None, get_plugin_manager([]) options, outdir, outdir / 'graph_mod.pdf', None, get_plugin_manager([])
+18 -10
View File
@@ -100,14 +100,14 @@ def test_jbig2_lossy(lossy, resources, outpdf):
check_ocrmypdf(*args) check_ocrmypdf(*args)
pdf = pikepdf.open(outpdf) with pikepdf.open(outpdf) as pdf:
pim = pikepdf.PdfImage(next(iter(pdf.pages[0].images.values()))) pim = pikepdf.PdfImage(next(iter(pdf.pages[0].images.values())))
assert pim.filters[0] == '/JBIG2Decode' assert pim.filters[0] == '/JBIG2Decode'
if lossy: if lossy:
assert '/JBIG2Globals' in pim.decode_parms[0] assert '/JBIG2Globals' in pim.decode_parms[0]
else: else:
assert len(pim.decode_parms) == 0 assert len(pim.decode_parms) == 0
@needs_pngquant @needs_pngquant
@@ -134,9 +134,9 @@ def test_flate_to_jbig2(resources, outdir):
'tests/plugins/tesseract_noop.py', 'tests/plugins/tesseract_noop.py',
) )
pdf = pikepdf.open(outdir / 'out.pdf') with pikepdf.open(outdir / 'out.pdf') as pdf:
pim = pikepdf.PdfImage(next(iter(pdf.pages[0].images.values()))) pim = pikepdf.PdfImage(next(iter(pdf.pages[0].images.values())))
assert pim.filters[0] == '/JBIG2Decode' assert pim.filters[0] == '/JBIG2Decode'
@needs_pngquant @needs_pngquant
@@ -204,3 +204,11 @@ def test_group3(resources, outdir):
assert ( assert (
opt.extract_image_filter(pdf, outdir, im, im.objgen[0]) is None opt.extract_image_filter(pdf, outdir, im, im.objgen[0]) is None
), "Group 3 should be disallowed" ), "Group 3 should be disallowed"
def test_find_formx(resources, outdir):
with pikepdf.open(resources / 'formxobject.pdf') as pdf:
working, pagenos = opt._find_image_xrefs(pdf)
assert len(working) == 1
xref = next(iter(working))
assert pagenos[xref] == 0
+5 -5
View File
@@ -237,11 +237,11 @@ def test_rotate_page_level(image_angle, page_angle, resources, outdir):
**IMG2PDF_KWARGS, **IMG2PDF_KWARGS,
) )
mempdf.seek(0) mempdf.seek(0)
pike = pikepdf.open(mempdf) with pikepdf.open(mempdf) as pdf:
pike.pages[0].Rotate = page_angle pdf.pages[0].Rotate = page_angle
target = outdir / f'{prefix}_{image_angle}_{page_angle}.pdf' target = outdir / f'{prefix}_{image_angle}_{page_angle}.pdf'
pike.save(target) pdf.save(target)
return target return target
reference = make_rotate_test('ref', 0, 0) reference = make_rotate_test('ref', 0, 0)
test = make_rotate_test('test', image_angle, page_angle) test = make_rotate_test('test', image_angle, page_angle)
+3 -8
View File
@@ -20,18 +20,13 @@ def poster(resources):
return resources / 'poster.pdf' return resources / 'poster.pdf'
def test_userunit_ghostscript_fails(poster, no_outpdf, caplog): @pytest.mark.parametrize("mode", ['pdf', 'pdfa'])
result = run_ocrmypdf_api(poster, no_outpdf, '--output-type=pdfa') def test_userunit_pdf_passes(mode, poster, outpdf):
assert result == ExitCode.input_file
assert 'not supported by Ghostscript' in caplog.text
def test_userunit_pdf_passes(poster, outpdf):
before = PdfInfo(poster) before = PdfInfo(poster)
check_ocrmypdf( check_ocrmypdf(
poster, poster,
outpdf, outpdf,
'--output-type=pdf', f'--output-type={mode}',
'--plugin', '--plugin',
'tests/plugins/tesseract_cache.py', 'tests/plugins/tesseract_cache.py',
) )