Compare commits

...
23 Commits
Author SHA1 Message Date
James R. Barlow 3dfde479e2 The world is not ready for := 2022-01-26 00:16:51 -08:00
James R. Barlow aea1862644 v13.3.0 release notes 2022-01-25 23:50:48 -08:00
James R. Barlow 3b406112d0 ghostscript: improve test coverage of error cases 2022-01-25 23:45:47 -08:00
James R. Barlow fcc4c2d371 ghostscript: improve error message if image cannot be opened 2022-01-25 23:12:51 -08:00
James R. Barlow 3de18ed612 tesseract: account for more tesseract 5 output differences 2022-01-25 00:19:49 -08:00
James R. Barlow 93cca42e20 optimize: don't try to optimize an image we can't save 2022-01-25 00:19:23 -08:00
James R. Barlow 2d0ac4707c Use better img2pdf settings where possible while supporting old versions
Fixes #894
2022-01-14 11:55:54 -08:00
James R. Barlow 7d208175cf unpaper: refactoring 2022-01-11 10:57:54 -08:00
James R. Barlow ea69e868ed unpaper: issue warning if image too large to clean 2022-01-11 10:44:38 -08:00
James R. Barlow beea603ab3 Revert "docs: add sphinx-panels"
This reverts commit 7966192d6e.
2022-01-04 11:44:20 -08:00
James R. Barlow 7966192d6e docs: add sphinx-panels 2022-01-04 11:43:28 -08:00
Anton GladkyandGitHub 5acbd7a252 Wrap exception on non-CMYK images into the log warning (#881) 2021-12-29 01:15:49 -08:00
Krasimir NedelchevandGitHub aed955ca8c Fix typo (#882) 2021-12-27 15:36:19 -08:00
James R. Barlow 298bdb8690 v13.2.0 release notes 2021-12-19 11:40:23 -08:00
James R. Barlow 1a58abcc6a concurrency: fix extra update of progressbar 2021-12-19 11:34:09 -08:00
James R. Barlow dbfceba020 Standardize ghostscript version default 2021-12-18 02:01:42 -08:00
James R. Barlow 0faa618c3c Replace deprecated distutils with packaging.version 2021-12-18 01:57:59 -08:00
James R. Barlow 7035002c03 Order of operations suppressed detailed Ghostscript missing error message 2021-12-18 01:27:39 -08:00
James R. Barlow f8fadaef41 _windows: remove use of deprecated distutils 2021-12-13 20:45:54 -08:00
James R. Barlow ee21bf9ef6 Update cache 2021-12-13 20:45:30 -08:00
James R. Barlow 190ca81951 v13.1.1 release notes 2021-12-10 21:49:04 -08:00
James R. Barlow d48254d477 Fix issue with attempting to deskew a blank page on Tesseract 5
Closes #868
2021-12-10 21:48:09 -08:00
James R. Barlow 1ec2ccca14 docs: add warning about multiproc on macOS 2021-12-10 17:42:50 -08:00
29 changed files with 284 additions and 111 deletions
+8
View File
@@ -68,6 +68,14 @@ OCRmyPDF, use processes.
not take at least one of these steps, process semantics will prevent
OCRmyPDF from working correctly.
.. warning::
On macOS with Python 3.7, you must call
:func:`multiprocessing.set_start_method("spawn")`. Without this, multiprocessing
will be unstable. From the command line, OCRmyPDF does this automatically,
but as an API user you must do this. See Python bpo-33725 for details.
Python 3.8+ also resolve this automatically.
Logging
-------
+30 -5
View File
@@ -10,13 +10,38 @@ that is, output messages may be improved at any release level, so parsing them
may be unreliable. Use the API to depend on precise behavior.
The public API may be useful in scripts that launch OCRmyPDF processes or that
wish to use some of its features for working with PDFs.
wish to use some of its features for working with PDFs..
.. note::
v13.3.0
=======
Python 3.6 reaches end of life on December 23, 2021. We have already ended support
for Python 3.6 but might release fixes for critical issues if necessary before that
date.
- Made a harmless but "scary" exception after failing to optimize an image less scary.
- Added a warning if a page image is too large for unpaper to clean. The image is
passed through without cleaning. This is due to a hard-coded limitation in a
C library used by unpaper so it cannot be rectified easily.
- We now use better default settings when calling img2pdf.
- We no longer try to optimize images that we failed to save in certain situations.
- We now account for some differences in text output from Tesseract 5 that differs
from Tesseract 4.
- Better handling of Ghostscript producing empty images when attempting to rasterize
page images.
v13.2.0
=======
- Removed all runtime uses of distutils since it is deprecated in standard library. We
previous used ``distutils.version`` to examine version numbers of dependencies
at run time, and now use ``packaging.version`` for this. This is a new
dependency.
- Fixed an error message advising the user that Ghostscript was not installed being
suppressed when this condition actually happens.
- Fixed an issue with incorrect page number and totals being displayed in the progress
bar. This was purely a display/presentation issue. :issue:`876`.
v13.1.1
=======
- Fixed issue with attempting to deskew a blank page on Tesseract 5. :issue:`868`.
v13.1.0
=======
+1
View File
@@ -48,6 +48,7 @@ install_requires =
Pillow>=8.2.0
coloredlogs>=14.0 # strictly optional
img2pdf>=0.3.0,<0.5 # pure Python
packaging>=20
pdfminer.six!=20200720,>=20191110,<=20211012
pikepdf>=4.0.0
pluggy>=0.13.0,<2
+25 -18
View File
@@ -18,7 +18,7 @@ from shutil import which
from subprocess import PIPE, CalledProcessError
from typing import Optional
from PIL import Image
from PIL import Image, UnidentifiedImageError
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
from ocrmypdf.helpers import Resolution
@@ -71,7 +71,8 @@ def jpeg_passthrough_available() -> bool:
def _gs_error_reported(stream) -> bool:
return True if re.search(r'error', stream, flags=re.IGNORECASE) else False
match = re.search(r'error', stream, flags=re.IGNORECASE)
return bool(match)
def rasterize_pdf(
@@ -124,20 +125,27 @@ def rasterize_pdf(
if _gs_error_reported(stderr):
log.error(stderr)
with Image.open(BytesIO(p.stdout)) as im:
if rotation is not None:
log.debug("Rotating output by %i", rotation)
# rotation is a clockwise angle and Image.ROTATE_* is
# counterclockwise so this cancels out the rotation
if rotation == 90:
im = im.transpose(Image.ROTATE_90)
elif rotation == 180:
im = im.transpose(Image.ROTATE_180)
elif rotation == 270:
im = im.transpose(Image.ROTATE_270)
if rotation % 180 == 90:
page_dpi = page_dpi.flip_axis()
im.save(fspath(output_file), dpi=page_dpi)
try:
with Image.open(BytesIO(p.stdout)) as im:
if rotation is not None:
log.debug("Rotating output by %i", rotation)
# rotation is a clockwise angle and Image.ROTATE_* is
# counterclockwise so this cancels out the rotation
if rotation == 90:
im = im.transpose(Image.ROTATE_90)
elif rotation == 180:
im = im.transpose(Image.ROTATE_180)
elif rotation == 270:
im = im.transpose(Image.ROTATE_270)
if rotation % 180 == 90:
page_dpi = page_dpi.flip_axis()
im.save(fspath(output_file), dpi=page_dpi)
except UnidentifiedImageError:
log.error(
f"Ghostscript (using {raster_device} at {raster_dpi} dpi) produced "
"an invalid page image file."
)
raise
class GhostscriptFollower:
@@ -161,8 +169,7 @@ class GhostscriptFollower:
)
return
else:
m = self.re_page.match(line.strip())
if m:
if self.re_page.match(line.strip()):
self.progressbar.update()
+52 -21
View File
@@ -9,13 +9,13 @@
import logging
import re
from distutils.version import StrictVersion
from math import pi
from os import fspath
from pathlib import Path
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
from typing import Dict, Iterator, List, Optional
from packaging.version import Version
from PIL import Image
from ocrmypdf.exceptions import (
@@ -60,25 +60,54 @@ class TesseractLoggerAdapter(logging.LoggerAdapter):
return '[tesseract] %s' % (msg), kwargs
class TesseractVersion(StrictVersion):
version_re = re.compile(
r'''
^(\d+) \. (\d+) (\. (\d+))? # groups: 1/major, 2/minor, 3/[skip], 4/patch
[-]? # optional hyphen separator
(?: ((?:alpha|beta|rc|dev)\d*)? [.\-\ ]? (\d+)? )? # 5/prerelease, 6/prerelease_num
(?:(?:-\d+)?-g[0-9a-f]+)? # untagged git version
$
''',
re.VERBOSE | re.ASCII,
TESSERACT_VERSION_PATTERN = r"""
v?
(?:
(?:(?P<epoch>[0-9]+)!)? # epoch
(?P<release>[0-9]+(?:\.[0-9]+)*) # release segment
(?P<pre> # pre-release
[-_\.]?
(?P<pre_l>(a|b|c|rc|alpha|beta|pre|preview))
[-_\.]?
(?P<pre_n>[0-9]+)?
)?
(?P<post> # post release
(?:-(?P<post_n1>[0-9]+))
|
(?:
[-_\.]?
(?P<post_l>post|rev|r)
[-_\.]?
(?P<post_n2>[0-9]+)?
)
)?
(?P<dev> # dev release
[-_\.]?
(?P<dev_l>dev)
[-_\.]?
(?P<dev_n>[0-9]+)?
)?
(?P<date>
[-_\.]
(?:20[0-9][0-9] [0-1][0-9] [0-3][0-9]) # yyyy mm dd
)?
(?P<gitcount>
[-_\.]?
[0-9]+
)?
(?P<gitcommit>
[-_\.]?
g[0-9a-f]{2,10}
)?
)
(?:\+(?P<local>[a-z0-9]+(?:[-_\.][a-z0-9]+)*))? # local version
"""
def parse(self, vstring):
try:
super().parse(vstring)
except TypeError as e:
if 'int() argument must be a string' in str(e):
super().parse(vstring + '-0')
class TesseractVersion(Version):
_regex = re.compile(
r"^\s*" + TESSERACT_VERSION_PATTERN + r"\s*$", re.VERBOSE | re.IGNORECASE
)
def version() -> str:
@@ -200,7 +229,9 @@ def get_deskew(
except CalledProcessError as e:
tesseract_log_output(e.stdout)
tesseract_log_output(e.stderr)
if b'Empty page!!' in e.output: # Not enough info for a skew angle
if b'Empty page!!' in e.output or (
e.output == b'' and e.returncode == 1
): # Not enough info for a skew angle - Tess 4 and 5 return different errors
return 0.0
raise SubprocessOutputError() from e
@@ -313,7 +344,7 @@ def generate_hocr(
_generate_null_hocr(output_hocr, output_text, input_file)
except CalledProcessError as e:
tesseract_log_output(e.output)
if b'Image too large' in e.output:
if b'Image too large' in e.output or b'Empty page!!' in e.output:
_generate_null_hocr(output_hocr, output_text, input_file)
return
@@ -385,7 +416,7 @@ def generate_pdf(
use_skip_page(output_pdf, output_text)
except CalledProcessError as e:
tesseract_log_output(e.output)
if b'Image too large' in e.output:
if b'Image too large' in e.output or b'Empty page!!' in e.output:
use_skip_page(output_pdf, output_text)
return
raise SubprocessOutputError() from e
+67 -37
View File
@@ -13,6 +13,7 @@
import logging
import os
import shlex
from contextlib import contextmanager
from decimal import Decimal
from pathlib import Path
from subprocess import PIPE, STDOUT
@@ -22,60 +23,84 @@ from typing import List, Optional, Tuple, Union
from PIL import Image
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
from ocrmypdf.subprocess import get_version
from ocrmypdf.subprocess import run as external_run
from ocrmypdf.subprocess import get_version, run
UNPAPER_IMAGE_PIXEL_LIMIT = 256 * 1024 * 1024
DecFloat = Union[Decimal, float]
log = logging.getLogger(__name__)
class UnpaperImageTooLargeError(Exception):
def __init__(
self,
w,
h,
message="Image with size {}x{} is too large for cleaning with 'unpaper'.",
):
self.w = w
self.h = h
self.message = message.format(w, h)
super().__init__(self.message)
def version() -> str:
return get_version('unpaper')
def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]:
def _convert_image(im: Image.Image) -> Tuple[Image.Image, bool, str]:
SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'}
with Image.open(input_file) as im:
im_modified = False
if im.mode not in SUFFIXES:
log.info("Converting image to other colorspace")
try:
if im.mode == 'P' and len(im.getcolors()) == 2:
im = im.convert(mode='1')
else:
im = im.convert(mode='RGB')
except OSError as e:
raise MissingDependencyError(
"Could not convert image with type " + im.mode
) from e
else:
im_modified = True
im_modified = False
if im.mode not in SUFFIXES:
log.info("Converting image to other colorspace")
try:
suffix = SUFFIXES[im.mode]
except KeyError:
if im.mode == 'P' and len(im.getcolors()) == 2:
im = im.convert(mode='1')
else:
im = im.convert(mode='RGB')
except OSError as e:
raise MissingDependencyError(
"Failed to convert image to a supported format."
) from None
if im_modified or input_file.suffix != '.pnm':
input_pnm = tmpdir / 'input.pnm'
im.save(input_pnm, format='PPM')
"Could not convert image with type " + im.mode
) from e
else:
# No changes, PNG input, just use the file we already have
input_pnm = input_file
output_pnm = tmpdir / f'output{suffix}'
return input_pnm, output_pnm
im_modified = True
try:
suffix = SUFFIXES[im.mode]
except KeyError:
raise MissingDependencyError(
"Failed to convert image to a supported format."
) from None
return im, im_modified, suffix
def run(
@contextmanager
def _setup_unpaper_io(input_file: Path) -> Tuple[Path, Path, Path]:
with Image.open(input_file) as im:
if im.width * im.height >= UNPAPER_IMAGE_PIXEL_LIMIT:
raise UnpaperImageTooLargeError(w=im.width, h=im.height)
im, im_modified, suffix = _convert_image(im)
with TemporaryDirectory() as tmpdir:
tmppath = Path(tmpdir)
if im_modified or input_file.suffix != '.pnm':
input_pnm = tmppath / 'input.pnm'
im.save(input_pnm, format='PPM')
else:
# No changes, PNG input, just use the file we already have
input_pnm = input_file
output_pnm = tmppath / f'output{suffix}'
yield input_pnm, output_pnm, tmppath
def run_unpaper(
input_file: Path, output_file: Path, *, dpi: DecFloat, mode_args: List[str]
) -> None:
args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args
with TemporaryDirectory() as tmpdir:
input_pnm, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file)
with _setup_unpaper_io(input_file) as (input_pnm, output_pnm, tmpdir):
# To prevent any shenanigans from accepting arbitrary parameters in
# --unpaper-args, we:
# 1) run with cwd set to a tmpdir with only unpaper's files
@@ -84,7 +109,7 @@ def run(
# This should ensure that a user cannot clobber some other file with
# their unpaper arguments (whether intentionally or otherwise)
args_unpaper.extend([os.fspath(input_pnm), os.fspath(output_pnm)])
external_run(
run(
args_unpaper,
close_fds=True,
check=True,
@@ -117,7 +142,7 @@ def clean(
*,
dpi: DecFloat,
unpaper_args: Optional[List[str]] = None,
):
) -> Path:
default_args = [
'--layout',
'none',
@@ -131,4 +156,9 @@ def clean(
]
if not unpaper_args:
unpaper_args = default_args
run(input_file, output_file, dpi=dpi, mode_args=unpaper_args)
try:
run_unpaper(input_file, output_file, dpi=dpi, mode_args=unpaper_args)
return output_file
except UnpaperImageTooLargeError as e:
log.warning(str(e))
return input_file
+5 -6
View File
@@ -18,7 +18,7 @@ from typing import Dict, Iterable, Optional
import img2pdf
import pikepdf
from pikepdf.models.metadata import encode_pdf_date
from PIL import Image, ImageColor, ImageDraw
from PIL import Image, ImageDraw
from ocrmypdf._concurrent import Executor
from ocrmypdf._exec import unpaper
@@ -32,7 +32,7 @@ from ocrmypdf.exceptions import (
PriorOcrFoundError,
UnsupportedImageFormatError,
)
from ocrmypdf.helpers import Resolution, safe_symlink
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
from ocrmypdf.hocrtransform import HocrTransform
from ocrmypdf.optimize import optimize
from ocrmypdf.pdfa import generate_pdfa_ps
@@ -98,8 +98,8 @@ def triage_image_file(input_file, output_file, options):
img2pdf.convert(
os.fspath(input_file),
layout_fun=layout_fun,
with_pdfrw=False,
outputstream=outf,
**IMG2PDF_KWARGS,
)
log.info("Successfully converted to PDF, processing...")
except img2pdf.ImageOpenError as e:
@@ -494,13 +494,12 @@ def preprocess_deskew(input_file: Path, page_context: PageContext):
def preprocess_clean(input_file: Path, page_context: PageContext):
output_file = page_context.get_path('pp_clean.png')
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
unpaper.clean(
return unpaper.clean(
input_file,
output_file,
dpi=dpi.x,
unpaper_args=page_context.options.unpaper_args,
)
return output_file
def create_ocr_image(image: Path, page_context: PageContext):
@@ -613,7 +612,7 @@ def create_pdf_page_from_image(
layout_fun = img2pdf.get_layout_fun(pagesize)
img2pdf.convert(
imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf
imfile, layout_fun=layout_fun, outputstream=pdf, **IMG2PDF_KWARGS
)
log.debug('convert done')
@@ -134,7 +134,6 @@ class StandardExecutor(Executor):
for future in as_completed(futures):
result = future.result()
task_finished(result, pbar)
pbar.update()
except KeyboardInterrupt:
# Terminate pool so we exit instantly
executor.shutdown(wait=False, cancel_futures=True)
+2 -2
View File
@@ -18,13 +18,13 @@ log = logging.getLogger(__name__)
@hookimpl
def check_options(options):
gs_version = ghostscript.version()
check_external_program(
program='gs',
package='ghostscript',
version_checker=gs_version,
version_checker=ghostscript.version,
need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports
)
gs_version = ghostscript.version()
if gs_version in ('9.24', '9.51'):
raise MissingDependencyError(
f"Ghostscript {gs_version} contains serious regressions and is not "
+1 -1
View File
@@ -142,7 +142,7 @@ Online documentation is located at:
'output_file',
metavar="output_pdf",
help="Output searchable PDF file (or '-' to write to standard output). "
"Existing files will be ovewritten. If same as input file, the "
"Existing files will be overwritten. If same as input file, the "
"input file will be updated only if processing is successful.",
)
parser.add_argument(
+11
View File
@@ -19,10 +19,21 @@ from math import isclose, isfinite
from pathlib import Path
from typing import Any, Sequence
import img2pdf
import pikepdf
from packaging.version import Version
log = logging.getLogger(__name__)
if Version(img2pdf.__version__) < Version('0.4.0'):
IMG2PDF_KWARGS = dict(without_pdfw=True)
elif Version(img2pdf.__version__) < Version('0.4.3'):
IMG2PDF_KWARGS = dict(engine=img2pdf.Engine.pikepdf)
else:
IMG2PDF_KWARGS = dict(
engine=img2pdf.Engine.pikepdf, rotation=img2pdf.Rotation.ifvalid
)
class Resolution(namedtuple('Resolution', ('x', 'y'))):
"""The number of pixels per inch in each 2D direction.
+7 -3
View File
@@ -41,7 +41,7 @@ from ocrmypdf._concurrent import Executor, SerialExecutor
from ocrmypdf._exec import jbig2enc, pngquant
from ocrmypdf._jobcontext import PdfContext
from ocrmypdf.exceptions import OutputFileAccessError
from ocrmypdf.helpers import safe_symlink
from ocrmypdf.helpers import IMG2PDF_KWARGS, safe_symlink
log = logging.getLogger(__name__)
@@ -200,7 +200,11 @@ def extract_image_generic(
elif not pim.indexed and pim.colorspace in pim.SIMPLE_COLORSPACES:
# An optimization opportunity here, not currently taken, is directly
# generating a PNG from compressed data
pim.as_pil_image().save(png_name(root, xref))
try:
pim.as_pil_image().save(png_name(root, xref))
except NotImplementedError:
log.warning("PDF contains an atypical image that cannot be optimized.")
return None
return XrefExt(xref, '.png')
elif (
not pim.indexed
@@ -450,7 +454,7 @@ def transcode_jpegs(
def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool:
output = filename.with_suffix('.png.pdf')
with output.open('wb') as f:
img2pdf.convert(fspath(filename), outputstream=f)
img2pdf.convert(fspath(filename), outputstream=f, **IMG2PDF_KWARGS)
with Pdf.open(output) as pdf_image:
foreign_image = next(iter(pdf_image.pages[0].images.values()))
+5 -4
View File
@@ -13,12 +13,13 @@ import re
import sys
from collections.abc import Mapping
from contextlib import suppress
from distutils.version import LooseVersion, Version
from functools import lru_cache
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
from subprocess import run as subprocess_run
from typing import Callable, Optional, Type, Union
from packaging.version import Version
from ocrmypdf.exceptions import MissingDependencyError
# pylint: disable=logging-format-interpolation
@@ -266,11 +267,11 @@ def check_external_program(
*,
program: str,
package: str,
version_checker: Union[str, Callable],
version_checker: Callable,
need_version: str,
required_for: Optional[str] = None,
recommended=False,
version_parser: Type[Version] = LooseVersion,
version_parser: Type[Version] = Version,
):
"""Check for required version of external program and raise exception if not.
@@ -291,7 +292,7 @@ def check_external_program(
try:
if callable(version_checker):
found_version = version_checker()
else:
else: # deprecated
found_version = version_checker
except (CalledProcessError, FileNotFoundError, MissingDependencyError):
_error_missing_program(program, package, required_for, recommended)
+14 -2
View File
@@ -8,7 +8,6 @@ import logging
import os
import shutil
import sys
from distutils.version import LooseVersion
from itertools import chain
from pathlib import Path
from typing import Any, Callable, Iterable, Iterator, Set, Tuple, TypeVar
@@ -23,6 +22,17 @@ log = logging.getLogger(__name__)
T = TypeVar('T')
def ghostscript_version_key(s: str) -> Tuple[int, int, int]:
"""Compare Ghostscript version numbers."""
try:
release = [int(elem) for elem in s.split('.', maxsplit=3)]
while len(release) < 3:
release.append(0)
return (release[0], release[1], release[2])
except ValueError:
return (0, 0, 0)
def registry_enum(
key: winreg.HKEYType, enum_fn: Callable[[winreg.HKEYType, int], T]
) -> Iterator[T]:
@@ -51,7 +61,9 @@ def registry_path_ghostscript(env=None) -> Iterator[Path]:
with winreg.OpenKey(
winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Artifex\GPL Ghostscript"
) as k:
latest_gs = max(registry_subkeys(k), key=LooseVersion, default='0')
latest_gs = max(
registry_subkeys(k), key=ghostscript_version_key, default=(0, 0, 0)
)
with winreg.OpenKey(
winreg.HKEY_LOCAL_MACHINE, fr"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
) as k:
+2
View File
@@ -80,3 +80,5 @@
{"tesseract_version": "4.1.1", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
{"tesseract_version": "4.1.1", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
{"tesseract_version": "4.1.1", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__--oem__1__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "--oem", "1", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
{"tesseract_version": "5.0.0", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__thresholding_method=1__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "-c", "thresholding_method=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
{"tesseract_version": "5.0.0", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__thresholding_method=2__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "-c", "thresholding_method=2", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
+20 -1
View File
@@ -6,11 +6,13 @@
import logging
import subprocess
from decimal import Decimal
from unittest.mock import patch
import pikepdf
import pytest
from PIL import Image
from PIL import Image, UnidentifiedImageError
from ocrmypdf._exec.ghostscript import rasterize_pdf
from ocrmypdf.exceptions import ExitCode
@@ -124,3 +126,20 @@ def test_ghostscript_feature_elision(resources, outpdf):
'--plugin',
'tests/plugins/gs_feature_elision.py',
)
def test_rasterize_pdf_errors(resources, no_outpdf, caplog):
with patch('ocrmypdf._exec.ghostscript.run') as mock:
# ghostscript can produce
mock.return_value = subprocess.CompletedProcess(
['fakegs'], returncode=0, stdout=b'', stderr=b'error this is an error'
)
with pytest.raises(UnidentifiedImageError):
rasterize_pdf(
resources / 'francais.pdf',
no_outpdf,
raster_device='pngmono',
raster_dpi=Resolution(100, 100),
)
assert "this is an error" in caplog.text
assert "invalid page image file" in caplog.text
+2 -2
View File
@@ -17,7 +17,7 @@ from PIL import Image, ImageDraw
from ocrmypdf import optimize as opt
from ocrmypdf._exec import jbig2enc, pngquant
from ocrmypdf._exec.ghostscript import rasterize_pdf
from ocrmypdf.helpers import Resolution
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution
from .conftest import check_ocrmypdf
@@ -139,8 +139,8 @@ def test_multiple_pngs(resources, outdir):
img2pdf.convert(
fspath(resources / 'baiona_colormapped.png'),
fspath(resources / 'baiona_gray.png'),
with_pdfrw=False,
outputstream=inpdf,
**IMG2PDF_KWARGS,
)
def mockquant(input_file, output_file, *_args):
+2 -2
View File
@@ -17,7 +17,7 @@ from reportlab.pdfgen.canvas import Canvas
from ocrmypdf import pdfinfo
from ocrmypdf.exceptions import InputFileError
from ocrmypdf.helpers import Resolution
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution
from ocrmypdf.pdfinfo import Colorspace, Encoding
from ocrmypdf.pdfinfo.layout import PDFPage
@@ -67,9 +67,9 @@ def test_single_page_image(eight_by_eight, outpdf):
img2pdf.convert(
bio,
producer="img2pdf",
with_pdfrw=False,
layout_fun=layout_fun,
outputstream=f,
**IMG2PDF_KWARGS,
)
info = pdfinfo.PdfInfo(outpdf)
+5
View File
@@ -40,6 +40,11 @@ def test_deskew(resources, outdir):
assert -0.5 < skew_angle < 0.5, "Deskewing failed"
def test_deskew_blank_page(resources, outpdf):
# Tesseract doesn't like blank pages - make sure we can get through
check_ocrmypdf(resources / 'blank.pdf', outpdf, '--deskew')
@pytest.mark.xfail(reason="remove background disabled")
def test_remove_background(resources, outdir):
# Ensure the input image does not contain pure white/black
+2 -1
View File
@@ -18,7 +18,7 @@ from reportlab.pdfgen.canvas import Canvas
from ocrmypdf._exec import ghostscript
from ocrmypdf._plugin_manager import get_plugin_manager
from ocrmypdf.helpers import Resolution
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution
from ocrmypdf.pdfinfo import PdfInfo
from .conftest import check_ocrmypdf, run_ocrmypdf
@@ -230,6 +230,7 @@ def test_rotate_page_level(image_angle, page_angle, resources, outdir):
memimg.read(),
layout_fun=img2pdf.get_fixed_dpi_layout_fun((200, 200)),
outputstream=mempdf,
**IMG2PDF_KWARGS,
)
mempdf.seek(0)
pike = pikepdf.open(mempdf)
+23 -5
View File
@@ -5,19 +5,24 @@
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
import logging
from os import fspath
from unittest.mock import patch
import pytest
from PIL import Image
from ocrmypdf._exec import unpaper
from ocrmypdf._plugin_manager import get_parser_options_plugins
from ocrmypdf._validation import check_options
from ocrmypdf.exceptions import ExitCode, MissingDependencyError
from .conftest import check_ocrmypdf, have_unpaper, run_ocrmypdf
from .conftest import check_ocrmypdf, have_unpaper, ocrmypdf_exec, run_ocrmypdf
# pylint: disable=redefined-outer-name
needs_unpaper = pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
def test_no_unpaper(resources, no_outpdf):
input_ = fspath(resources / "c02-22.pdf")
@@ -45,7 +50,7 @@ def test_old_unpaper(resources, no_outpdf):
mock.assert_called()
@pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
@needs_unpaper
def test_clean(resources, outpdf):
check_ocrmypdf(
resources / "skew.pdf",
@@ -56,7 +61,7 @@ def test_clean(resources, outpdf):
)
@pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
@needs_unpaper
def test_unpaper_args_valid(resources, outpdf):
check_ocrmypdf(
resources / "skew.pdf",
@@ -69,7 +74,7 @@ def test_unpaper_args_valid(resources, outpdf):
)
@pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
@needs_unpaper
def test_unpaper_args_invalid_filename(resources, outpdf):
p = run_ocrmypdf(
resources / "skew.pdf",
@@ -84,7 +89,7 @@ def test_unpaper_args_invalid_filename(resources, outpdf):
assert p.returncode == ExitCode.bad_args
@pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
@needs_unpaper
def test_unpaper_args_invalid(resources, outpdf):
p = run_ocrmypdf(
resources / "skew.pdf",
@@ -98,3 +103,16 @@ def test_unpaper_args_invalid(resources, outpdf):
# Can't tell difference between unpaper choking on bad arguments or some
# other unpaper failure
assert p.returncode == ExitCode.child_process_error
@needs_unpaper
def test_unpaper_image_too_big(resources, outdir, caplog):
with patch('ocrmypdf._exec.unpaper.UNPAPER_IMAGE_PIXEL_LIMIT', 42):
infile = resources / 'crom.png'
unpaper.clean(infile, outdir / 'out.png', dpi=300) == infile
assert any(
'too large for cleaning' in rec.message
for rec in caplog.get_records('call')
if rec.levelno == logging.WARNING
)