Compare commits

...
15 Commits
Author SHA1 Message Date
James R. Barlow 47ef1914d4 v11.4.4 release notes 2021-01-01 01:39:24 -08:00
James R. Barlow df157552f3 Make ocrmypdf.ocr take a threading lock 2021-01-01 01:37:09 -08:00
James R. Barlow 0b3a526049 Partial fix crash on 'userunit' None (#700)
Our method of getting data from pdfminer would silently consume a StopIteration
if pdfminer returned no processed pages, leading to odd error message.

We improve an error from pdfminer properly, and returning a more
descriptive error of our own.

It would be possible for ocrmypdf to repair the file before sending it to
pdfminer, but this seems to be rare enough that we won't do that yet.
2021-01-01 01:11:32 -08:00
James R. Barlow 1e80d412fa tesseract: fix typing of some optional arguments 2021-01-01 00:46:00 -08:00
James R. Barlow df6e106203 concurrent: simplify results loop 2021-01-01 00:44:46 -08:00
James R. Barlow bd0f005861 tests: tag tests that need pngquant, jbig2enc 2020-12-30 01:58:57 -08:00
James R. Barlow 6ba4b7b3f3 ci: temporarily disable pngquant on Windows
Looks like a packaging error, choco complains of bad hashes.
2020-12-30 01:40:56 -08:00
James R. Barlow 2c11349ee8 Merge branch 'master' of github.com:jbarlow83/OCRmyPDF 2020-12-29 21:40:46 -08:00
James R. Barlow b0afef09ef v11.4.3 release notes 2020-12-29 21:40:35 -08:00
James R. Barlow 72fa347c38 tests: skip metadata test for two pikepdf versions that warn incorrectly 2020-12-29 01:47:52 -08:00
James R. Barlow 96d68c2413 pipeline: refactor metadata_fixup 2020-12-29 01:47:32 -08:00
James R. Barlow babc76fa74 tests: assert that most patched functions are called
We were not actually checking if functions we patched we called when
expected.
2020-12-28 23:58:33 -08:00
Tim GatesandGitHub dc06990e5d docs: fix simple typo, instsalled -> installed (#704)
There is a small typo in docs/installation.rst.

Should read `installed` rather than `instsalled`.
2020-12-28 15:28:34 -08:00
James R. Barlow 0ff0d2f8d1 Remove PDF/A overprint debug message
Since we currently log all of a process's output at debug it's
redundant to log this separate message.
2020-12-27 16:19:05 -08:00
James R. Barlow 81602cf420 Fix test not patching properly after Ghostscript polling change 2020-12-27 16:01:50 -08:00
25 changed files with 180 additions and 77 deletions
+1 -1
View File
@@ -32,7 +32,7 @@ stages:
choco install --yes --no-progress --pre tesseract
choco install --yes --no-progress python3
choco install --yes --no-progress ghostscript
choco install --yes --no-progress pngquant
# choco install --yes --no-progress pngquant
displayName: "Install system packages"
- pwsh: |
refreshenv
+5
View File
@@ -56,6 +56,11 @@ Programs that call ``ocrmypdf.ocr()`` should also install a SIGBUS signal
handler (except on Windows), to raise an exception if access to a memory
mapped file fails. OCRmyPDF may use memory mapping.
``ocrmypdf.ocr()`` will take a threading lock to prevent multiple runs of itself
in the same Python interpreter process. This is not thread-safe, because of how
OCRmyPDF's plugins and Python's library import system work. If you need to parallelize
OCRmyPDF, use processes.
.. warning::
On Windows and macOS, the script that calls ``ocrmypdf.ocr()`` must be
+1 -1
View File
@@ -637,7 +637,7 @@ Installing with Python pip
OCRmyPDF is delivered by PyPI because it is a convenient way to install
the latest version. However, PyPI and ``pip`` cannot address the fact
that ``ocrmypdf`` depends on certain non-Python system libraries and
programs being instsalled.
programs being installed.
For best results, first install `your platform's
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
+16
View File
@@ -12,6 +12,22 @@ may be unreliable. Use the API to depend on precise behavior.
The public API may be useful in scripts that launch OCRmyPDF processes or that
wish to use some of its features for working with PDFs.
v11.4.4
=======
- Fixed ``AttributeError: 'NoneType' object has no attribute 'userunit'``, issue #700,
related to OCRmyPDF not properly forwarded an error message from pdfminer.six.
- Adjusted typing of some arguments.
- ``ocrmypdf.ocr`` now takes a ``threading.Lock`` for reasons outlined in the
documentation.
v11.4.3
=======
- Removed a redundant debug message.
- Test suite now asserts that most patched functions are called when they should be.
- Test suite now skips a test that fails on two particular versions of piekpdf.
v11.4.2
=======
+5 -9
View File
@@ -109,15 +109,11 @@ def exec_progress_pool(
)
try:
results = pool.imap_unordered(task, task_arguments)
while True:
try:
result = results.next()
if task_finished:
task_finished(result, pbar)
else:
pbar.update()
except StopIteration:
break
for result in results:
if task_finished:
task_finished(result, pbar)
else:
pbar.update()
except KeyboardInterrupt:
# Terminate pool so we exit instantly
pool.terminate()
+2 -8
View File
@@ -254,6 +254,8 @@ def generate_pdfa(
raise SubprocessOutputError('Ghostscript PDF/A rendering failed') from e
else:
stderr = p.stderr
# If there is an error we log the whole stderr, except for filtering
# duplicates.
if _gs_error_reported(stderr):
last_part = None
repcount = 0
@@ -266,11 +268,3 @@ def generate_pdfa(
else:
repcount += 1
last_part = part
elif 'overprint mode not set' in stderr:
# Unless someone is going to print PDF/A documents on a
# magical sRGB printer I can't see the removal of overprinting
# being a problem....
log.debug(
"Ghostscript had to remove PDF 'overprinting' from the "
"input file to complete PDF/A conversion. "
)
+3 -3
View File
@@ -14,7 +14,7 @@ from collections import namedtuple
from os import fspath
from pathlib import Path
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
from typing import List
from typing import List, Optional
from PIL import Image
@@ -118,7 +118,7 @@ def get_languages():
return set(lang.strip() for lang in rest)
def tess_base_args(langs: List[str], engine_mode: int) -> List[str]:
def tess_base_args(langs: List[str], engine_mode: Optional[int]) -> List[str]:
args = ['tesseract']
if langs:
args.extend(['-l', '+'.join(langs)])
@@ -127,7 +127,7 @@ def tess_base_args(langs: List[str], engine_mode: int) -> List[str]:
return args
def get_orientation(input_file: Path, engine_mode: int, timeout: float):
def get_orientation(input_file: Path, engine_mode: Optional[int], timeout: float):
args_tesseract = tess_base_args(['osd'], engine_mode) + [
'--psm',
'0',
+10 -12
View File
@@ -756,19 +756,17 @@ def metadata_fixup(working_file: Path, context: PdfContext):
if 'xmp:CreateDate' not in meta:
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
# Ghostscript likes to set title to Untitled if omitted from input.
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
# and the XMP Spec do not make this recommendation.
if meta.get('dc:title') == 'Untitled':
with original.open_metadata(
set_pikepdf_as_editor=False, update_docinfo=False
) as original_meta:
if 'dc:title' not in original_meta:
with original.open_metadata(
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
) as meta_original:
if meta.get('dc:title') == 'Untitled':
# Ghostscript likes to set title to Untitled if omitted from input.
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
# and the XMP Spec do not make this recommendation.
if 'dc:title' not in meta_original:
del meta['dc:title']
meta_original = original.open_metadata()
missing = set(meta_original.keys()) - set(meta.keys())
report_on_metadata(missing)
missing = set(meta_original.keys()) - set(meta.keys())
report_on_metadata(missing)
pdf.save(
output_file,
+16 -7
View File
@@ -8,6 +8,7 @@
import logging
import os
import sys
import threading
from enum import IntEnum
from io import IOBase
from pathlib import Path
@@ -30,6 +31,8 @@ except ModuleNotFoundError:
StrPath = Union[os.PathLike, AnyStr]
PathOrIO = Union[BinaryIO, StrPath]
_api_lock = threading.Lock()
class Verbosity(IntEnum):
"""Verbosity level for configure_logging."""
@@ -306,12 +309,18 @@ def ocr( # pylint: disable=unused-argument
parser = get_parser()
create_options_kwargs['parser'] = parser
plugin_manager = get_plugin_manager(plugins)
plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
if 'verbose' in kwargs:
warn("ocrmypdf.ocr(verbose=) is ignored. Use ocrmypdf.configure_logging().")
with _api_lock:
# We can't allow multiple ocrmypdf.ocr() threads to run in parallel, because
# they might install different plugins, and generally speaking we have areas
# of code that use global state.
options = create_options(**create_options_kwargs)
check_options(options, plugin_manager)
return run_pipeline(options=options, plugin_manager=plugin_manager, api=True)
plugin_manager = get_plugin_manager(plugins)
plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
if 'verbose' in kwargs:
warn("ocrmypdf.ocr(verbose=) is ignored. Use ocrmypdf.configure_logging().")
options = create_options(**create_options_kwargs)
check_options(options, plugin_manager)
return run_pipeline(options=options, plugin_manager=plugin_manager, api=True)
+3 -1
View File
@@ -22,7 +22,7 @@ import pikepdf
from pikepdf import Object, Pdf, PdfMatrix
from ocrmypdf._concurrent import exec_progress_pool
from ocrmypdf.exceptions import EncryptedPdfError
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap
from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes
@@ -598,6 +598,8 @@ def _pdf_pageinfo_concurrent(
def update_pageinfo(result, pbar):
page = result
if not page:
raise InputFileError("Could read a page in the PDF")
pages[page.pageno] = page
pbar.update()
+8 -3
View File
@@ -21,7 +21,7 @@ from pdfminer.pdffont import PDFSimpleFont, PDFUnicodeNotDefined
from pdfminer.pdfpage import PDFPage
from pdfminer.utils import bbox2str, matrix2str
from ocrmypdf.exceptions import EncryptedPdfError
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
STRIP_NAME = re.compile(r'[0-9]+')
@@ -236,8 +236,13 @@ def get_page_analysis(infile, pageno, pscript5_mode):
try:
with Path(infile).open('rb') as f:
page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0)
interp.process_page(next(page))
page_iter = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0)
page = next(page_iter, None)
if page is None:
raise InputFileError(
f"pdfminer could not process page {pageno} (counting from 0)."
)
interp.process_page(page)
except PDFTextExtractionNotAllowed as e:
raise EncryptedPdfError() from e
finally:
+7 -5
View File
@@ -23,21 +23,22 @@ from unittest.mock import patch
from ocrmypdf import hookimpl
from ocrmypdf.builtin_plugins import ghostscript
from ocrmypdf.subprocess import run
from ocrmypdf.subprocess import run_polling_stderr
elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1
not permitted in PDF/A-2, overprint mode not set"""
def run_append_stderr(*args, **kwargs):
proc = run(*args, **kwargs)
proc.stderr = b'\n'.join([proc.stderr, elision_warning.encode('utf-8')])
proc = run_polling_stderr(*args, **kwargs)
proc.stderr += '\n' + elision_warning + '\n'
return proc
@hookimpl
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
with patch('ocrmypdf._exec.ghostscript.run', new=run_append_stderr):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
mock.side_effect = run_append_stderr
ghostscript.generate_pdfa(
pdf_pages=pdf_pages,
pdfmark=pdfmark,
@@ -47,4 +48,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
pdfa_part=pdfa_part,
progressbar_class=None,
)
return output_file
mock.assert_called_once()
return output_file
+3 -1
View File
@@ -39,7 +39,8 @@ def run_rig_args(args, **kwargs):
@hookimpl
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=run_rig_args):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
mock.side_effect = run_rig_args
ghostscript.generate_pdfa(
pdf_pages=pdf_pages,
pdfmark=pdfmark,
@@ -49,4 +50,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
pdfa_part=pdfa_part,
progressbar_class=None,
)
mock.assert_called()
return output_file
+3 -1
View File
@@ -44,7 +44,8 @@ def rasterize_pdf_page(
rotation=None,
filter_vector=False,
) -> Path:
with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail):
with patch('ocrmypdf._exec.ghostscript.run') as mock:
mock.side_effect = raise_gs_fail
ghostscript.rasterize_pdf_page(
input_file=input_file,
output_file=output_file,
@@ -55,4 +56,5 @@ def rasterize_pdf_page(
rotation=rotation,
filter_vector=filter_vector,
)
mock.assert_called()
return output_file
+3 -1
View File
@@ -34,7 +34,8 @@ def raise_gs_fail(*args, **kwargs):
@hookimpl
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=raise_gs_fail):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
mock.side_effect = raise_gs_fail
ghostscript.generate_pdfa(
pdf_pages=pdf_pages,
pdfmark=pdfmark,
@@ -44,4 +45,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
pdfa_part=pdfa_part,
progressbar_class=None,
)
mock.assert_called()
return output_file
+11 -2
View File
@@ -26,6 +26,7 @@ that is not UTF-8 compatible, so we are forced to check that we can convert it
and present it to the user.
"""
from contextlib import contextmanager
from subprocess import CalledProcessError
from unittest.mock import patch
@@ -42,17 +43,25 @@ def bad_utf8(*args, **kwargs):
)
@contextmanager
def patch_tesseract_run():
with patch('ocrmypdf._exec.tesseract.run') as mock:
mock.side_effect = bad_utf8
yield
mock.assert_called()
class BadUtf8OcrEngine(TesseractOcrEngine):
@staticmethod
def generate_hocr(input_file, output_hocr, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8):
with patch_tesseract_run():
TesseractOcrEngine.generate_hocr(
input_file, output_hocr, output_text, options
)
@staticmethod
def generate_pdf(input_file, output_pdf, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8):
with patch_tesseract_run():
TesseractOcrEngine.generate_pdf(
input_file, output_pdf, output_text, options
)
+12 -3
View File
@@ -19,6 +19,7 @@
# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
from contextlib import contextmanager
from subprocess import CalledProcessError
from unittest.mock import patch
@@ -35,22 +36,30 @@ def raise_size_exception(*args, **kwargs):
)
@contextmanager
def patch_tesseract_run():
with patch('ocrmypdf._exec.tesseract.run') as mock:
mock.side_effect = raise_size_exception
yield
mock.assert_called()
class BigImageErrorOcrEngine(TesseractOcrEngine):
@staticmethod
def get_orientation(input_file, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
with patch_tesseract_run():
return TesseractOcrEngine.get_orientation(input_file, options)
@staticmethod
def generate_hocr(input_file, output_hocr, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
with patch_tesseract_run():
TesseractOcrEngine.generate_hocr(
input_file, output_hocr, output_text, options
)
@staticmethod
def generate_pdf(input_file, output_pdf, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
with patch_tesseract_run():
TesseractOcrEngine.generate_pdf(
input_file, output_pdf, output_text, options
)
+12 -3
View File
@@ -20,6 +20,7 @@
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
import signal
from contextlib import contextmanager
from subprocess import CalledProcessError
from unittest.mock import patch
@@ -37,22 +38,30 @@ def raise_crash(*args, **kwargs):
)
@contextmanager
def patch_tesseract_run():
with patch('ocrmypdf._exec.tesseract.run') as mock:
mock.side_effect = raise_crash
yield
mock.assert_called()
class CrashOcrEngine(TesseractOcrEngine):
@staticmethod
def get_orientation(input_file, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
with patch_tesseract_run():
return TesseractOcrEngine.get_orientation(input_file, options)
@staticmethod
def generate_hocr(input_file, output_hocr, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
with patch_tesseract_run():
TesseractOcrEngine.generate_hocr(
input_file, output_hocr, output_text, options
)
@staticmethod
def generate_pdf(input_file, output_pdf, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
with patch_tesseract_run():
TesseractOcrEngine.generate_pdf(
input_file, output_pdf, output_text, options
)
+5
View File
@@ -36,12 +36,17 @@ class TestSafeSymlink:
def test_no_cpu_count(monkeypatch):
invoked = False
def cpu_count_raises():
nonlocal invoked
invoked = True
raise NotImplementedError()
monkeypatch.setattr(multiprocessing, 'cpu_count', cpu_count_raises)
with pytest.warns(expected_warning=UserWarning):
assert helpers.available_cpu_count() == 1
assert invoked, "Patched function called during test"
def test_deprecated():
+2 -1
View File
@@ -65,11 +65,12 @@ def test_cmyk_no_icc(caplog, resources, no_outpdf):
def test_img2pdf_fails(resources, no_outpdf):
with patch(
'ocrmypdf._pipeline.img2pdf.convert', side_effect=img2pdf.ImageOpenError()
):
) as mock:
rc = run_ocrmypdf_api(
resources / 'baiona_gray.png', no_outpdf, '--image-dpi', '200'
)
assert rc == ocrmypdf.ExitCode.input_file
mock.assert_called()
def test_jpeg_in_jpeg_out(resources, outpdf):
+4 -1
View File
@@ -306,6 +306,9 @@ def test_kodak_toc(resources, outpdf):
assert isinstance(p.Root.Outlines.First, pikepdf.Dictionary)
@pytest.mark.skipif(
pikepdf.__version__ in ('2.2.2', '2.2.3'), reason="Raises wrong warning"
)
def test_metadata_fixup_warning(resources, outdir, caplog):
options = get_parser().parse_args(
args=['--output-type', 'pdfa-2', 'graph.pdf', 'out.pdf']
@@ -318,7 +321,7 @@ def test_metadata_fixup_warning(resources, outdir, caplog):
)
metadata_fixup(working_file=outdir / 'graph.pdf', context=context)
for record in caplog.records:
assert record.levelname != 'WARNING'
assert record.levelname != 'WARNING', "Unexpected warning"
# Now add some metadata that will not be copyable
graph = pikepdf.open(outdir / 'graph.pdf')
+17 -7
View File
@@ -21,7 +21,15 @@ from ocrmypdf.helpers import Resolution
check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=e1101
needs_pngquant = pytest.mark.skipif(
not pngquant.available(), reason="pngquant not installed"
)
needs_jbig2enc = pytest.mark.skipif(
not jbig2enc.available(), reason="jbig2enc not installed"
)
@needs_pngquant
@pytest.mark.parametrize('pdf', ['multipage.pdf', 'palette.pdf'])
def test_basic(resources, pdf, outpdf):
infile = resources / pdf
@@ -30,6 +38,7 @@ def test_basic(resources, pdf, outpdf):
assert 0.98 * Path(outpdf).stat().st_size <= Path(infile).stat().st_size
@needs_pngquant
def test_mono_not_inverted(resources, outdir):
infile = resources / '2400dpi.pdf'
opt.main(infile, outdir / 'out.pdf', level=3)
@@ -45,7 +54,7 @@ def test_mono_not_inverted(resources, outdir):
assert im.getpixel((0, 0)) == 255, "Expected white background"
@pytest.mark.skipif(not pngquant.available(), reason='need pngquant')
@needs_pngquant
def test_jpg_png_params(resources, outpdf):
check_ocrmypdf(
resources / 'crom.png',
@@ -63,7 +72,7 @@ def test_jpg_png_params(resources, outpdf):
)
@pytest.mark.skipif(not jbig2enc.available(), reason='need jbig2enc')
@needs_jbig2enc
@pytest.mark.parametrize('lossy', [False, True])
def test_jbig2_lossy(lossy, resources, outpdf):
args = [
@@ -95,10 +104,8 @@ def test_jbig2_lossy(lossy, resources, outpdf):
assert len(pim.decode_parms) == 0
@pytest.mark.skipif(
not jbig2enc.available() or not pngquant.available(),
reason='need jbig2enc and pngquant',
)
@needs_pngquant
@needs_jbig2enc
def test_flate_to_jbig2(resources, outdir):
# This test requires an image that pngquant is capable of converting to
# to 1bpp - so use an existing 1bpp image, convert up, confirm it can
@@ -126,6 +133,7 @@ def test_flate_to_jbig2(resources, outdir):
assert pim.filters[0] == '/JBIG2Decode'
@needs_pngquant
def test_multiple_pngs(resources, outdir):
with Path.open(outdir / 'in.pdf', 'wb') as inpdf:
img2pdf.convert(
@@ -141,7 +149,8 @@ def test_multiple_pngs(resources, outdir):
draw.rectangle((0, 0, im.width, im.height), fill=128)
im.save(output_file)
with patch('ocrmypdf.optimize.pngquant.quantize', new=mockquant):
with patch('ocrmypdf.optimize.pngquant.quantize') as mock:
mock.side_effect = mockquant
check_ocrmypdf(
outdir / 'in.pdf',
outdir / 'out.pdf',
@@ -155,6 +164,7 @@ def test_multiple_pngs(resources, outdir):
'--plugin',
'tests/plugins/tesseract_noop.py',
)
mock.assert_called()
with pikepdf.open(outdir / 'in.pdf') as inpdf, pikepdf.open(
outdir / 'out.pdf'
+19
View File
@@ -15,7 +15,11 @@ from PIL import Image
from reportlab.pdfgen.canvas import Canvas
from ocrmypdf import pdfinfo
from ocrmypdf.exceptions import InputFileError
from ocrmypdf.pdfinfo import Colorspace, Encoding
from ocrmypdf.pdfinfo.layout import PDFPage
run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api
# pylint: disable=protected-access
@@ -179,3 +183,18 @@ def test_stack_abuse():
with pytest.warns(None):
with pytest.raises(RuntimeError):
pdfinfo.info._interpret_contents(stream)
def test_pages_issue700(monkeypatch, resources):
def get_no_pages(*args, **kwargs):
return iter([])
monkeypatch.setattr(PDFPage, 'get_pages', get_no_pages)
with pytest.raises(InputFileError, match="pdfminer"):
pdfinfo.PdfInfo(
resources / 'cardinal.pdf',
detailed_analysis=True,
progbar=False,
max_workers=1,
)
+6 -4
View File
@@ -28,11 +28,12 @@ def test_no_unpaper(resources, no_outpdf):
output = fspath(no_outpdf)
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version:
mock_unpaper_version.side_effect = FileNotFoundError("unpaper")
with patch("ocrmypdf._exec.unpaper.version") as mock:
mock.side_effect = FileNotFoundError("unpaper")
with pytest.raises(MissingDependencyError):
check_options(options, pm)
mock.assert_called()
def test_old_unpaper(resources, no_outpdf):
@@ -40,11 +41,12 @@ def test_old_unpaper(resources, no_outpdf):
output = fspath(no_outpdf)
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version:
mock_unpaper_version.return_value = '0.5'
with patch("ocrmypdf._exec.unpaper.version") as mock:
mock.return_value = '0.5'
with pytest.raises(MissingDependencyError):
check_options(options, pm)
mock.assert_called()
@pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
+6 -3
View File
@@ -188,18 +188,20 @@ def test_language_warning(caplog):
caplog.set_level(logging.DEBUG)
with patch(
'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8')
):
) as mock:
vd.check_options_languages(opts, {'eng'})
assert opts.languages == {'eng'}
assert '' in caplog.text
mock.assert_called_once()
opts = make_opts(language=None)
with patch(
'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8')
):
) as mock:
vd.check_options_languages(opts, {'eng'})
assert opts.languages == {'eng'}
assert 'assuming --language' in caplog.text
mock.assert_called_once()
def test_version_comparison():
@@ -265,7 +267,8 @@ def test_pagesegmode_warning(caplog):
def test_two_languages():
with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True):
with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True) as mock:
vd._check_options(
*make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'}
)
mock.assert_called()