Compare commits

...
10 Commits
Author SHA1 Message Date
James R. Barlow bd0f005861 tests: tag tests that need pngquant, jbig2enc 2020-12-30 01:58:57 -08:00
James R. Barlow 6ba4b7b3f3 ci: temporarily disable pngquant on Windows
Looks like a packaging error, choco complains of bad hashes.
2020-12-30 01:40:56 -08:00
James R. Barlow 2c11349ee8 Merge branch 'master' of github.com:jbarlow83/OCRmyPDF 2020-12-29 21:40:46 -08:00
James R. Barlow b0afef09ef v11.4.3 release notes 2020-12-29 21:40:35 -08:00
James R. Barlow 72fa347c38 tests: skip metadata test for two pikepdf versions that warn incorrectly 2020-12-29 01:47:52 -08:00
James R. Barlow 96d68c2413 pipeline: refactor metadata_fixup 2020-12-29 01:47:32 -08:00
James R. Barlow babc76fa74 tests: assert that most patched functions are called
We were not actually checking if functions we patched we called when
expected.
2020-12-28 23:58:33 -08:00
Tim GatesandGitHub dc06990e5d docs: fix simple typo, instsalled -> installed (#704)
There is a small typo in docs/installation.rst.

Should read `installed` rather than `instsalled`.
2020-12-28 15:28:34 -08:00
James R. Barlow 0ff0d2f8d1 Remove PDF/A overprint debug message
Since we currently log all of a process's output at debug it's
redundant to log this separate message.
2020-12-27 16:19:05 -08:00
James R. Barlow 81602cf420 Fix test not patching properly after Ghostscript polling change 2020-12-27 16:01:50 -08:00
18 changed files with 112 additions and 54 deletions
+1 -1
View File
@@ -32,7 +32,7 @@ stages:
choco install --yes --no-progress --pre tesseract
choco install --yes --no-progress python3
choco install --yes --no-progress ghostscript
choco install --yes --no-progress pngquant
# choco install --yes --no-progress pngquant
displayName: "Install system packages"
- pwsh: |
refreshenv
+1 -1
View File
@@ -637,7 +637,7 @@ Installing with Python pip
OCRmyPDF is delivered by PyPI because it is a convenient way to install
the latest version. However, PyPI and ``pip`` cannot address the fact
that ``ocrmypdf`` depends on certain non-Python system libraries and
programs being instsalled.
programs being installed.
For best results, first install `your platform's
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
+7
View File
@@ -12,6 +12,13 @@ may be unreliable. Use the API to depend on precise behavior.
The public API may be useful in scripts that launch OCRmyPDF processes or that
wish to use some of its features for working with PDFs.
v11.4.3
=======
- Removed a redundant debug message.
- Test suite now asserts that most patched functions are called when they should be.
- Test suite now skips a test that fails on two particular versions of piekpdf.
v11.4.2
=======
+2 -8
View File
@@ -254,6 +254,8 @@ def generate_pdfa(
raise SubprocessOutputError('Ghostscript PDF/A rendering failed') from e
else:
stderr = p.stderr
# If there is an error we log the whole stderr, except for filtering
# duplicates.
if _gs_error_reported(stderr):
last_part = None
repcount = 0
@@ -266,11 +268,3 @@ def generate_pdfa(
else:
repcount += 1
last_part = part
elif 'overprint mode not set' in stderr:
# Unless someone is going to print PDF/A documents on a
# magical sRGB printer I can't see the removal of overprinting
# being a problem....
log.debug(
"Ghostscript had to remove PDF 'overprinting' from the "
"input file to complete PDF/A conversion. "
)
+10 -12
View File
@@ -756,19 +756,17 @@ def metadata_fixup(working_file: Path, context: PdfContext):
if 'xmp:CreateDate' not in meta:
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
# Ghostscript likes to set title to Untitled if omitted from input.
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
# and the XMP Spec do not make this recommendation.
if meta.get('dc:title') == 'Untitled':
with original.open_metadata(
set_pikepdf_as_editor=False, update_docinfo=False
) as original_meta:
if 'dc:title' not in original_meta:
with original.open_metadata(
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
) as meta_original:
if meta.get('dc:title') == 'Untitled':
# Ghostscript likes to set title to Untitled if omitted from input.
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
# and the XMP Spec do not make this recommendation.
if 'dc:title' not in meta_original:
del meta['dc:title']
meta_original = original.open_metadata()
missing = set(meta_original.keys()) - set(meta.keys())
report_on_metadata(missing)
missing = set(meta_original.keys()) - set(meta.keys())
report_on_metadata(missing)
pdf.save(
output_file,
+7 -5
View File
@@ -23,21 +23,22 @@ from unittest.mock import patch
from ocrmypdf import hookimpl
from ocrmypdf.builtin_plugins import ghostscript
from ocrmypdf.subprocess import run
from ocrmypdf.subprocess import run_polling_stderr
elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1
not permitted in PDF/A-2, overprint mode not set"""
def run_append_stderr(*args, **kwargs):
proc = run(*args, **kwargs)
proc.stderr = b'\n'.join([proc.stderr, elision_warning.encode('utf-8')])
proc = run_polling_stderr(*args, **kwargs)
proc.stderr += '\n' + elision_warning + '\n'
return proc
@hookimpl
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
with patch('ocrmypdf._exec.ghostscript.run', new=run_append_stderr):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
mock.side_effect = run_append_stderr
ghostscript.generate_pdfa(
pdf_pages=pdf_pages,
pdfmark=pdfmark,
@@ -47,4 +48,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
pdfa_part=pdfa_part,
progressbar_class=None,
)
return output_file
mock.assert_called_once()
return output_file
+3 -1
View File
@@ -39,7 +39,8 @@ def run_rig_args(args, **kwargs):
@hookimpl
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=run_rig_args):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
mock.side_effect = run_rig_args
ghostscript.generate_pdfa(
pdf_pages=pdf_pages,
pdfmark=pdfmark,
@@ -49,4 +50,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
pdfa_part=pdfa_part,
progressbar_class=None,
)
mock.assert_called()
return output_file
+3 -1
View File
@@ -44,7 +44,8 @@ def rasterize_pdf_page(
rotation=None,
filter_vector=False,
) -> Path:
with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail):
with patch('ocrmypdf._exec.ghostscript.run') as mock:
mock.side_effect = raise_gs_fail
ghostscript.rasterize_pdf_page(
input_file=input_file,
output_file=output_file,
@@ -55,4 +56,5 @@ def rasterize_pdf_page(
rotation=rotation,
filter_vector=filter_vector,
)
mock.assert_called()
return output_file
+3 -1
View File
@@ -34,7 +34,8 @@ def raise_gs_fail(*args, **kwargs):
@hookimpl
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=raise_gs_fail):
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
mock.side_effect = raise_gs_fail
ghostscript.generate_pdfa(
pdf_pages=pdf_pages,
pdfmark=pdfmark,
@@ -44,4 +45,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
pdfa_part=pdfa_part,
progressbar_class=None,
)
mock.assert_called()
return output_file
+11 -2
View File
@@ -26,6 +26,7 @@ that is not UTF-8 compatible, so we are forced to check that we can convert it
and present it to the user.
"""
from contextlib import contextmanager
from subprocess import CalledProcessError
from unittest.mock import patch
@@ -42,17 +43,25 @@ def bad_utf8(*args, **kwargs):
)
@contextmanager
def patch_tesseract_run():
with patch('ocrmypdf._exec.tesseract.run') as mock:
mock.side_effect = bad_utf8
yield
mock.assert_called()
class BadUtf8OcrEngine(TesseractOcrEngine):
@staticmethod
def generate_hocr(input_file, output_hocr, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8):
with patch_tesseract_run():
TesseractOcrEngine.generate_hocr(
input_file, output_hocr, output_text, options
)
@staticmethod
def generate_pdf(input_file, output_pdf, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8):
with patch_tesseract_run():
TesseractOcrEngine.generate_pdf(
input_file, output_pdf, output_text, options
)
+12 -3
View File
@@ -19,6 +19,7 @@
# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
from contextlib import contextmanager
from subprocess import CalledProcessError
from unittest.mock import patch
@@ -35,22 +36,30 @@ def raise_size_exception(*args, **kwargs):
)
@contextmanager
def patch_tesseract_run():
with patch('ocrmypdf._exec.tesseract.run') as mock:
mock.side_effect = raise_size_exception
yield
mock.assert_called()
class BigImageErrorOcrEngine(TesseractOcrEngine):
@staticmethod
def get_orientation(input_file, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
with patch_tesseract_run():
return TesseractOcrEngine.get_orientation(input_file, options)
@staticmethod
def generate_hocr(input_file, output_hocr, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
with patch_tesseract_run():
TesseractOcrEngine.generate_hocr(
input_file, output_hocr, output_text, options
)
@staticmethod
def generate_pdf(input_file, output_pdf, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
with patch_tesseract_run():
TesseractOcrEngine.generate_pdf(
input_file, output_pdf, output_text, options
)
+12 -3
View File
@@ -20,6 +20,7 @@
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
import signal
from contextlib import contextmanager
from subprocess import CalledProcessError
from unittest.mock import patch
@@ -37,22 +38,30 @@ def raise_crash(*args, **kwargs):
)
@contextmanager
def patch_tesseract_run():
with patch('ocrmypdf._exec.tesseract.run') as mock:
mock.side_effect = raise_crash
yield
mock.assert_called()
class CrashOcrEngine(TesseractOcrEngine):
@staticmethod
def get_orientation(input_file, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
with patch_tesseract_run():
return TesseractOcrEngine.get_orientation(input_file, options)
@staticmethod
def generate_hocr(input_file, output_hocr, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
with patch_tesseract_run():
TesseractOcrEngine.generate_hocr(
input_file, output_hocr, output_text, options
)
@staticmethod
def generate_pdf(input_file, output_pdf, output_text, options):
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
with patch_tesseract_run():
TesseractOcrEngine.generate_pdf(
input_file, output_pdf, output_text, options
)
+5
View File
@@ -36,12 +36,17 @@ class TestSafeSymlink:
def test_no_cpu_count(monkeypatch):
invoked = False
def cpu_count_raises():
nonlocal invoked
invoked = True
raise NotImplementedError()
monkeypatch.setattr(multiprocessing, 'cpu_count', cpu_count_raises)
with pytest.warns(expected_warning=UserWarning):
assert helpers.available_cpu_count() == 1
assert invoked, "Patched function called during test"
def test_deprecated():
+2 -1
View File
@@ -65,11 +65,12 @@ def test_cmyk_no_icc(caplog, resources, no_outpdf):
def test_img2pdf_fails(resources, no_outpdf):
with patch(
'ocrmypdf._pipeline.img2pdf.convert', side_effect=img2pdf.ImageOpenError()
):
) as mock:
rc = run_ocrmypdf_api(
resources / 'baiona_gray.png', no_outpdf, '--image-dpi', '200'
)
assert rc == ocrmypdf.ExitCode.input_file
mock.assert_called()
def test_jpeg_in_jpeg_out(resources, outpdf):
+4 -1
View File
@@ -306,6 +306,9 @@ def test_kodak_toc(resources, outpdf):
assert isinstance(p.Root.Outlines.First, pikepdf.Dictionary)
@pytest.mark.skipif(
pikepdf.__version__ in ('2.2.2', '2.2.3'), reason="Raises wrong warning"
)
def test_metadata_fixup_warning(resources, outdir, caplog):
options = get_parser().parse_args(
args=['--output-type', 'pdfa-2', 'graph.pdf', 'out.pdf']
@@ -318,7 +321,7 @@ def test_metadata_fixup_warning(resources, outdir, caplog):
)
metadata_fixup(working_file=outdir / 'graph.pdf', context=context)
for record in caplog.records:
assert record.levelname != 'WARNING'
assert record.levelname != 'WARNING', "Unexpected warning"
# Now add some metadata that will not be copyable
graph = pikepdf.open(outdir / 'graph.pdf')
+17 -7
View File
@@ -21,7 +21,15 @@ from ocrmypdf.helpers import Resolution
check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=e1101
needs_pngquant = pytest.mark.skipif(
not pngquant.available(), reason="pngquant not installed"
)
needs_jbig2enc = pytest.mark.skipif(
not jbig2enc.available(), reason="jbig2enc not installed"
)
@needs_pngquant
@pytest.mark.parametrize('pdf', ['multipage.pdf', 'palette.pdf'])
def test_basic(resources, pdf, outpdf):
infile = resources / pdf
@@ -30,6 +38,7 @@ def test_basic(resources, pdf, outpdf):
assert 0.98 * Path(outpdf).stat().st_size <= Path(infile).stat().st_size
@needs_pngquant
def test_mono_not_inverted(resources, outdir):
infile = resources / '2400dpi.pdf'
opt.main(infile, outdir / 'out.pdf', level=3)
@@ -45,7 +54,7 @@ def test_mono_not_inverted(resources, outdir):
assert im.getpixel((0, 0)) == 255, "Expected white background"
@pytest.mark.skipif(not pngquant.available(), reason='need pngquant')
@needs_pngquant
def test_jpg_png_params(resources, outpdf):
check_ocrmypdf(
resources / 'crom.png',
@@ -63,7 +72,7 @@ def test_jpg_png_params(resources, outpdf):
)
@pytest.mark.skipif(not jbig2enc.available(), reason='need jbig2enc')
@needs_jbig2enc
@pytest.mark.parametrize('lossy', [False, True])
def test_jbig2_lossy(lossy, resources, outpdf):
args = [
@@ -95,10 +104,8 @@ def test_jbig2_lossy(lossy, resources, outpdf):
assert len(pim.decode_parms) == 0
@pytest.mark.skipif(
not jbig2enc.available() or not pngquant.available(),
reason='need jbig2enc and pngquant',
)
@needs_pngquant
@needs_jbig2enc
def test_flate_to_jbig2(resources, outdir):
# This test requires an image that pngquant is capable of converting to
# to 1bpp - so use an existing 1bpp image, convert up, confirm it can
@@ -126,6 +133,7 @@ def test_flate_to_jbig2(resources, outdir):
assert pim.filters[0] == '/JBIG2Decode'
@needs_pngquant
def test_multiple_pngs(resources, outdir):
with Path.open(outdir / 'in.pdf', 'wb') as inpdf:
img2pdf.convert(
@@ -141,7 +149,8 @@ def test_multiple_pngs(resources, outdir):
draw.rectangle((0, 0, im.width, im.height), fill=128)
im.save(output_file)
with patch('ocrmypdf.optimize.pngquant.quantize', new=mockquant):
with patch('ocrmypdf.optimize.pngquant.quantize') as mock:
mock.side_effect = mockquant
check_ocrmypdf(
outdir / 'in.pdf',
outdir / 'out.pdf',
@@ -155,6 +164,7 @@ def test_multiple_pngs(resources, outdir):
'--plugin',
'tests/plugins/tesseract_noop.py',
)
mock.assert_called()
with pikepdf.open(outdir / 'in.pdf') as inpdf, pikepdf.open(
outdir / 'out.pdf'
+6 -4
View File
@@ -28,11 +28,12 @@ def test_no_unpaper(resources, no_outpdf):
output = fspath(no_outpdf)
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version:
mock_unpaper_version.side_effect = FileNotFoundError("unpaper")
with patch("ocrmypdf._exec.unpaper.version") as mock:
mock.side_effect = FileNotFoundError("unpaper")
with pytest.raises(MissingDependencyError):
check_options(options, pm)
mock.assert_called()
def test_old_unpaper(resources, no_outpdf):
@@ -40,11 +41,12 @@ def test_old_unpaper(resources, no_outpdf):
output = fspath(no_outpdf)
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version:
mock_unpaper_version.return_value = '0.5'
with patch("ocrmypdf._exec.unpaper.version") as mock:
mock.return_value = '0.5'
with pytest.raises(MissingDependencyError):
check_options(options, pm)
mock.assert_called()
@pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
+6 -3
View File
@@ -188,18 +188,20 @@ def test_language_warning(caplog):
caplog.set_level(logging.DEBUG)
with patch(
'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8')
):
) as mock:
vd.check_options_languages(opts, {'eng'})
assert opts.languages == {'eng'}
assert '' in caplog.text
mock.assert_called_once()
opts = make_opts(language=None)
with patch(
'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8')
):
) as mock:
vd.check_options_languages(opts, {'eng'})
assert opts.languages == {'eng'}
assert 'assuming --language' in caplog.text
mock.assert_called_once()
def test_version_comparison():
@@ -265,7 +267,8 @@ def test_pagesegmode_warning(caplog):
def test_two_languages():
with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True):
with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True) as mock:
vd._check_options(
*make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'}
)
mock.assert_called()