diff --git a/src/ocrmypdf/_pipeline.py b/src/ocrmypdf/_pipeline.py index 445d0faa..6435804e 100644 --- a/src/ocrmypdf/_pipeline.py +++ b/src/ocrmypdf/_pipeline.py @@ -47,8 +47,8 @@ VECTOR_PAGE_DPI = 400 def triage_image_file(input_file, output_file, options, log): + log.info("Input file is not a PDF, checking if it is an image...") try: - log.info("Input file is not a PDF, checking if it is an image...") im = Image.open(input_file) except EnvironmentError as e: # Recover the original filename @@ -85,9 +85,9 @@ def triage_image_file(input_file, output_file, options, log): if 'iccprofile' not in im.info: if im.mode == 'RGB': - log.info('Input image has no ICC profile, assuming sRGB') + log.info("Input image has no ICC profile, assuming sRGB") elif im.mode == 'CMYK': - log.info('Input CMYK image has no ICC profile, not usable') + log.error("Input CMYK image has no ICC profile, not usable") raise UnsupportedImageFormatError() try: diff --git a/tests/resources/baiona_cmyk.jpg b/tests/resources/baiona_cmyk.jpg new file mode 100644 index 00000000..01d6badf Binary files /dev/null and b/tests/resources/baiona_cmyk.jpg differ diff --git a/tests/test_acroform.py b/tests/test_acroform.py new file mode 100644 index 00000000..48b9ff91 --- /dev/null +++ b/tests/test_acroform.py @@ -0,0 +1,34 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +import pytest + +import ocrmypdf + + +check_ocrmypdf = pytest.helpers.check_ocrmypdf + + +@pytest.fixture +def acroform(resources): + return resources / 'acroform.pdf' + + +def test_acroform_and_redo(acroform, caplog, no_outpdf): + with pytest.raises(ocrmypdf.exceptions.InputFileError): + check_ocrmypdf(acroform, no_outpdf, '--redo-ocr') + assert '--redo-ocr is not currently possible' in caplog.text diff --git a/tests/test_image_input.py b/tests/test_image_input.py new file mode 100644 index 00000000..b5deedb4 --- /dev/null +++ b/tests/test_image_input.py @@ -0,0 +1,92 @@ +# © 2019 James R. Barlow: github.com/jbarlow83 +# +# This file is part of OCRmyPDF. +# +# OCRmyPDF is free software: you can redistribute it and/or modify +# it under the terms of the GNU General Public License as published by +# the Free Software Foundation, either version 3 of the License, or +# (at your option) any later version. +# +# OCRmyPDF is distributed in the hope that it will be useful, +# but WITHOUT ANY WARRANTY; without even the implied warranty of +# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# GNU General Public License for more details. +# +# You should have received a copy of the GNU General Public License +# along with OCRmyPDF. If not, see . + +from unittest.mock import patch + +import pytest +from PIL import Image +import img2pdf +import pikepdf + +import ocrmypdf + +check_ocrmypdf = pytest.helpers.check_ocrmypdf +run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api + + +@pytest.fixture +def baiona(resources): + return Image.open(resources / 'baiona_gray.png') + + +def test_image_to_pdf(spoof_tesseract_noop, resources, outpdf): + check_ocrmypdf( + resources / 'crom.png', outpdf, '--image-dpi', '200', env=spoof_tesseract_noop + ) + + +def test_no_dpi_info(caplog, baiona, outdir, no_outpdf): + im = baiona + assert 'dpi' not in im.info + input_image = outdir / 'baiona_no_dpi.png' + im.save(input_image) + + rc = run_ocrmypdf_api(input_image, no_outpdf) + assert rc == ocrmypdf.ExitCode.input_file + assert "--image-dpi" in caplog.text + + +def test_dpi_not_credible(caplog, baiona, outdir, no_outpdf): + im = baiona + assert 'dpi' not in im.info + input_image = outdir / 'baiona_no_dpi.png' + im.save(input_image, dpi=(30, 30)) + + rc = run_ocrmypdf_api(input_image, no_outpdf) + assert rc == ocrmypdf.ExitCode.input_file + assert "not credible" in caplog.text + + +def test_cmyk_no_icc(caplog, resources, no_outpdf): + rc = run_ocrmypdf_api(resources / 'baiona_cmyk.jpg', no_outpdf) + assert rc == ocrmypdf.ExitCode.input_file + assert "no ICC profile" in caplog.text + + +def test_img2pdf_fails(resources, no_outpdf): + with patch( + 'ocrmypdf._pipeline.img2pdf.convert', side_effect=img2pdf.ImageOpenError() + ): + rc = run_ocrmypdf_api( + resources / 'baiona_gray.png', no_outpdf, '--image-dpi', '200' + ) + assert rc == ocrmypdf.ExitCode.input_file + + +def test_jpeg_in_jpeg_out(resources, outpdf, spoof_tesseract_noop): + check_ocrmypdf( + resources / 'congress.jpg', + outpdf, + '--image-dpi', + '100', + '--output-type', + 'pdf', # specifically check pdf because Ghostscript may convert to JPEG + '--remove-background', + env=spoof_tesseract_noop, + ) + with pikepdf.open(outpdf) as pdf: + assert next(pdf.pages[0].images.values()).Filter == pikepdf.Name.DCTDecode diff --git a/tests/test_main.py b/tests/test_main.py index 4ef299fb..0380508b 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -350,12 +350,6 @@ def test_algo4(resources, spoof_tesseract_noop, outpdf): assert p.returncode == ExitCode.encrypted_pdf -def test_image_to_pdf(spoof_tesseract_noop, resources, outpdf): - check_ocrmypdf( - resources / 'crom.png', outpdf, '--image-dpi', '200', env=spoof_tesseract_noop - ) - - def test_jbig2_passthrough(spoof_tesseract_cache, resources, outpdf): out = check_ocrmypdf( resources / 'jbig2.pdf',