Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bd0f005861 | ||
|
|
6ba4b7b3f3 | ||
|
|
2c11349ee8 | ||
|
|
b0afef09ef | ||
|
|
72fa347c38 | ||
|
|
96d68c2413 | ||
|
|
babc76fa74 | ||
|
|
dc06990e5d | ||
|
|
0ff0d2f8d1 | ||
|
|
81602cf420 |
+1
-1
@@ -32,7 +32,7 @@ stages:
|
||||
choco install --yes --no-progress --pre tesseract
|
||||
choco install --yes --no-progress python3
|
||||
choco install --yes --no-progress ghostscript
|
||||
choco install --yes --no-progress pngquant
|
||||
# choco install --yes --no-progress pngquant
|
||||
displayName: "Install system packages"
|
||||
- pwsh: |
|
||||
refreshenv
|
||||
|
||||
@@ -637,7 +637,7 @@ Installing with Python pip
|
||||
OCRmyPDF is delivered by PyPI because it is a convenient way to install
|
||||
the latest version. However, PyPI and ``pip`` cannot address the fact
|
||||
that ``ocrmypdf`` depends on certain non-Python system libraries and
|
||||
programs being instsalled.
|
||||
programs being installed.
|
||||
|
||||
For best results, first install `your platform's
|
||||
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
|
||||
|
||||
@@ -12,6 +12,13 @@ may be unreliable. Use the API to depend on precise behavior.
|
||||
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||
wish to use some of its features for working with PDFs.
|
||||
|
||||
v11.4.3
|
||||
=======
|
||||
|
||||
- Removed a redundant debug message.
|
||||
- Test suite now asserts that most patched functions are called when they should be.
|
||||
- Test suite now skips a test that fails on two particular versions of piekpdf.
|
||||
|
||||
v11.4.2
|
||||
=======
|
||||
|
||||
|
||||
@@ -254,6 +254,8 @@ def generate_pdfa(
|
||||
raise SubprocessOutputError('Ghostscript PDF/A rendering failed') from e
|
||||
else:
|
||||
stderr = p.stderr
|
||||
# If there is an error we log the whole stderr, except for filtering
|
||||
# duplicates.
|
||||
if _gs_error_reported(stderr):
|
||||
last_part = None
|
||||
repcount = 0
|
||||
@@ -266,11 +268,3 @@ def generate_pdfa(
|
||||
else:
|
||||
repcount += 1
|
||||
last_part = part
|
||||
elif 'overprint mode not set' in stderr:
|
||||
# Unless someone is going to print PDF/A documents on a
|
||||
# magical sRGB printer I can't see the removal of overprinting
|
||||
# being a problem....
|
||||
log.debug(
|
||||
"Ghostscript had to remove PDF 'overprinting' from the "
|
||||
"input file to complete PDF/A conversion. "
|
||||
)
|
||||
|
||||
+10
-12
@@ -756,19 +756,17 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
||||
if 'xmp:CreateDate' not in meta:
|
||||
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
||||
|
||||
# Ghostscript likes to set title to Untitled if omitted from input.
|
||||
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
||||
# and the XMP Spec do not make this recommendation.
|
||||
if meta.get('dc:title') == 'Untitled':
|
||||
with original.open_metadata(
|
||||
set_pikepdf_as_editor=False, update_docinfo=False
|
||||
) as original_meta:
|
||||
if 'dc:title' not in original_meta:
|
||||
with original.open_metadata(
|
||||
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
||||
) as meta_original:
|
||||
if meta.get('dc:title') == 'Untitled':
|
||||
# Ghostscript likes to set title to Untitled if omitted from input.
|
||||
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
||||
# and the XMP Spec do not make this recommendation.
|
||||
if 'dc:title' not in meta_original:
|
||||
del meta['dc:title']
|
||||
|
||||
meta_original = original.open_metadata()
|
||||
missing = set(meta_original.keys()) - set(meta.keys())
|
||||
report_on_metadata(missing)
|
||||
missing = set(meta_original.keys()) - set(meta.keys())
|
||||
report_on_metadata(missing)
|
||||
|
||||
pdf.save(
|
||||
output_file,
|
||||
|
||||
@@ -23,21 +23,22 @@ from unittest.mock import patch
|
||||
|
||||
from ocrmypdf import hookimpl
|
||||
from ocrmypdf.builtin_plugins import ghostscript
|
||||
from ocrmypdf.subprocess import run
|
||||
from ocrmypdf.subprocess import run_polling_stderr
|
||||
|
||||
elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1
|
||||
not permitted in PDF/A-2, overprint mode not set"""
|
||||
|
||||
|
||||
def run_append_stderr(*args, **kwargs):
|
||||
proc = run(*args, **kwargs)
|
||||
proc.stderr = b'\n'.join([proc.stderr, elision_warning.encode('utf-8')])
|
||||
proc = run_polling_stderr(*args, **kwargs)
|
||||
proc.stderr += '\n' + elision_warning + '\n'
|
||||
return proc
|
||||
|
||||
|
||||
@hookimpl
|
||||
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
||||
with patch('ocrmypdf._exec.ghostscript.run', new=run_append_stderr):
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
|
||||
mock.side_effect = run_append_stderr
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=pdf_pages,
|
||||
pdfmark=pdfmark,
|
||||
@@ -47,4 +48,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=None,
|
||||
)
|
||||
return output_file
|
||||
mock.assert_called_once()
|
||||
return output_file
|
||||
|
||||
@@ -39,7 +39,8 @@ def run_rig_args(args, **kwargs):
|
||||
|
||||
@hookimpl
|
||||
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=run_rig_args):
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
|
||||
mock.side_effect = run_rig_args
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=pdf_pages,
|
||||
pdfmark=pdfmark,
|
||||
@@ -49,4 +50,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=None,
|
||||
)
|
||||
mock.assert_called()
|
||||
return output_file
|
||||
|
||||
@@ -44,7 +44,8 @@ def rasterize_pdf_page(
|
||||
rotation=None,
|
||||
filter_vector=False,
|
||||
) -> Path:
|
||||
with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail):
|
||||
with patch('ocrmypdf._exec.ghostscript.run') as mock:
|
||||
mock.side_effect = raise_gs_fail
|
||||
ghostscript.rasterize_pdf_page(
|
||||
input_file=input_file,
|
||||
output_file=output_file,
|
||||
@@ -55,4 +56,5 @@ def rasterize_pdf_page(
|
||||
rotation=rotation,
|
||||
filter_vector=filter_vector,
|
||||
)
|
||||
mock.assert_called()
|
||||
return output_file
|
||||
|
||||
@@ -34,7 +34,8 @@ def raise_gs_fail(*args, **kwargs):
|
||||
|
||||
@hookimpl
|
||||
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', new=raise_gs_fail):
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
|
||||
mock.side_effect = raise_gs_fail
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=pdf_pages,
|
||||
pdfmark=pdfmark,
|
||||
@@ -44,4 +45,5 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=None,
|
||||
)
|
||||
mock.assert_called()
|
||||
return output_file
|
||||
|
||||
@@ -26,6 +26,7 @@ that is not UTF-8 compatible, so we are forced to check that we can convert it
|
||||
and present it to the user.
|
||||
"""
|
||||
|
||||
from contextlib import contextmanager
|
||||
from subprocess import CalledProcessError
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -42,17 +43,25 @@ def bad_utf8(*args, **kwargs):
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def patch_tesseract_run():
|
||||
with patch('ocrmypdf._exec.tesseract.run') as mock:
|
||||
mock.side_effect = bad_utf8
|
||||
yield
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
class BadUtf8OcrEngine(TesseractOcrEngine):
|
||||
@staticmethod
|
||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_hocr(
|
||||
input_file, output_hocr, output_text, options
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def generate_pdf(input_file, output_pdf, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_pdf(
|
||||
input_file, output_pdf, output_text, options
|
||||
)
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
from contextlib import contextmanager
|
||||
from subprocess import CalledProcessError
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -35,22 +36,30 @@ def raise_size_exception(*args, **kwargs):
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def patch_tesseract_run():
|
||||
with patch('ocrmypdf._exec.tesseract.run') as mock:
|
||||
mock.side_effect = raise_size_exception
|
||||
yield
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
class BigImageErrorOcrEngine(TesseractOcrEngine):
|
||||
@staticmethod
|
||||
def get_orientation(input_file, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
|
||||
with patch_tesseract_run():
|
||||
return TesseractOcrEngine.get_orientation(input_file, options)
|
||||
|
||||
@staticmethod
|
||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_hocr(
|
||||
input_file, output_hocr, output_text, options
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def generate_pdf(input_file, output_pdf, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_pdf(
|
||||
input_file, output_pdf, output_text, options
|
||||
)
|
||||
|
||||
@@ -20,6 +20,7 @@
|
||||
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
import signal
|
||||
from contextlib import contextmanager
|
||||
from subprocess import CalledProcessError
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -37,22 +38,30 @@ def raise_crash(*args, **kwargs):
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def patch_tesseract_run():
|
||||
with patch('ocrmypdf._exec.tesseract.run') as mock:
|
||||
mock.side_effect = raise_crash
|
||||
yield
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
class CrashOcrEngine(TesseractOcrEngine):
|
||||
@staticmethod
|
||||
def get_orientation(input_file, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
|
||||
with patch_tesseract_run():
|
||||
return TesseractOcrEngine.get_orientation(input_file, options)
|
||||
|
||||
@staticmethod
|
||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_hocr(
|
||||
input_file, output_hocr, output_text, options
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def generate_pdf(input_file, output_pdf, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_pdf(
|
||||
input_file, output_pdf, output_text, options
|
||||
)
|
||||
|
||||
@@ -36,12 +36,17 @@ class TestSafeSymlink:
|
||||
|
||||
|
||||
def test_no_cpu_count(monkeypatch):
|
||||
invoked = False
|
||||
|
||||
def cpu_count_raises():
|
||||
nonlocal invoked
|
||||
invoked = True
|
||||
raise NotImplementedError()
|
||||
|
||||
monkeypatch.setattr(multiprocessing, 'cpu_count', cpu_count_raises)
|
||||
with pytest.warns(expected_warning=UserWarning):
|
||||
assert helpers.available_cpu_count() == 1
|
||||
assert invoked, "Patched function called during test"
|
||||
|
||||
|
||||
def test_deprecated():
|
||||
|
||||
@@ -65,11 +65,12 @@ def test_cmyk_no_icc(caplog, resources, no_outpdf):
|
||||
def test_img2pdf_fails(resources, no_outpdf):
|
||||
with patch(
|
||||
'ocrmypdf._pipeline.img2pdf.convert', side_effect=img2pdf.ImageOpenError()
|
||||
):
|
||||
) as mock:
|
||||
rc = run_ocrmypdf_api(
|
||||
resources / 'baiona_gray.png', no_outpdf, '--image-dpi', '200'
|
||||
)
|
||||
assert rc == ocrmypdf.ExitCode.input_file
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
def test_jpeg_in_jpeg_out(resources, outpdf):
|
||||
|
||||
@@ -306,6 +306,9 @@ def test_kodak_toc(resources, outpdf):
|
||||
assert isinstance(p.Root.Outlines.First, pikepdf.Dictionary)
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
pikepdf.__version__ in ('2.2.2', '2.2.3'), reason="Raises wrong warning"
|
||||
)
|
||||
def test_metadata_fixup_warning(resources, outdir, caplog):
|
||||
options = get_parser().parse_args(
|
||||
args=['--output-type', 'pdfa-2', 'graph.pdf', 'out.pdf']
|
||||
@@ -318,7 +321,7 @@ def test_metadata_fixup_warning(resources, outdir, caplog):
|
||||
)
|
||||
metadata_fixup(working_file=outdir / 'graph.pdf', context=context)
|
||||
for record in caplog.records:
|
||||
assert record.levelname != 'WARNING'
|
||||
assert record.levelname != 'WARNING', "Unexpected warning"
|
||||
|
||||
# Now add some metadata that will not be copyable
|
||||
graph = pikepdf.open(outdir / 'graph.pdf')
|
||||
|
||||
+17
-7
@@ -21,7 +21,15 @@ from ocrmypdf.helpers import Resolution
|
||||
|
||||
check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=e1101
|
||||
|
||||
needs_pngquant = pytest.mark.skipif(
|
||||
not pngquant.available(), reason="pngquant not installed"
|
||||
)
|
||||
needs_jbig2enc = pytest.mark.skipif(
|
||||
not jbig2enc.available(), reason="jbig2enc not installed"
|
||||
)
|
||||
|
||||
|
||||
@needs_pngquant
|
||||
@pytest.mark.parametrize('pdf', ['multipage.pdf', 'palette.pdf'])
|
||||
def test_basic(resources, pdf, outpdf):
|
||||
infile = resources / pdf
|
||||
@@ -30,6 +38,7 @@ def test_basic(resources, pdf, outpdf):
|
||||
assert 0.98 * Path(outpdf).stat().st_size <= Path(infile).stat().st_size
|
||||
|
||||
|
||||
@needs_pngquant
|
||||
def test_mono_not_inverted(resources, outdir):
|
||||
infile = resources / '2400dpi.pdf'
|
||||
opt.main(infile, outdir / 'out.pdf', level=3)
|
||||
@@ -45,7 +54,7 @@ def test_mono_not_inverted(resources, outdir):
|
||||
assert im.getpixel((0, 0)) == 255, "Expected white background"
|
||||
|
||||
|
||||
@pytest.mark.skipif(not pngquant.available(), reason='need pngquant')
|
||||
@needs_pngquant
|
||||
def test_jpg_png_params(resources, outpdf):
|
||||
check_ocrmypdf(
|
||||
resources / 'crom.png',
|
||||
@@ -63,7 +72,7 @@ def test_jpg_png_params(resources, outpdf):
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.skipif(not jbig2enc.available(), reason='need jbig2enc')
|
||||
@needs_jbig2enc
|
||||
@pytest.mark.parametrize('lossy', [False, True])
|
||||
def test_jbig2_lossy(lossy, resources, outpdf):
|
||||
args = [
|
||||
@@ -95,10 +104,8 @@ def test_jbig2_lossy(lossy, resources, outpdf):
|
||||
assert len(pim.decode_parms) == 0
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not jbig2enc.available() or not pngquant.available(),
|
||||
reason='need jbig2enc and pngquant',
|
||||
)
|
||||
@needs_pngquant
|
||||
@needs_jbig2enc
|
||||
def test_flate_to_jbig2(resources, outdir):
|
||||
# This test requires an image that pngquant is capable of converting to
|
||||
# to 1bpp - so use an existing 1bpp image, convert up, confirm it can
|
||||
@@ -126,6 +133,7 @@ def test_flate_to_jbig2(resources, outdir):
|
||||
assert pim.filters[0] == '/JBIG2Decode'
|
||||
|
||||
|
||||
@needs_pngquant
|
||||
def test_multiple_pngs(resources, outdir):
|
||||
with Path.open(outdir / 'in.pdf', 'wb') as inpdf:
|
||||
img2pdf.convert(
|
||||
@@ -141,7 +149,8 @@ def test_multiple_pngs(resources, outdir):
|
||||
draw.rectangle((0, 0, im.width, im.height), fill=128)
|
||||
im.save(output_file)
|
||||
|
||||
with patch('ocrmypdf.optimize.pngquant.quantize', new=mockquant):
|
||||
with patch('ocrmypdf.optimize.pngquant.quantize') as mock:
|
||||
mock.side_effect = mockquant
|
||||
check_ocrmypdf(
|
||||
outdir / 'in.pdf',
|
||||
outdir / 'out.pdf',
|
||||
@@ -155,6 +164,7 @@ def test_multiple_pngs(resources, outdir):
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
mock.assert_called()
|
||||
|
||||
with pikepdf.open(outdir / 'in.pdf') as inpdf, pikepdf.open(
|
||||
outdir / 'out.pdf'
|
||||
|
||||
@@ -28,11 +28,12 @@ def test_no_unpaper(resources, no_outpdf):
|
||||
output = fspath(no_outpdf)
|
||||
|
||||
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
|
||||
with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version:
|
||||
mock_unpaper_version.side_effect = FileNotFoundError("unpaper")
|
||||
with patch("ocrmypdf._exec.unpaper.version") as mock:
|
||||
mock.side_effect = FileNotFoundError("unpaper")
|
||||
|
||||
with pytest.raises(MissingDependencyError):
|
||||
check_options(options, pm)
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
def test_old_unpaper(resources, no_outpdf):
|
||||
@@ -40,11 +41,12 @@ def test_old_unpaper(resources, no_outpdf):
|
||||
output = fspath(no_outpdf)
|
||||
|
||||
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
|
||||
with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version:
|
||||
mock_unpaper_version.return_value = '0.5'
|
||||
with patch("ocrmypdf._exec.unpaper.version") as mock:
|
||||
mock.return_value = '0.5'
|
||||
|
||||
with pytest.raises(MissingDependencyError):
|
||||
check_options(options, pm)
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
@pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
|
||||
|
||||
@@ -188,18 +188,20 @@ def test_language_warning(caplog):
|
||||
caplog.set_level(logging.DEBUG)
|
||||
with patch(
|
||||
'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8')
|
||||
):
|
||||
) as mock:
|
||||
vd.check_options_languages(opts, {'eng'})
|
||||
assert opts.languages == {'eng'}
|
||||
assert '' in caplog.text
|
||||
mock.assert_called_once()
|
||||
|
||||
opts = make_opts(language=None)
|
||||
with patch(
|
||||
'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8')
|
||||
):
|
||||
) as mock:
|
||||
vd.check_options_languages(opts, {'eng'})
|
||||
assert opts.languages == {'eng'}
|
||||
assert 'assuming --language' in caplog.text
|
||||
mock.assert_called_once()
|
||||
|
||||
|
||||
def test_version_comparison():
|
||||
@@ -265,7 +267,8 @@ def test_pagesegmode_warning(caplog):
|
||||
|
||||
|
||||
def test_two_languages():
|
||||
with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True):
|
||||
with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True) as mock:
|
||||
vd._check_options(
|
||||
*make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'}
|
||||
)
|
||||
mock.assert_called()
|
||||
|
||||
Reference in New Issue
Block a user