diff --git a/tests/conftest.py b/tests/conftest.py index d7d4b437..7be5f1f8 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -220,8 +220,12 @@ def check_ocrmypdf(input_file, output_file, *args, env=None): _parser, options, plugin_manager = get_parser_options_plugins(args=args) api.check_options(options, plugin_manager) if env: - options.tesseract_env = env - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) + first = env['_OCRMYPDF_TEST_PATH'].split(os.pathsep)[0] + if 'tesseract_noop' in first: + options.plugins = ['tests/plugins/tesseract_noop.py'] + else: + options.tesseract_env = env + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) result = api.run_pipeline(options, plugin_manager=plugin_manager, api=True) assert result == 0 @@ -243,12 +247,19 @@ def run_ocrmypdf_api(input_file, output_file, *args, env=None): ] _parser, options, plugin_manager = get_parser_options_plugins(args=args) if env: - options.tesseract_env = env.copy() - options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) - first_path = env.get('_OCRMYPDF_TEST_PATH', '').split(os.pathsep)[0] - if 'spoof' in first_path: - assert 'gs' not in first_path, "use run_ocrmypdf() for gs" - assert 'tesseract' in first_path + try: + first = env['_OCRMYPDF_TEST_PATH'].split(os.pathsep)[0] + if 'tesseract_noop' in first: + options.plugins = ['tests/plugins/tesseract_noop.py'] + else: + options.tesseract_env = env.copy() + options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file) + first_path = env.get('_OCRMYPDF_TEST_PATH', '').split(os.pathsep)[0] + if 'spoof' in first_path: + assert 'gs' not in first_path, "use run_ocrmypdf() for gs" + assert 'tesseract' in first_path + except KeyError: + pass if options.tesseract_env: assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values()) diff --git a/tests/plugins/tesseract_noop.py b/tests/plugins/tesseract_noop.py new file mode 100644 index 00000000..1db91fe4 --- /dev/null +++ b/tests/plugins/tesseract_noop.py @@ -0,0 +1,111 @@ +#!/usr/bin/env python3 +# © 2016 James R. Barlow: github.com/jbarlow83 +# +# Permission is hereby granted, free of charge, to any person obtaining a +# copy of this software and associated documentation files (the +# "Software"), to deal in the Software without restriction, including +# without limitation the rights to use, copy, modify, merge, publish, +# distribute, sublicense, and/or sell copies of the Software, and to +# permit persons to whom the Software is furnished to do so, subject to +# the following conditions: +# +# The above copyright notice and this permission notice shall be included +# in all copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + +"""Tesseract no-op spoof + +To quickly run tests where getting OCR output is not necessary. + +In 'hocr' mode, create a .hocr file that specifies no text found. + +In 'pdf' mode, convert the image to PDF using another program. + +In orientation check mode, report the orientation is upright. +""" + +import sys +from pathlib import Path + +import img2pdf +import pikepdf +from PIL import Image + +from ocrmypdf import OcrEngine, OrientationConfidence, hookimpl + +HOCR_TEMPLATE = ''' + + + + + + + + + +
+
+

+ + +

+
+
+ +''' + + +class NoopOcrEngine(OcrEngine): + @staticmethod + def version(): + return '4.0.0' + + @staticmethod + def creator_tag(options): + tag = '-PDF' if options.pdf_renderer == 'sandwich' else '' + return f"NO-OP {tag} {NoopOcrEngine.version()}" + + def __str__(self): + return f"NO-OP {NoopOcrEngine.version()}" + + @staticmethod + def languages(options): + return {'eng'} + + @staticmethod + def get_orientation(input_file, options): + return OrientationConfidence(angle=0, confidence=0.0) + + @staticmethod + def generate_hocr(input_file, output_hocr, output_text, options): + with Image.open(input_file) as im, open( + output_hocr, 'w', encoding='utf-8' + ) as f: + w, h = im.size + f.write(HOCR_TEMPLATE.format(str(w), str(h))) + with open(output_text, 'w') as f: + f.write('') + + @staticmethod + def generate_pdf(input_file, output_pdf, output_text, options): + with Image.open(input_file) as im: + dpi = im.info['dpi'] + pagesize = im.size[0] / dpi[0], im.size[1] / dpi[1] + ptsize = pagesize[0] * 72, pagesize[1] * 72 + pdf = pikepdf.new() + pdf.add_blank_page(page_size=ptsize) + pdf.save(output_pdf, static_id=True) + output_text.write_text('') + + +@hookimpl +def get_ocr_engine(): + return NoopOcrEngine() diff --git a/tests/test_main.py b/tests/test_main.py index c3a715ce..8c417d76 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -182,14 +182,10 @@ def test_maximum_options( ) -def test_tesseract_missing_tessdata(resources, no_outpdf, tmpdir): - env = os.environ.copy() - env['TESSDATA_PREFIX'] = os.fspath(tmpdir) - +def test_tesseract_missing_tessdata(monkeypatch, resources, no_outpdf, tmpdir): + monkeypatch.setenv("TESSDATA_PREFIX", os.fspath(tmpdir)) with pytest.raises(MissingDependencyError): - run_ocrmypdf_api( - resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text', env=env - ) + run_ocrmypdf_api(resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text') def test_invalid_input_pdf(resources, no_outpdf):