Compare commits

...
Author SHA1 Message Date
James R. Barlow f255723e8d v6.2.5 release notes 2018-10-28 14:31:48 -07:00
James R. Barlow 88b6c3df8f Refactor the detailed error messages
Cherry-picked and resolved from c74f2ee
2018-10-28 14:26:49 -07:00
James R. Barlow 190bfe8859 Fix suppression of tesseract config error messages
Cherry-pick and merge from fb8b161
2018-10-28 14:22:01 -07:00
James R. Barlow 0cc23b999c test: fix pytest warning about direct use of a fixture 2018-10-28 14:08:03 -07:00
James R. Barlow ce930aa16e tesseract: account for behavior changes when params are missing
Tesseract 4.0-rc1 now accepts invalid parameters in config and
won't return an error anymore. We prefer to raise an error if this
occurs.

See: https://github.com/tesseract-ocr/tesseract/commit/741ea00d7059d8ff7c55797ffc525a461d7f3ced
2018-10-28 14:07:47 -07:00
James R. Barlow 1ce6c8daf3 test: this error message changed case in newer Tesseract 2018-10-28 14:07:22 -07:00
6 changed files with 75 additions and 44 deletions
+6
View File
@@ -10,6 +10,12 @@ The OCRmyPDF package itself does not contain a public API, although it is fairly
replace: `#$1 <https://github.com/jbarlow83/OCRmyPDF/issues/$1>`_ replace: `#$1 <https://github.com/jbarlow83/OCRmyPDF/issues/$1>`_
v6.2.5
------
- Backport compatibility fixes for Tesseract 4.0.0-rcN from v7.2.0
v6.2.4 v6.2.4
------ ------
+23 -32
View File
@@ -587,9 +587,27 @@ def do_ruffus_exception(ruffus_five_tuple, options, log):
description of the error message that occurred.""" description of the error message that occurred."""
exit_code = None exit_code = None
task_name, job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple _task_name, _job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple
job_name = job_name # unused
if exc_name == 'builtins.SystemExit': if isinstance(exc_name, type):
# ruffus is full of mystery... sometimes (probably when the process
# group leader is killed) exc_name is the class object of the exception,
# rather than a str. So reach into the object and get its name.
exc_name = exc_name.__name__
if exc_name.startswith('ocrmypdf.exceptions.'):
base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '')
exc_class = getattr(ocrmypdf_exceptions, base_exc_name)
exit_code = getattr(exc_class, 'exit_code', ExitCode.other_error)
try:
if isinstance(exc_value, exc_class):
exc_msg = str(exc_value)
else:
exc_msg = str(exc_class())
except Exception:
exc_msg = "Unknown"
if exc_name in ('builtins.SystemExit', 'SystemExit'):
match = re.search(r"\.(.+?)\)", exc_value) match = re.search(r"\.(.+?)\)", exc_value)
exit_code_name = match.groups()[0] exit_code_name = match.groups()[0]
exit_code = getattr(ExitCode, exit_code_name, 'other_error') exit_code = getattr(ExitCode, exit_code_name, 'other_error')
@@ -611,36 +629,9 @@ def do_ruffus_exception(ruffus_five_tuple, options, log):
msg = "Error occurred while running this command:" msg = "Error occurred while running this command:"
log.error(msg + '\n' + exc_value) log.error(msg + '\n' + exc_value)
exit_code = ExitCode.child_process_error exit_code = ExitCode.child_process_error
elif (exc_name == 'PyPDF2.utils.PdfReadError' and \
'not been decrypted' in exc_value) or \
(exc_name == 'ocrmypdf.exceptions.EncryptedPdfError'):
log.error(textwrap.dedent("""\
Input PDF is encrypted. The encryption must be removed to
perform OCR.
For information about this PDF's security use
qpdf --show-encryption infilename
You can remove the encryption using
qpdf --decrypt [--password=[password]] infilename
"""))
exit_code = ExitCode.encrypted_pdf
elif exc_name == 'ocrmypdf.exceptions.PdfMergeFailedError':
log.error(textwrap.dedent("""\
Failed to merge PDF image layer with OCR layer
Usually this happens because the input PDF file is mal-formed and
ocrmypdf cannot automatically correct the problem on its own.
Try using
ocrmypdf --pdf-renderer tesseract [..other args..]
"""))
exit_code = ExitCode.input_file
elif exc_name.startswith('ocrmypdf.exceptions.'): elif exc_name.startswith('ocrmypdf.exceptions.'):
base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '') if exc_msg:
exc_class = getattr(ocrmypdf_exceptions, base_exc_name) log.error(exc_msg)
exit_code = exc_class.exit_code
elif exc_name == 'PIL.Image.DecompressionBombError': elif exc_name == 'PIL.Image.DecompressionBombError':
msg = cleanup_ruffus_error_message(exc_value) msg = cleanup_ruffus_error_message(exc_value)
msg += ("\nUse the --max-image-mpixels argument to set increase the " msg += ("\nUse the --max-image-mpixels argument to set increase the "
+27
View File
@@ -17,6 +17,7 @@
from enum import IntEnum from enum import IntEnum
from textwrap import dedent
class ExitCode(IntEnum): class ExitCode(IntEnum):
ok = 0 ok = 0
@@ -35,6 +36,13 @@ class ExitCode(IntEnum):
class ExitCodeException(Exception): class ExitCodeException(Exception):
exit_code = ExitCode.other_error exit_code = ExitCode.other_error
message = ""
def __str__(self):
super_msg = super().__str__() # Don't do str(super())
if self.message:
return self.message.format(super_msg)
return super_msg
class BadArgsError(ExitCodeException): class BadArgsError(ExitCodeException):
@@ -43,7 +51,15 @@ class BadArgsError(ExitCodeException):
class PdfMergeFailedError(ExitCodeException): class PdfMergeFailedError(ExitCodeException):
exit_code = ExitCode.input_file exit_code = ExitCode.input_file
message = dedent('''\
Failed to merge PDF image layer with OCR layer
Usually this happens because the input PDF file is malformed and
ocrmypdf cannot automatically correct the problem on its own.
Try using
ocrmypdf --pdf-renderer sandwich [..other args..]
''')
class MissingDependencyError(ExitCodeException): class MissingDependencyError(ExitCodeException):
exit_code = ExitCode.missing_dependency exit_code = ExitCode.missing_dependency
@@ -75,7 +91,18 @@ class SubprocessOutputError(ExitCodeException):
class EncryptedPdfError(ExitCodeException): class EncryptedPdfError(ExitCodeException):
exit_code = ExitCode.encrypted_pdf exit_code = ExitCode.encrypted_pdf
message = dedent('''\
Input PDF is encrypted. The encryption must be removed to
perform OCR.
For information about this PDF's security use
qpdf --show-encryption infilename
You can remove the encryption using
qpdf --decrypt [--password=[password]] infilename
''')
class TesseractConfigError(ExitCodeException): class TesseractConfigError(ExitCodeException):
exit_code = ExitCode.invalid_config exit_code = ExitCode.invalid_config
message = "Error occurred while parsing a Tesseract configuration file"
+10 -7
View File
@@ -67,8 +67,8 @@ def v4():
@lru_cache(maxsize=1) @lru_cache(maxsize=1)
def has_textonly_pdf(): def has_textonly_pdf():
"""Does Tesseract have textonly_pdf capability? """Does Tesseract have textonly_pdf capability?
Available in 3.05.01, and v4.00.00alpha since January 2017. Best to Available in 3.05.01, and v4.00.00alpha since January 2017. Best to
parse the parameter list parse the parameter list
""" """
args_tess = [ args_tess = [
@@ -191,6 +191,14 @@ def tesseract_log_output(log, stdout, input_file):
log.warning(prefix + "lots of diacritics - possibly poor OCR") log.warning(prefix + "lots of diacritics - possibly poor OCR")
elif line.startswith('OSD: Weak margin'): elif line.startswith('OSD: Weak margin'):
log.warning(prefix + "unsure about page orientation") log.warning(prefix + "unsure about page orientation")
elif 'Error in pixScanForForeground' in line:
pass # Appears to be spurious/problem with nonwhite borders
elif 'Error in boxClipToRectangle' in line:
pass # Always appears with pixScanForForeground message
elif 'parameter not found: ' in line.lower():
log.error(prefix + line.strip())
problem = line.split('found: ')[1]
raise TesseractConfigError(problem)
elif 'error' in line.lower() or 'exception' in line.lower(): elif 'error' in line.lower() or 'exception' in line.lower():
log.error(prefix + line.strip()) log.error(prefix + line.strip())
elif 'warning' in line.lower(): elif 'warning' in line.lower():
@@ -264,8 +272,6 @@ def generate_hocr(input_file, output_files, language: list, engine_mode,
_generate_null_hocr(output_hocr, output_sidecar, input_file) _generate_null_hocr(output_hocr, output_sidecar, input_file)
except CalledProcessError as e: except CalledProcessError as e:
tesseract_log_output(log, e.output, input_file) tesseract_log_output(log, e.output, input_file)
if b'read_params_file: parameter not found' in e.output:
raise TesseractConfigError() from e
if b'Image too large' in e.output: if b'Image too large' in e.output:
_generate_null_hocr(output_hocr, output_sidecar, input_file) _generate_null_hocr(output_hocr, output_sidecar, input_file)
return return
@@ -362,9 +368,6 @@ def generate_pdf(*, input_image, skip_pdf, output_pdf, output_text,
use_skip_page(text_only, skip_pdf, output_pdf, output_text) use_skip_page(text_only, skip_pdf, output_pdf, output_text)
except CalledProcessError as e: except CalledProcessError as e:
tesseract_log_output(log, e.output, input_image) tesseract_log_output(log, e.output, input_image)
if b'read_params_file: parameter not found' in e.output:
raise TesseractConfigError() from e
if b'Image too large' in e.output: if b'Image too large' in e.output:
use_skip_page(text_only, skip_pdf, output_pdf, output_text) use_skip_page(text_only, skip_pdf, output_pdf, output_text)
return return
+1 -1
View File
@@ -734,7 +734,7 @@ THIS FILE IS INVALID
resources / 'ccitt.pdf', outdir / 'out.pdf', resources / 'ccitt.pdf', outdir / 'out.pdf',
'--pdf-renderer', renderer, '--pdf-renderer', renderer,
'--tesseract-config', cfg_file) '--tesseract-config', cfg_file)
assert "parameter not found" in err, "No error message" assert "parameter not found" in err.lower(), "No error message"
assert p.returncode == ExitCode.invalid_config assert p.returncode == ExitCode.invalid_config
+8 -4
View File
@@ -29,8 +29,7 @@ from pathlib import Path
spoof = pytest.helpers.spoof spoof = pytest.helpers.spoof
@pytest.fixture def _ensure_tess4():
def ensure_tess4():
if tesseract.v4(): if tesseract.v4():
# "tesseract" on $PATH is already v4 # "tesseract" on $PATH is already v4
return os.environ.copy() return os.environ.copy()
@@ -49,6 +48,11 @@ def ensure_tess4():
raise EnvironmentError("Can't find Tesseract 4") raise EnvironmentError("Can't find Tesseract 4")
@pytest.fixture
def ensure_tess4():
return _ensure_tess4()
@contextmanager @contextmanager
def modified_os_environ(env): def modified_os_environ(env):
old_env = os.environ.copy() old_env = os.environ.copy()
@@ -63,8 +67,8 @@ def tess4_available():
""" """
try: try:
# ensure_tess4 locates the tess4 binary we are going to check # _ensure_tess4 locates the tess4 binary we are going to check
env = ensure_tess4() env = _ensure_tess4()
with modified_os_environ(env): with modified_os_environ(env):
# Now jump into this environment and make sure it really is Tess4 # Now jump into this environment and make sure it really is Tess4
return tesseract.v4() and tesseract.has_textonly_pdf() return tesseract.v4() and tesseract.has_textonly_pdf()