From 190bfe88599af65a61e02ef15748cdc684728bfe Mon Sep 17 00:00:00 2001 From: "James R. Barlow" Date: Wed, 3 Oct 2018 17:39:50 -0700 Subject: [PATCH] Fix suppression of tesseract config error messages Cherry-pick and merge from fb8b161 --- src/ocrmypdf/__main__.py | 27 ++++++++++++++++++--------- src/ocrmypdf/exec/tesseract.py | 15 ++++++++++----- 2 files changed, 28 insertions(+), 14 deletions(-) diff --git a/src/ocrmypdf/__main__.py b/src/ocrmypdf/__main__.py index dc89d303..77cd0530 100755 --- a/src/ocrmypdf/__main__.py +++ b/src/ocrmypdf/__main__.py @@ -587,9 +587,20 @@ def do_ruffus_exception(ruffus_five_tuple, options, log): description of the error message that occurred.""" exit_code = None - task_name, job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple - job_name = job_name # unused - if exc_name == 'builtins.SystemExit': + _task_name, _job_name, exc_name, exc_value, exc_stack = ruffus_five_tuple + + if isinstance(exc_name, type): + # ruffus is full of mystery... sometimes (probably when the process + # group leader is killed) exc_name is the class object of the exception, + # rather than a str. So reach into the object and get its name. + exc_name = exc_name.__name__ + + if exc_name.startswith('ocrmypdf.exceptions.'): + base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '') + exc_class = getattr(ocrmypdf_exceptions, base_exc_name) + exit_code = getattr(exc_class, 'exit_code', ExitCode.other_error) + + if exc_name in ('builtins.SystemExit', 'SystemExit'): match = re.search(r"\.(.+?)\)", exc_value) exit_code_name = match.groups()[0] exit_code = getattr(ExitCode, exit_code_name, 'other_error') @@ -625,7 +636,6 @@ def do_ruffus_exception(ruffus_five_tuple, options, log): qpdf --decrypt [--password=[password]] infilename """)) - exit_code = ExitCode.encrypted_pdf elif exc_name == 'ocrmypdf.exceptions.PdfMergeFailedError': log.error(textwrap.dedent("""\ Failed to merge PDF image layer with OCR layer @@ -636,11 +646,10 @@ def do_ruffus_exception(ruffus_five_tuple, options, log): Try using ocrmypdf --pdf-renderer tesseract [..other args..] """)) - exit_code = ExitCode.input_file - elif exc_name.startswith('ocrmypdf.exceptions.'): - base_exc_name = exc_name.replace('ocrmypdf.exceptions.', '') - exc_class = getattr(ocrmypdf_exceptions, base_exc_name) - exit_code = exc_class.exit_code + elif exc_name == 'ocrmypdf.exceptions.TesseractConfigError': + log.error(textwrap.dedent("""\ + Error occurred while parsing a tesseract configuration file + """)) elif exc_name == 'PIL.Image.DecompressionBombError': msg = cleanup_ruffus_error_message(exc_value) msg += ("\nUse the --max-image-mpixels argument to set increase the " diff --git a/src/ocrmypdf/exec/tesseract.py b/src/ocrmypdf/exec/tesseract.py index 2b156315..0575793f 100644 --- a/src/ocrmypdf/exec/tesseract.py +++ b/src/ocrmypdf/exec/tesseract.py @@ -67,8 +67,8 @@ def v4(): @lru_cache(maxsize=1) def has_textonly_pdf(): """Does Tesseract have textonly_pdf capability? - - Available in 3.05.01, and v4.00.00alpha since January 2017. Best to + + Available in 3.05.01, and v4.00.00alpha since January 2017. Best to parse the parameter list """ args_tess = [ @@ -191,13 +191,18 @@ def tesseract_log_output(log, stdout, input_file): log.warning(prefix + "lots of diacritics - possibly poor OCR") elif line.startswith('OSD: Weak margin'): log.warning(prefix + "unsure about page orientation") + elif 'Error in pixScanForForeground' in line: + pass # Appears to be spurious/problem with nonwhite borders + elif 'Error in boxClipToRectangle' in line: + pass # Always appears with pixScanForForeground message + elif 'parameter not found: ' in line.lower(): + log.error(prefix + line.strip()) + problem = line.split('found: ')[1] + raise TesseractConfigError(problem) elif 'error' in line.lower() or 'exception' in line.lower(): log.error(prefix + line.strip()) elif 'warning' in line.lower(): log.warning(prefix + line.strip()) - elif 'parameter not found: ' in line.lower(): - problem = line.split('found: ')[1] - raise TesseractConfigError(problem) elif 'read_params_file' in line.lower(): log.error(prefix + line.strip()) else: