diff --git a/docs/installation.rst b/docs/installation.rst index dd53dc88..cb006566 100644 --- a/docs/installation.rst +++ b/docs/installation.rst @@ -616,8 +616,8 @@ manager. ``pip`` cannot provide them. The following versions are required: -- Python 3.7 or newer -- Ghostscript 9.23 or newer +- Python 3.8 or newer +- Ghostscript 9.50 or newer - Tesseract 4.0.0 or newer - jbig2enc 0.29 or newer - pngquant 2.5 or newer diff --git a/docs/introduction.rst b/docs/introduction.rst index 7c11818b..563844bb 100644 --- a/docs/introduction.rst +++ b/docs/introduction.rst @@ -190,8 +190,7 @@ Ghostscript also imposes some limitations: behavior can be suppressed by setting ``--pdfa-image-compression`` to ``jpeg`` or ``lossless`` to set all images to one type or the other. Ghostscript has no option to maintain the input image's format. - (Ghostscript 9.25+ can copy JPEG images without transcoding them; - earlier versions will transcode.) + (Modern Ghostscript can copy JPEG images without transcoding them.) - Ghostscript's PDF/A conversion removes any XMP metadata that is not one of the standard XMP metadata namespaces for PDFs. In particular, PRISM Metdata is removed. diff --git a/src/ocrmypdf/_exec/ghostscript.py b/src/ocrmypdf/_exec/ghostscript.py index 9e21c33c..d67044e7 100644 --- a/src/ocrmypdf/_exec/ghostscript.py +++ b/src/ocrmypdf/_exec/ghostscript.py @@ -47,21 +47,6 @@ def version(): return get_version(GS) -def jpeg_passthrough_available() -> bool: - """Returns True if the installed version of Ghostscript supports JPEG passthru - - Prior to 9.23, Ghostscript decoded and re-encoded JPEGs internally. In 9.23 - it gained the ability to keep JPEGs unmodified. However, the 9.23 - implementation was buggy and would deletes the last two bytes of images in - some cases, as reported here. - https://bugs.ghostscript.com/show_bug.cgi?id=699216 - - The issue was fixed for 9.24, hence that is the first version we consider - the feature available. (Ghostscript 9.24 has its own problems is blacklisted.) - """ - return version() >= '9.24' - - def _gs_error_reported(stream) -> bool: match = re.search(r'error', stream, flags=re.IGNORECASE) return bool(match) @@ -201,20 +186,9 @@ def generate_pdfa( ] strategy = 'LeaveColorUnchanged' - # Older versions of Ghostscript expect a leading slash in - # sColorConversionStrategy, newer ones should not have it. See Ghostscript - # git commit fe1c025d. gs_version = version() - strategy = ('/' + strategy) if gs_version < '9.19' else strategy - - if gs_version == '9.23': - # 9.23: added JPEG passthrough as a new feature, but with a bug that - # incorrectly formats some images. Fixed as of 9.24. So we disable this - # feature for 9.23. - # https://bugs.ghostscript.com/show_bug.cgi?id=699216 - compression_args.append('-dPassThroughJPEGImages=false') - elif gs_version == '9.56.0': - # 9.56.0 breaks our OCR...? + if gs_version == '9.56.0': + # 9.56.0 introduced a new rendering mode that breaks our OCR compression_args.append('-dNEWPDF=false') # nb no need to specify ProcessColorModel when ColorConversionStrategy diff --git a/src/ocrmypdf/builtin_plugins/ghostscript.py b/src/ocrmypdf/builtin_plugins/ghostscript.py index 64375ad8..7ab8d7c9 100644 --- a/src/ocrmypdf/builtin_plugins/ghostscript.py +++ b/src/ocrmypdf/builtin_plugins/ghostscript.py @@ -21,37 +21,19 @@ def check_options(options): program='gs', package='ghostscript', version_checker=ghostscript.version, - need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports + need_version='9.50', # Ubuntu 20.04's version ) gs_version = ghostscript.version() - if gs_version in ('9.24', '9.51'): + if gs_version in ('9.51',): raise MissingDependencyError( f"Ghostscript {gs_version} contains serious regressions and is not " "supported. Please upgrade to a newer version, or downgrade to the " "previous version." ) - # We have these constraints to check for. - # 1. Ghostscript < 9.20 mangles multibyte Unicode - # 2. hocr doesn't work on non-Latin languages (so don't select it) - is_latin = options.languages.issubset(HOCR_OK_LANGS) - if gs_version < '9.20' and options.output_type != 'pdf' and not is_latin: - # https://bugs.ghostscript.com/show_bug.cgi?id=696874 - # Ghostscript < 9.20 fails to encode multibyte characters properly - log.warning( - f"The installed version of Ghostscript ({gs_version}) does not work " - "correctly with the OCR languages you specified. Use --output-type pdf or " - "upgrade to Ghostscript 9.20 or later to avoid this issue." - ) - if options.output_type == 'pdfa': options.output_type = 'pdfa-2' - if options.output_type == 'pdfa-3' and ghostscript.version() < '9.19': - raise MissingDependencyError( - "--output-type pdfa-3 requires Ghostscript 9.19 or later" - ) - @hookimpl def rasterize_pdf_page( diff --git a/tests/plugins/gs_feature_elision.py b/tests/plugins/gs_feature_elision.py index 0f63c502..ce829f5f 100644 --- a/tests/plugins/gs_feature_elision.py +++ b/tests/plugins/gs_feature_elision.py @@ -9,13 +9,13 @@ from ocrmypdf import hookimpl from ocrmypdf.builtin_plugins import ghostscript from ocrmypdf.subprocess import run_polling_stderr -elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1 +ELISION_WARNING = """GPL Ghostscript 9.50: Setting Overprint Mode to 1 not permitted in PDF/A-2, overprint mode not set""" def run_append_stderr(*args, **kwargs): proc = run_polling_stderr(*args, **kwargs) - proc.stderr += '\n' + elision_warning + '\n' + proc.stderr += '\n' + ELISION_WARNING + '\n' return proc diff --git a/tests/test_main.py b/tests/test_main.py index 739df7b7..c975665b 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -716,11 +716,9 @@ def test_compression_changed(ocrmypdf_exec, resources, image, compression, outpd if compression == "jpeg": assert pdfimage.enc == Encoding.jpeg else: - if ghostscript.jpeg_passthrough_available(): - # Ghostscript 9.23 adds JPEG passthrough, which allows a JPEG to be - # copied without transcoding - so report - if image.endswith('jpg'): - assert pdfimage.enc == Encoding.jpeg + if image.endswith('jpg'): + # Ghostscript JPEG passthrough - no issue + assert pdfimage.enc == Encoding.jpeg else: assert pdfimage.enc not in (Encoding.jpeg, Encoding.jpeg2000) diff --git a/tests/test_validation.py b/tests/test_validation.py index fca0cc32..ee29f312 100644 --- a/tests/test_validation.py +++ b/tests/test_validation.py @@ -48,22 +48,6 @@ def test_hocr_notlatin_warning(caplog): assert 'PDF renderer is known to cause' in caplog.text -def test_old_ghostscript(caplog): - with patch('ocrmypdf._exec.ghostscript.version', return_value='9.19'), patch( - 'ocrmypdf._exec.tesseract.get_languages', return_value={'eng', 'chi_sim'} - ): - vd.check_options(*make_opts_pm(language='chi_sim', output_type='pdfa')) - assert 'does not work correctly' in caplog.text - - with patch('ocrmypdf._exec.ghostscript.version', return_value='9.18'): - with pytest.raises(MissingDependencyError): - vd.check_options(*make_opts_pm(output_type='pdfa-3')) - - with patch('ocrmypdf._exec.ghostscript.version', return_value='9.24'): - with pytest.raises(MissingDependencyError): - vd.check_options(*make_opts_pm()) - - def test_old_tesseract_error(): with patch('ocrmypdf._exec.tesseract.version', return_value='4.00.00alpha'): with pytest.raises(MissingDependencyError):