Merge branch 'feature/fix-raster-dpi-too-high' into v15

This commit is contained in:
James R. Barlow
2023-09-21 00:05:56 -07:00
17 changed files with 366 additions and 121 deletions
+41 -2
View File
@@ -18,6 +18,33 @@ log = logging.getLogger(__name__)
BLACKLISTED_GS_VERSIONS = frozenset()
@hookimpl
def add_options(parser):
gs = parser.add_argument_group("Ghostscript", "Advanced control of Ghostscript")
gs.add_argument(
'--color-conversion-strategy',
action='store',
type=str,
metavar='STRATEGY',
choices=ghostscript.COLOR_CONVERSION_STRATEGIES,
default='LeaveColorUnchanged',
help="Set Ghostscript color conversion strategy",
)
gs.add_argument(
'--pdfa-image-compression',
choices=['auto', 'jpeg', 'lossless'],
default='auto',
help="Specify how to compress images in the output PDF/A. 'auto' lets "
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
"JPEG compression. 'lossless' uses PNG-style lossless compression "
"for all images. Monochrome images are always compressed using a "
"lossless codec. Compression settings "
"are applied to all pages, including those for which OCR was "
"skipped. Not supported for --output-type=pdf ; that setting "
"preserves the original compression of all images.",
)
@hookimpl
def check_options(options):
"""Check that the options are valid for this plugin."""
@@ -37,6 +64,17 @@ def check_options(options):
if options.output_type == 'pdfa':
options.output_type = 'pdfa-2'
if options.color_conversion_strategy not in ghostscript.COLOR_CONVERSION_STRATEGIES:
raise ValueError(
f"Invalid color conversion strategy: {options.color_conversion_strategy}"
)
if options.pdfa_image_compression != 'auto' and not options.output_type.startswith(
'pdfa'
):
log.warning(
"--pdfa-image-compression argument only applies when "
"--output-type is one of 'pdfa', 'pdfa-1', or 'pdfa-2'"
)
@hookimpl
@@ -71,7 +109,7 @@ def generate_pdfa(
pdf_pages,
pdfmark,
output_file,
compression,
context,
pdf_version,
pdfa_part,
progressbar_class,
@@ -81,7 +119,8 @@ def generate_pdfa(
ghostscript.generate_pdfa(
pdf_pages=[*pdf_pages, pdfmark],
output_file=output_file,
compression=compression,
compression=context.options.pdfa_image_compression,
color_conversion_strategy=context.options.color_conversion_strategy,
pdf_version=pdf_version,
pdfa_part=pdfa_part,
progressbar_class=progressbar_class,
+15 -3
View File
@@ -30,7 +30,7 @@ def add_options(parser):
action='append',
metavar='CFG',
default=[],
help="Additional Tesseract configuration files -- see documentation",
help="Additional Tesseract configuration files -- see documentation.",
)
tess.add_argument(
'--tesseract-pagesegmode',
@@ -38,7 +38,7 @@ def add_options(parser):
type=int,
metavar='PSM',
choices=range(0, 14),
help="Set Tesseract page segmentation mode (see tesseract --help)",
help="Set Tesseract page segmentation mode (see tesseract --help).",
)
tess.add_argument(
'--tesseract-oem',
@@ -75,7 +75,10 @@ def add_options(parser):
metavar='SECONDS',
help=(
"Give up on OCR after the timeout, but copy the preprocessed page "
"into the final output."
"into the final output. This timeout is only used when using Tesseract "
"for OCR. When Tesseract is used for other operations such as "
"deskewing and orientation, the timeout is controlled by "
"--tesseract-non-ocr-timeout."
),
)
tess.add_argument(
@@ -175,6 +178,15 @@ def validate(pdfinfo, options):
tess_threads = int(os.environ['OMP_THREAD_LIMIT'])
log.debug("Using Tesseract OpenMP thread limit %d", tess_threads)
if (
options.tesseract_downsample_above != 32767
and not options.tesseract_downsample_large_images
):
log.warning(
"The --tesseract-downsample-above argument will have no effect unless "
"--tesseract-downsample-large-images is also given."
)
@hookimpl
def filter_ocr_image(page: PageContext, image: Image.Image) -> Image.Image: