Merge branch 'feature/fix-raster-dpi-too-high' into v15

This commit is contained in:
James R. Barlow
2023-09-21 00:05:56 -07:00
17 changed files with 366 additions and 121 deletions
+15 -3
View File
@@ -28,6 +28,17 @@ except AttributeError:
# Pillow 9 shim
Transpose = Image # type: ignore
COLOR_CONVERSION_STRATEGIES = frozenset(
[
'CMYK',
'Gray',
'LeaveColorUnchanged',
'RGB',
'UseDeviceIndependentColor',
]
)
log = logging.getLogger(__name__)
@@ -151,6 +162,7 @@ def generate_pdfa(
output_file: os.PathLike,
*,
compression: str,
color_conversion_strategy: str,
pdf_version: str = '1.5',
pdfa_part: str = '2',
progressbar_class=None,
@@ -200,16 +212,16 @@ def generate_pdfa(
"-dBATCH",
"-dNOPAUSE",
"-dSAFER",
"-dCompatibilityLevel=" + str(pdf_version),
f"-dCompatibilityLevel={str(pdf_version)}",
"-sDEVICE=pdfwrite",
"-dAutoRotatePages=/None",
"-sColorConversionStrategy=" + strategy,
f"-sColorConversionStrategy={color_conversion_strategy}",
]
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
+ compression_args
+ [
"-dJPEGQ=95",
"-dPDFA=" + pdfa_part,
f"-dPDFA={pdfa_part}",
"-dPDFACompatibilityPolicy=1",
"-o",
"-",
+22 -3
View File
@@ -359,8 +359,12 @@ def is_ocr_required(page_context: PageContext) -> bool:
def rasterize_preview(input_file: Path, page_context: PageContext) -> Path:
"""Generate a lower quality preview image."""
output_file = page_context.get_path('rasterize_preview.jpg')
canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options)
page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
canvas_dpi = Resolution(300.0, 300.0).take_min(
[get_canvas_square_dpi(page_context.pageinfo, page_context.options)]
)
page_dpi = Resolution(300.0, 300.0).take_min(
[get_page_square_dpi(page_context.pageinfo, page_context.options)]
)
page_context.plugin_manager.hook.rasterize_pdf_page(
input_file=input_file,
output_file=output_file,
@@ -490,6 +494,21 @@ def rasterize(
canvas_dpi = get_canvas_square_dpi(pageinfo, page_context.options)
page_dpi = get_page_square_dpi(pageinfo, page_context.options)
dpi_profile = pageinfo.page_dpi_profile()
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
log.warning(
"Weight average DPI is %0.1f, max DPI is %0.1f. "
"The discrepancy may indicate a high detail region on this page, "
"but could also indicate a problem with the input PDF file. "
"An image will be rendered at %0.1f DPI.",
dpi_profile.weighted_dpi,
dpi_profile.max_dpi,
dpi_profile.weighted_dpi,
)
canvas_dpi = page_dpi = Resolution(
dpi_profile.weighted_dpi, dpi_profile.weighted_dpi
)
page_context.plugin_manager.hook.rasterize_pdf_page(
input_file=input_file,
output_file=output_file,
@@ -792,7 +811,7 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext) -
pdf_pages=[fix_docinfo_file],
pdfmark=input_ps_stub,
output_file=output_file,
compression=options.pdfa_image_compression,
context=context,
pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3
progressbar_class=(
context.plugin_manager.hook.get_progressbar_class()
-11
View File
@@ -205,16 +205,6 @@ def check_options_ocr_behavior(options: Namespace) -> None:
options.pages = _pages_from_ranges(options.pages)
def check_options_advanced(options: Namespace) -> None:
if options.pdfa_image_compression != 'auto' and not options.output_type.startswith(
'pdfa'
):
log.warning(
"--pdfa-image-compression argument only applies when "
"--output-type is one of 'pdfa', 'pdfa-1', or 'pdfa-2'"
)
def check_options_metadata(options: Namespace) -> None:
docinfo = [options.title, options.author, options.keywords, options.subject]
for s in (m for m in docinfo if m):
@@ -241,7 +231,6 @@ def _check_plugin_invariant_options(options: Namespace) -> None:
check_options_sidecar(options)
check_options_preprocessing(options)
check_options_ocr_behavior(options)
check_options_advanced(options)
check_options_pillow(options)
+41 -2
View File
@@ -18,6 +18,33 @@ log = logging.getLogger(__name__)
BLACKLISTED_GS_VERSIONS = frozenset()
@hookimpl
def add_options(parser):
gs = parser.add_argument_group("Ghostscript", "Advanced control of Ghostscript")
gs.add_argument(
'--color-conversion-strategy',
action='store',
type=str,
metavar='STRATEGY',
choices=ghostscript.COLOR_CONVERSION_STRATEGIES,
default='LeaveColorUnchanged',
help="Set Ghostscript color conversion strategy",
)
gs.add_argument(
'--pdfa-image-compression',
choices=['auto', 'jpeg', 'lossless'],
default='auto',
help="Specify how to compress images in the output PDF/A. 'auto' lets "
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
"JPEG compression. 'lossless' uses PNG-style lossless compression "
"for all images. Monochrome images are always compressed using a "
"lossless codec. Compression settings "
"are applied to all pages, including those for which OCR was "
"skipped. Not supported for --output-type=pdf ; that setting "
"preserves the original compression of all images.",
)
@hookimpl
def check_options(options):
"""Check that the options are valid for this plugin."""
@@ -37,6 +64,17 @@ def check_options(options):
if options.output_type == 'pdfa':
options.output_type = 'pdfa-2'
if options.color_conversion_strategy not in ghostscript.COLOR_CONVERSION_STRATEGIES:
raise ValueError(
f"Invalid color conversion strategy: {options.color_conversion_strategy}"
)
if options.pdfa_image_compression != 'auto' and not options.output_type.startswith(
'pdfa'
):
log.warning(
"--pdfa-image-compression argument only applies when "
"--output-type is one of 'pdfa', 'pdfa-1', or 'pdfa-2'"
)
@hookimpl
@@ -71,7 +109,7 @@ def generate_pdfa(
pdf_pages,
pdfmark,
output_file,
compression,
context,
pdf_version,
pdfa_part,
progressbar_class,
@@ -81,7 +119,8 @@ def generate_pdfa(
ghostscript.generate_pdfa(
pdf_pages=[*pdf_pages, pdfmark],
output_file=output_file,
compression=compression,
compression=context.options.pdfa_image_compression,
color_conversion_strategy=context.options.color_conversion_strategy,
pdf_version=pdf_version,
pdfa_part=pdfa_part,
progressbar_class=progressbar_class,
+15 -3
View File
@@ -30,7 +30,7 @@ def add_options(parser):
action='append',
metavar='CFG',
default=[],
help="Additional Tesseract configuration files -- see documentation",
help="Additional Tesseract configuration files -- see documentation.",
)
tess.add_argument(
'--tesseract-pagesegmode',
@@ -38,7 +38,7 @@ def add_options(parser):
type=int,
metavar='PSM',
choices=range(0, 14),
help="Set Tesseract page segmentation mode (see tesseract --help)",
help="Set Tesseract page segmentation mode (see tesseract --help).",
)
tess.add_argument(
'--tesseract-oem',
@@ -75,7 +75,10 @@ def add_options(parser):
metavar='SECONDS',
help=(
"Give up on OCR after the timeout, but copy the preprocessed page "
"into the final output."
"into the final output. This timeout is only used when using Tesseract "
"for OCR. When Tesseract is used for other operations such as "
"deskewing and orientation, the timeout is controlled by "
"--tesseract-non-ocr-timeout."
),
)
tess.add_argument(
@@ -175,6 +178,15 @@ def validate(pdfinfo, options):
tess_threads = int(os.environ['OMP_THREAD_LIMIT'])
log.debug("Using Tesseract OpenMP thread limit %d", tess_threads)
if (
options.tesseract_downsample_above != 32767
and not options.tesseract_downsample_large_images
):
log.warning(
"The --tesseract-downsample-above argument will have no effect unless "
"--tesseract-downsample-large-images is also given."
)
@hookimpl
def filter_ocr_image(page: PageContext, image: Image.Image) -> Image.Image:
+3 -14
View File
@@ -177,7 +177,9 @@ Online documentation is located at:
'--image-dpi',
metavar='DPI',
type=int,
help="For input image instead of PDF, use this DPI instead of file's.",
help="When the input file is an image, not a PDF, use this DPI instead "
"of the DPI claimed by the input file. If the input does not claim a "
"sensible DPI, this option will be required.",
)
parser.add_argument(
'--output-type',
@@ -402,19 +404,6 @@ Online documentation is located at:
help="Only rotate pages when confidence is above this value (arbitrary "
"units reported by tesseract)",
)
advanced.add_argument(
'--pdfa-image-compression',
choices=['auto', 'jpeg', 'lossless'],
default='auto',
help="Specify how to compress images in the output PDF/A. 'auto' lets "
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
"JPEG compression. 'lossless' uses PNG-style lossless compression "
"for all images. Monochrome images are always compressed using a "
"lossless codec. Compression settings "
"are applied to all pages, including those for which OCR was "
"skipped. Not supported for --output-type=pdf ; that setting "
"preserves the original compression of all images.",
)
advanced.add_argument(
'--fast-web-view',
type=numeric(float, 0),
+40 -10
View File
@@ -15,7 +15,16 @@ from contextlib import suppress
from io import StringIO
from math import isclose, isfinite
from pathlib import Path
from typing import Any, Generic, Sequence, SupportsFloat, SupportsRound, TypeVar
from statistics import harmonic_mean
from typing import (
Any,
Callable,
Generic,
Sequence,
SupportsFloat,
SupportsRound,
TypeVar,
)
import img2pdf
import pikepdf
@@ -73,17 +82,38 @@ class Resolution(Generic[T]):
return isfinite(self.x) and isfinite(self.y)
return True
def to_scalar(self) -> float:
"""Return the harmonic mean of x and y as a 1D approximation.
In most cases, Resolution is 2D, but typically it is "square" (x == y) and
can be approximated as a single number. When not square, the harmonic mean
is used to approximate the 2D resolution as a single number.
"""
return harmonic_mean([self.x, self.y])
def _take_minmax(
self, vals: Iterable[Any], yvals: Iterable[Any], cmp: Callable
) -> Resolution:
"""Return a new Resolution object with the maximum resolution of inputs."""
if yvals is not None:
return Resolution(cmp(self.x, *vals), cmp(self.y, *yvals))
cmp_x, cmp_y = self.x, self.y
for x, y in vals:
cmp_x = cmp(x, cmp_x)
cmp_y = cmp(y, cmp_y)
return Resolution(cmp_x, cmp_y)
def take_max(
self, vals: Iterable[Any], yvals: Iterable[Any] | None = None
) -> Resolution:
"""Return a new Resolution object with the maximum resolution of inputs."""
if yvals is not None:
return Resolution(max(self.x, *vals), max(self.y, *yvals))
max_x, max_y = self.x, self.y
for x, y in vals:
max_x = max(x, max_x)
max_y = max(y, max_y)
return Resolution(max_x, max_y)
return self._take_minmax(vals, yvals, max)
def take_min(
self, vals: Iterable[Any], yvals: Iterable[Any] | None = None
) -> Resolution:
"""Return a new Resolution object with the minimum resolution of inputs."""
return self._take_minmax(vals, yvals, min)
def flip_axis(self) -> Resolution[T]:
"""Return a new Resolution object with x and y swapped."""
@@ -95,11 +125,11 @@ class Resolution(Generic[T]):
def __str__(self):
"""Return a string representation of the resolution."""
return f"{self.x:f}x{self.y:f}"
return f"{self.x:f}×{self.y:f}"
def __repr__(self): # pragma: no cover
"""Return a repr() of the resolution."""
return f"Resolution({self.x}x{self.y} dpi)"
return f"Resolution({self.x!r}, {self.y!r})"
def __eq__(self, other):
"""Return True if the resolution is equal to another resolution."""
+76 -5
View File
@@ -420,12 +420,12 @@ class ImageInfo:
return self._type
@property
def width(self):
def width(self) -> int:
"""Width of the image in pixels."""
return self._width
@property
def height(self):
def height(self) -> int:
"""Height of the image in pixels."""
return self._height
@@ -458,17 +458,24 @@ class ImageInfo:
return self.dpi.is_finite and self.width >= 0 and self.height >= 0
@property
def dpi(self):
def dpi(self) -> Resolution:
"""Dots per inch of the image.
Calculated based on where and how the image is drawn in the PDF.
"""
return _get_dpi(self._shorthand, (self._width, self._height))
@property
def printed_area(self) -> float:
"""Physical area of the image in square inches."""
if not self.renderable:
return 0.0
return float(self.width * self.dpi.x * self.height * self.dpi.y)
def __repr__(self):
"""Return a string representation of the image."""
return (
f"<ImageInfo '{self.name}' {self.type_} {self.width}x{self.height} "
f"<ImageInfo '{self.name}' {self.type_} {self.width}×{self.height} "
f"{self.color} {self.comp} {self.bpc} {self.enc} {self.dpi}>"
)
@@ -747,12 +754,38 @@ def _pdf_pageinfo_concurrent(
return pages
class PageResolutionProfile(NamedTuple):
"""Information about the resolutions of a page."""
weighted_dpi: float
"""The weighted average DPI of the page, weighted by the area of each image."""
max_dpi: float
"""The maximum DPI of an image on the page."""
average_to_max_dpi_ratio: float
"""The average DPI of the page divided by the maximum DPI of the page.
This indicates the intensity of the resolution variation on the page.
If the average is 1.0 or close to 1.0, has all of its content at a uniform
resolution. If the average is much lower than 1.0, some content is at a
higher resolution than the rest of the page.
"""
area_ratio: float
"""The maximum-DPI area of the page divided by the total drawn area.
This indicates the prevalence of high-resolution content on the page.
"""
class PageInfo:
"""Information about type of contents on each page in a PDF."""
_has_text: bool | None
_has_vector: bool | None
_images: list[ImageInfo]
_images: list[ImageInfo] = []
def __init__(
self,
@@ -939,6 +972,44 @@ class PageInfo:
else:
return '1.5'
def page_dpi_profile(self) -> PageResolutionProfile | None:
"""Return information about the DPIs of the page.
This is useful to detect pages with a small proportion of high-resolution
content that is forcing us to use a high DPI for the whole page. The ratio
is weighted by the area of each image. If images overlap, the overlapped
area counts.
Vector graphics and text are ignored.
Returns None if there is no meaningful DPI for the page.
"""
image_dpis = [
image.dpi.to_scalar() for image in self._images if image.renderable
]
image_areas = [image.printed_area for image in self._images if image.renderable]
total_drawn_area = sum(image_areas)
if total_drawn_area == 0:
return None
weights = [area / total_drawn_area for area in image_areas]
# Calculate harmonic mean of DPIs weighted by area
# When the minimum version is Python 3.10, change this to
# statistics.harmonic_mean with the weights parameter
# rather than doing it manually.
weighted_dpi = sum(weights) / sum(
weight / dpi for weight, dpi in zip(weights, image_dpis)
)
max_dpi = max(image_dpis)
dpi_average_max_ratio = weighted_dpi / max_dpi
arg_max_dpi = image_dpis.index(max_dpi)
max_area_ratio = image_areas[arg_max_dpi] / total_drawn_area
return PageResolutionProfile(
weighted_dpi, max_dpi, dpi_average_max_ratio, max_area_ratio
)
def __repr__(self):
"""Return string representation."""
return (
+7 -7
View File
@@ -351,7 +351,6 @@ def filter_pdf_page(page: PageContext, image_filename: Path, output_pdf: Path) -
This hook will be called from child processes. Modifying global state
will not affect the main process or other child processes.
Note:
This is a :ref:`firstresult hook<firstresult>`.
"""
@@ -466,7 +465,7 @@ def generate_pdfa(
pdf_pages: list[Path],
pdfmark: Path,
output_file: Path,
compression: str,
context: PdfContext,
pdf_version: str,
pdfa_part: str,
progressbar_class,
@@ -484,11 +483,7 @@ def generate_pdfa(
pdfmark: A PostScript file intended for Ghostscript with details on
how to perform the PDF/A conversion.
output_file: The name of the desired output file.
compression: One of ``'jpeg'``, ``'lossless'``, ``''``. For ``'jpeg'``,
the PDF/A generator should convert all images to JPEG encoding where
possible. For lossless, all images should be converted to FlateEncode
(lossless PNG). If an empty string, the PDF generator should make its
own decisions about how to encode images.
context: The current context.
pdf_version: The minimum PDF version that the output file should be.
At its own discretion, the PDF/A generator may raise the version,
but should not lower it.
@@ -514,6 +509,11 @@ def generate_pdfa(
Note:
This is a :ref:`firstresult hook<firstresult>`.
Note:
Before version 15.0.0, the ``context`` was not provided and ``compression``
was provided instead. Plugins should now read the context object to determine
if compression is requested.
See Also:
https://github.com/tqdm/tqdm
"""