Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3a75b20740 | ||
|
|
d35d008806 | ||
|
|
f5662d5eb0 | ||
|
|
39010dd255 | ||
|
|
fbaad570c7 | ||
|
|
f974e3b3c1 | ||
|
|
46b49cc176 | ||
|
|
5256e74d0c | ||
|
|
621d6a0b89 | ||
|
|
08be7c8bbe | ||
|
|
980a5472b6 | ||
|
|
51c618e357 | ||
|
|
4dde3786c2 | ||
|
|
d544342602 | ||
|
|
fac91fca2a | ||
|
|
6edf756849 | ||
|
|
4fb1bb4de6 | ||
|
|
6a8eb7daaa | ||
|
|
0544d06c3d |
@@ -228,6 +228,59 @@ then run ocrmypdf as follows (along with any other desired arguments):
|
||||
Some combinations of control parameters will break Tesseract or break
|
||||
assumptions that OCRmyPDF makes about Tesseract's output.
|
||||
|
||||
Changing page segmentation mode
|
||||
-------------------------------
|
||||
|
||||
The directive ``--tesseract-pagesegmode Nmode`` forwards the desired page segmentation
|
||||
mode to Tesseract OCR. The default is 3.
|
||||
|
||||
Page segmentation can improve OCR results when you know that a PDF ought to be
|
||||
analyzed a particular way, such as PDFs whose pages contain only a single line of
|
||||
text. For the vast majority of users, changing the page segmentation mode will only
|
||||
make things worse.
|
||||
|
||||
As of June 2024, the Tesseract page segmentation modes are:
|
||||
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| ID | Description |
|
||||
+=====+==================================================================================+
|
||||
| 0 | Orientation and script detection (OSD) only. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 1 | Automatic page segmentation with OSD. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 2 | Automatic page segmentation, but no OSD, or OCR. (not implemented) |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 3 | Fully automatic page segmentation, but no OSD. (Default) |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 4 | Assume a single column of text of variable sizes. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 5 | Assume a single uniform block of vertically aligned text. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 6 | Assume a single uniform block of text. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 7 | Treat the image as a single text line. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 8 | Treat the image as a single word. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 9 | Treat the image as a single word in a circle. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 10 | Treat the image as a single character. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 11 | Sparse text. Find as much text as possible in no particular order. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 12 | Sparse text with OSD. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 13 | Raw line. Treat the image as a single text line, bypassing hacks that are |
|
||||
| | Tesseract-specific. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
|
||||
Modes 0, 1, 2, and 12 (all of those that enable orientation and script detection)
|
||||
are not compatible with OCRmyPDF, which performs OSD in a separate step from OCR.
|
||||
Their use may interfere with ``--rotate-pages`` and other features.
|
||||
|
||||
It is currently not possible to use advanced Tesseract OCR features, such as creating
|
||||
OCR information, when using Tesseract through OCRmyPDF.
|
||||
|
||||
Changing the PDF renderer
|
||||
=========================
|
||||
|
||||
|
||||
@@ -224,7 +224,7 @@ standard tooling needed to build packages, such as a compiler and binary tools.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo pacman -S base-devel
|
||||
sudo pacman -S --needed base-devel
|
||||
|
||||
Now you are ready to install the OCRmyPDF package.
|
||||
|
||||
@@ -681,7 +681,7 @@ To install all of the development and test requirements:
|
||||
.. code-block:: bash
|
||||
|
||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
python -m .venv
|
||||
python -m venv .venv
|
||||
source .venv/bin/activate
|
||||
cd OCRmyPDF
|
||||
pip install -e .[test]
|
||||
|
||||
@@ -30,6 +30,35 @@ OCRmyPDF typically supports the three most recent Python versions.
|
||||
|
||||
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||
|
||||
v16.4.3
|
||||
=======
|
||||
|
||||
- Work around pdfminer.six issue where a token on the buffer boundary is incorrectly
|
||||
parsed as two tokens. :issue:`1361`
|
||||
- New rules are applied to stencil masks and explicit masks when calculating the
|
||||
optimal page DPI for rendering. :issue:`1362`
|
||||
- Fixed attempts to use an incompatible jbig2.EXE provided by TeX Live. :issue:`1363`
|
||||
|
||||
v16.4.2
|
||||
=======
|
||||
|
||||
- Fixed order of filenames passed to Ghostscript for PDF/A generation. :issue:`1359`
|
||||
- Suppressed missing jbig2dec warning message. :issue:`1358`
|
||||
- Fixed calculation of image size when soft mask dimensions don't match image
|
||||
dimension. :issue:`1351`
|
||||
- Several fixes to documentation. Thanks to users Iris and JoKalliauer
|
||||
who contributed these changes.
|
||||
- Fixed error on processing PDFs that are missing certain image metadata. :issue:`1315`
|
||||
|
||||
v16.4.1
|
||||
=======
|
||||
|
||||
- Fixed calculation of image printed area (used in finding weighted DPI for OCR).
|
||||
:issue:`1334`
|
||||
- Fixed "NotImplementedError: not sure how to get colorspace" error
|
||||
messages in logs which simply records a failure to optimize images with
|
||||
print production colorspaces. :issue:`1315`
|
||||
|
||||
v16.4.0
|
||||
=======
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from subprocess import PIPE
|
||||
from subprocess import PIPE, CalledProcessError
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
@@ -14,7 +14,13 @@ from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*'))
|
||||
try:
|
||||
version = get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
except CalledProcessError as e:
|
||||
# TeX Live for Windows provides an incompatible jbig2.EXE which may
|
||||
# be on the PATH.
|
||||
raise MissingDependencyError('jbig2enc') from e
|
||||
return Version(version)
|
||||
|
||||
|
||||
def available():
|
||||
|
||||
@@ -475,7 +475,7 @@ def calculate_raster_dpi(page_context: PageContext):
|
||||
page_dpi = get_page_square_dpi(page_context, image_dpi)
|
||||
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
|
||||
log.warning(
|
||||
"Weight average image DPI is %0.1f, max DPI is %0.1f. "
|
||||
"Weighted average image DPI is %0.1f, max DPI is %0.1f. "
|
||||
"The discrepancy may indicate a high detail region on this page, "
|
||||
"but could also indicate a problem with the input PDF file. "
|
||||
"Page image will be rendered at %0.1f DPI.",
|
||||
|
||||
@@ -129,7 +129,7 @@ def generate_pdfa(
|
||||
):
|
||||
"""Generate a PDF/A from the list of PDF pages and PDF/A metadata."""
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[*pdf_pages, pdfmark],
|
||||
pdf_pages=[pdfmark, *pdf_pages],
|
||||
output_file=output_file,
|
||||
compression=context.options.pdfa_image_compression,
|
||||
color_conversion_strategy=context.options.color_conversion_strategy,
|
||||
|
||||
@@ -268,7 +268,9 @@ def check_pdf(input_file: Path) -> bool:
|
||||
return False
|
||||
else:
|
||||
with pdf:
|
||||
messages = pdf.check()
|
||||
with warnings.catch_warnings():
|
||||
warnings.filterwarnings('ignore', message=r'pikepdf.*JBIG2.*')
|
||||
messages = pdf.check()
|
||||
success = True
|
||||
for msg in messages:
|
||||
if 'error' in msg.lower():
|
||||
|
||||
@@ -28,6 +28,7 @@ from pikepdf import (
|
||||
Stream,
|
||||
UnsupportedImageTypeError,
|
||||
)
|
||||
from pikepdf.models.image import HifiPrintImageNotTranscodableError
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||
@@ -200,7 +201,7 @@ def extract_image_generic(
|
||||
with imgname.open('wb') as f:
|
||||
ext = pim.extract_to(stream=f)
|
||||
imgname.rename(imgname.with_suffix(ext))
|
||||
except UnsupportedImageTypeError:
|
||||
except (UnsupportedImageTypeError, HifiPrintImageNotTranscodableError):
|
||||
return None
|
||||
return XrefExt(xref, ext)
|
||||
elif (
|
||||
|
||||
@@ -369,8 +369,21 @@ class ImageInfo:
|
||||
pim = PdfImage(pdfimage)
|
||||
else:
|
||||
raise ValueError("Either pdfimage or inline must be set")
|
||||
|
||||
self._width = pim.width
|
||||
self._height = pim.height
|
||||
if (smask := pim.obj.get(Name.SMask, None)) is not None:
|
||||
# SMask is pretty much an alpha channel, but in PDF it's possible
|
||||
# for channel to have different dimensions than the image
|
||||
# itself. Some PDF writers use this to create a grayscale stencil
|
||||
# mask. For our purposes, the effective size is the size of the
|
||||
# larger component (image or smask).
|
||||
self._width = max(smask.get(Name.Width, 0), self._width)
|
||||
self._height = max(smask.get(Name.Height, 0), self._height)
|
||||
if (mask := pim.obj.get(Name.Mask, None)) is not None:
|
||||
# If the image has a /Mask entry, it has an explicit mask.
|
||||
self._width = max(mask.get(Name.Width, 0), self._width)
|
||||
self._height = max(mask.get(Name.Height, 0), self._height)
|
||||
|
||||
# If /ImageMask is true, then this image is a stencil mask
|
||||
# (Images that draw with this stencil mask will have a reference to
|
||||
@@ -474,9 +487,18 @@ class ImageInfo:
|
||||
def renderable(self) -> bool:
|
||||
"""Whether the image is renderable.
|
||||
|
||||
Some PDFs in the wild have invalid images that are not renderable.
|
||||
Some PDFs in the wild have invalid images that are not renderable,
|
||||
due to unusual dimensions.
|
||||
|
||||
Stencil masks are not also not renderable, since they are not
|
||||
drawn, but rather they control how rendering happens.
|
||||
"""
|
||||
return self.dpi.is_finite and self.width >= 0 and self.height >= 0
|
||||
return (
|
||||
self.dpi.is_finite
|
||||
and self.width >= 0
|
||||
and self.height >= 0
|
||||
and self.type_ != 'stencil'
|
||||
)
|
||||
|
||||
@property
|
||||
def dpi(self) -> Resolution:
|
||||
@@ -491,7 +513,7 @@ class ImageInfo:
|
||||
"""Physical area of the image in square inches."""
|
||||
if not self.renderable:
|
||||
return 0.0
|
||||
return float(self.width * self.dpi.x * self.height * self.dpi.y)
|
||||
return float((self.width / self.dpi.x) * (self.height / self.dpi.y))
|
||||
|
||||
def __repr__(self):
|
||||
"""Return a string representation of the image."""
|
||||
@@ -573,7 +595,7 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
||||
xobjs = resources[Name.XObject].as_dict()
|
||||
for xobj in xobjs:
|
||||
candidate = xobjs[xobj]
|
||||
if candidate is None or candidate[Name.Subtype] != Name.Form:
|
||||
if candidate is None or candidate.get(Name.Subtype) != Name.Form:
|
||||
continue
|
||||
|
||||
form_xobject = candidate
|
||||
@@ -1055,10 +1077,14 @@ class PageInfo:
|
||||
|
||||
Returns None if there is no meaningful DPI for the page.
|
||||
"""
|
||||
image_dpis = [
|
||||
image.dpi.to_scalar() for image in self._images if image.renderable
|
||||
]
|
||||
image_areas = [image.printed_area for image in self._images if image.renderable]
|
||||
image_dpis = []
|
||||
image_areas = []
|
||||
for image in self._images:
|
||||
if not image.renderable:
|
||||
continue
|
||||
image_dpis.append(image.dpi.to_scalar())
|
||||
image_areas.append(image.printed_area)
|
||||
|
||||
total_drawn_area = sum(image_areas)
|
||||
if total_drawn_area == 0:
|
||||
return None
|
||||
@@ -1071,7 +1097,6 @@ class PageInfo:
|
||||
|
||||
arg_max_dpi = image_dpis.index(max_dpi)
|
||||
max_area_ratio = image_areas[arg_max_dpi] / total_drawn_area
|
||||
|
||||
return PageResolutionProfile(
|
||||
weighted_dpi,
|
||||
max_dpi,
|
||||
|
||||
@@ -17,6 +17,7 @@ import pdfminer
|
||||
import pdfminer.encodingdb
|
||||
import pdfminer.pdfdevice
|
||||
import pdfminer.pdfinterp
|
||||
import pdfminer.psparser
|
||||
from pdfminer.converter import PDFLayoutAnalyzer
|
||||
from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox
|
||||
from pdfminer.pdfcolor import PDFColorSpace
|
||||
@@ -58,9 +59,10 @@ def pdfsimplefont__init__(
|
||||
|
||||
setattr(PDFSimpleFont, '__init__', pdfsimplefont__init__)
|
||||
|
||||
#
|
||||
# pdfminer patches when creator is PScript5.dll
|
||||
#
|
||||
# Patch pdfminer.six buffer size
|
||||
# The parser doesn't properly handle keyword tokens are split across the end of the
|
||||
# buffer, so increase the buffer size something far larger than will ever be seen.
|
||||
pdfminer.psparser.PSBaseParser.BUFSIZ = 256 * 1024 * 1024
|
||||
|
||||
|
||||
def pdftype3font__pscript5_get_height(self):
|
||||
|
||||
Reference in New Issue
Block a user