Compare commits

...
6 Commits
7 changed files with 142 additions and 114 deletions
+6
View File
@@ -12,6 +12,12 @@ may be unreliable. Use the API to depend on precise behavior.
The public API may be useful in scripts that launch OCRmyPDF processes or that The public API may be useful in scripts that launch OCRmyPDF processes or that
wish to use some of its features for working with PDFs. wish to use some of its features for working with PDFs.
v11.4.2
=======
- Fixed support for Cygwin, hopefully.
- watcher.py: Fixed an issue with the OCR_LOGLEVEL not being interpreted.
v11.4.1 v11.4.1
======= =======
+8 -3
View File
@@ -44,7 +44,7 @@ DESKEW = bool(os.getenv('OCR_DESKEW', ''))
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1')) POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
USE_POLLING = bool(os.getenv('OCR_USE_POLLING', '')) USE_POLLING = bool(os.getenv('OCR_USE_POLLING', ''))
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper() LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
PATTERNS = ['*.pdf', '*.PDF'] PATTERNS = ['*.pdf', '*.PDF']
log = logging.getLogger('ocrmypdf-watcher') log = logging.getLogger('ocrmypdf-watcher')
@@ -117,7 +117,12 @@ class HandleObserverEvent(PatternMatchingEventHandler):
def main(): def main():
ocrmypdf.configure_logging( ocrmypdf.configure_logging(
verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True verbosity=(
ocrmypdf.Verbosity.default
if LOGLEVEL != 'DEBUG'
else ocrmypdf.Verbosity.debug
),
manage_root_logger=True,
) )
log.setLevel(LOGLEVEL) log.setLevel(LOGLEVEL)
log.info( log.info(
@@ -135,7 +140,7 @@ def main():
f"ARGS: {OCR_JSON_SETTINGS}\n" f"ARGS: {OCR_JSON_SETTINGS}\n"
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
f"USE_POLLING: {USE_POLLING}\n" f"USE_POLLING: {USE_POLLING}\n"
f"LOGLEVEL: {LOGLEVEL}\n" f"LOGLEVEL: {LOGLEVEL}"
) )
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS: if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS:
+4 -2
View File
@@ -12,6 +12,7 @@ import os
import signal import signal
import sys import sys
import threading import threading
from contextlib import suppress
from multiprocessing import Pool as ProcessPool from multiprocessing import Pool as ProcessPool
from multiprocessing.dummy import Pool as ThreadPool from multiprocessing.dummy import Pool as ThreadPool
from typing import Callable, Iterable, Optional from typing import Callable, Iterable, Optional
@@ -56,7 +57,7 @@ def process_init(queue, user_init, loglevel):
signal.signal(signal.SIGINT, signal.SIG_IGN) signal.signal(signal.SIGINT, signal.SIG_IGN)
# Install SIGBUS handler (so our parent process can abort somewhat gracefully) # Install SIGBUS handler (so our parent process can abort somewhat gracefully)
if hasattr(signal, 'SIGBUS'): with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS
signal.signal(signal.SIGBUS, process_sigbus) signal.signal(signal.SIGBUS, process_sigbus)
# Reconfigure the root logger for this process to send all messages to a queue # Reconfigure the root logger for this process to send all messages to a queue
@@ -72,7 +73,8 @@ def process_init(queue, user_init, loglevel):
def thread_init(_queue, user_init, _loglevel): def thread_init(_queue, user_init, _loglevel):
# As a thread, block SIGBUS so the main thread deals with it... # As a thread, block SIGBUS so the main thread deals with it...
if hasattr(signal, 'SIGBUS'): with suppress(AttributeError):
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS}) signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS})
if user_init: if user_init:
user_init() user_init()
+1 -1
View File
@@ -150,7 +150,7 @@ def get_pdfinfo(
progbar=False, progbar=False,
max_workers=None, max_workers=None,
check_pages=None, check_pages=None,
): ) -> PdfInfo:
try: try:
return PdfInfo( return PdfInfo(
input_file, input_file,
+117 -103
View File
@@ -15,11 +15,11 @@ from functools import partial
from math import hypot, isclose from math import hypot, isclose
from os import PathLike from os import PathLike
from pathlib import Path from pathlib import Path
from typing import Any, Dict, List, Optional, Union from typing import Any, Container, Dict, Iterator, List, Optional, Tuple, Union
from warnings import warn from warnings import warn
import pikepdf import pikepdf
from pikepdf import PdfMatrix from pikepdf import Object, Pdf, PdfMatrix
from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._concurrent import exec_progress_pool
from ocrmypdf.exceptions import EncryptedPdfError from ocrmypdf.exceptions import EncryptedPdfError
@@ -115,7 +115,7 @@ def _normalize_stack(graphobjs):
yield (operands, operator) yield (operands, operator)
def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
"""Interpret the PDF content stream. """Interpret the PDF content stream.
The stack represents the state of the PDF graphics stack. We are only The stack represents the state of the PDF graphics stack. We are only
@@ -204,7 +204,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
) )
def _get_dpi(ctm_shorthand, image_size): def _get_dpi(ctm_shorthand, image_size) -> Resolution:
"""Given the transformation matrix and image size, find the image DPI. """Given the transformation matrix and image size, find the image DPI.
PDFs do not include image resolution information within image data. PDFs do not include image resolution information within image data.
@@ -271,8 +271,14 @@ def _get_dpi(ctm_shorthand, image_size):
class ImageInfo: class ImageInfo:
DPI_PREC = Decimal('1.000') DPI_PREC = Decimal('1.000')
def __init__(self, *, name='', pdfimage=None, inline=None, shorthand=None): def __init__(
self,
*,
name='',
pdfimage: Optional[Object] = None,
inline: Optional[Object] = None,
shorthand=None,
):
self._name = str(name) self._name = str(name)
self._shorthand = shorthand self._shorthand = shorthand
@@ -282,6 +288,8 @@ class ImageInfo:
elif pdfimage is not None: elif pdfimage is not None:
self._origin = 'xobject' self._origin = 'xobject'
pim = pikepdf.PdfImage(pdfimage) pim = pikepdf.PdfImage(pdfimage)
else:
raise ValueError("Either pdfimage or inline must be set")
self._width = pim.width self._width = pim.width
self._height = pim.height self._height = pim.height
@@ -371,7 +379,7 @@ class ImageInfo:
).format(**class_locals) ).format(**class_locals)
def _find_inline_images(contentsinfo): def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
"Find inline images in the contentstream" "Find inline images in the contentstream"
for n, inline in enumerate(contentsinfo.inline_images): for n, inline in enumerate(contentsinfo.inline_images):
@@ -380,7 +388,7 @@ def _find_inline_images(contentsinfo):
) )
def _image_xobjects(container): def _image_xobjects(container) -> Iterator[Tuple[Object, str]]:
"""Search for all XObject-based images in the container """Search for all XObject-based images in the container
Usually the container is a page, but it could also be a Form XObject Usually the container is a page, but it could also be a Form XObject
@@ -400,7 +408,7 @@ def _image_xobjects(container):
return return
xobjs = resources['/XObject'].as_dict() xobjs = resources['/XObject'].as_dict()
for xobj in xobjs: for xobj in xobjs:
candidate = xobjs[xobj] candidate: Object = xobjs[xobj]
if not '/Subtype' in candidate: if not '/Subtype' in candidate:
continue continue
if candidate['/Subtype'] == '/Image': if candidate['/Subtype'] == '/Image':
@@ -408,7 +416,9 @@ def _image_xobjects(container):
yield (pdfimage, xobj) yield (pdfimage, xobj)
def _find_regular_images(container, contentsinfo): def _find_regular_images(
container: Object, contentsinfo: ContentsInfo
) -> Iterator[ImageInfo]:
"""Find images stored in the container's /Resources /XObject """Find images stored in the container's /Resources /XObject
Usually the container is a page, but it could also be a Form XObject Usually the container is a page, but it could also be a Form XObject
@@ -432,7 +442,7 @@ def _find_regular_images(container, contentsinfo):
yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand) yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand)
def _find_form_xobject_images(pdf, container, contentsinfo): def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo):
"""Find any images that are in Form XObjects in the container """Find any images that are in Form XObjects in the container
The container may be a page, or a parent Form XObject. The container may be a page, or a parent Form XObject.
@@ -464,7 +474,9 @@ def _find_form_xobject_images(pdf, container, contentsinfo):
) )
def _process_content_streams(*, pdf, container, shorthand=None): def _process_content_streams(
*, pdf: Pdf, container: Object, shorthand=None
) -> Iterator[Union[VectorMarker, TextMarker, ImageInfo]]:
"""Find all individual instances of images drawn in the container """Find all individual instances of images drawn in the container
Usually the container is a page, but it may also be a Form XObject. Usually the container is a page, but it may also be a Form XObject.
@@ -526,7 +538,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
margin_ratio * ph, # bottom (first quadrant: bottom < top) margin_ratio * ph, # bottom (first quadrant: bottom < top)
) )
def rects_intersect(a, b): def rects_intersect(a, b) -> bool:
""" """
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3) Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
@@ -542,7 +554,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
return has_text return has_text
def simplify_textboxes(miner, textbox_getter): def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]:
"""Extract only limited content from text boxes """Extract only limited content from text boxes
We do this to save memory and ensure that our objects are pickleable. We do this to save memory and ensure that our objects are pickleable.
@@ -556,74 +568,6 @@ def simplify_textboxes(miner, textbox_getter):
yield TextboxInfo(box.bbox, visible, corrupt) yield TextboxInfo(box.bbox, visible, corrupt)
def _pdf_get_pageinfo(
pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool
):
pageinfo: Dict[str, Any] = {}
pageinfo['pageno'] = pageno
pageinfo['images'] = []
page = pdf.pages[pageno]
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
width_pt = mediabox[2] - mediabox[0]
height_pt = mediabox[3] - mediabox[1]
check_this_page = pageno in check_pages
if check_this_page and detailed_analysis:
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
miner = get_page_analysis(infile, pageno, pscript5_mode)
pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes))
bboxes = (box.bbox for box in pageinfo['textboxes'])
pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt)
else:
pageinfo['textboxes'] = []
pageinfo['has_text'] = None # i.e. "no information"
userunit = page.get('/UserUnit', Decimal(1.0))
if not isinstance(userunit, Decimal):
userunit = Decimal(userunit)
pageinfo['userunit'] = userunit
pageinfo['width_inches'] = width_pt * userunit / Decimal(72.0)
pageinfo['height_inches'] = height_pt * userunit / Decimal(72.0)
try:
pageinfo['rotate'] = int(page['/Rotate'])
except KeyError:
pageinfo['rotate'] = 0
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
if check_this_page:
pageinfo['has_vector'] = False
pageinfo['has_text'] = False
pageinfo['images'] = []
for ci in _process_content_streams(
pdf=pdf, container=page, shorthand=userunit_shorthand
):
if isinstance(ci, VectorMarker):
pageinfo['has_vector'] = True
elif isinstance(ci, TextMarker):
pageinfo['has_text'] = True
elif isinstance(ci, ImageInfo):
pageinfo['images'].append(ci)
else:
raise NotImplementedError()
else:
pageinfo['has_vector'] = None # i.e. "no information"
pageinfo['has_text'] = None
pageinfo['images'] = None
if pageinfo['images']:
dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo['images'])
pageinfo['dpi'] = dpi
pageinfo['width_pixels'] = int(round(dpi.x * float(pageinfo['width_inches'])))
pageinfo['height_pixels'] = int(round(dpi.y * float(pageinfo['height_inches'])))
return pageinfo
worker_pdf = None worker_pdf = None
@@ -693,13 +637,85 @@ def _pdf_pageinfo_concurrent(
class PageInfo: class PageInfo:
def __init__(self, pdf, pageno, infile, check_pages, detailed_analysis=False): def __init__(
self,
pdf: Pdf,
pageno: int,
infile: PathLike,
check_pages: Container[int],
detailed_analysis: bool = False,
):
self._pageno = pageno self._pageno = pageno
self._infile = infile self._infile = infile
self._detailed_analysis = detailed_analysis self._detailed_analysis = detailed_analysis
self._pageinfo = _pdf_get_pageinfo( self._gather_pageinfo(pdf, pageno, infile, check_pages, detailed_analysis)
pdf, pageno, infile, check_pages, detailed_analysis
) def _gather_pageinfo(
self,
pdf: Pdf,
pageno: int,
infile: PathLike,
check_pages: Container[int],
detailed_analysis: bool,
):
page = pdf.pages[pageno]
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
width_pt = mediabox[2] - mediabox[0]
height_pt = mediabox[3] - mediabox[1]
check_this_page = pageno in check_pages
if check_this_page and detailed_analysis:
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
miner = get_page_analysis(infile, pageno, pscript5_mode)
self._textboxes = list(simplify_textboxes(miner, get_text_boxes))
bboxes = (box.bbox for box in self._textboxes)
self._has_text = _page_has_text(bboxes, width_pt, height_pt)
else:
self._textboxes = []
self._has_text = None # i.e. "no information"
userunit = page.get('/UserUnit', Decimal(1.0))
if not isinstance(userunit, Decimal):
userunit = Decimal(userunit)
self._userunit = userunit
self._width_inches = width_pt * userunit / Decimal(72.0)
self._height_inches = height_pt * userunit / Decimal(72.0)
try:
self._rotate = int(page['/Rotate'])
except KeyError:
self._rotate = 0
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
if check_this_page:
self._has_vector = False
self._has_text = False
self._images = []
for ci in _process_content_streams(
pdf=pdf, container=page, shorthand=userunit_shorthand
):
if isinstance(ci, VectorMarker):
self._has_vector = True
elif isinstance(ci, TextMarker):
self._has_text = True
elif isinstance(ci, ImageInfo):
self._images.append(ci)
else:
raise NotImplementedError()
else:
self._has_vector = None # i.e. "no information"
self._has_text = None
self._images = None
self._dpi = None
if self._images:
dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in self._images)
self._dpi = dpi
self._width_pixels = int(round(dpi.x * float(self._width_inches)))
self._height_pixels = int(round(dpi.y * float(self._height_inches)))
@property @property
def pageno(self) -> int: def pageno(self) -> int:
@@ -707,25 +723,25 @@ class PageInfo:
@property @property
def has_text(self) -> bool: def has_text(self) -> bool:
return self._pageinfo['has_text'] return self._has_text
@property @property
def has_corrupt_text(self) -> bool: def has_corrupt_text(self) -> bool:
if not self._detailed_analysis: if not self._detailed_analysis:
raise NotImplementedError('Did not do detailed analysis') raise NotImplementedError('Did not do detailed analysis')
return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes']) return any(tbox.is_corrupt for tbox in self._textboxes)
@property @property
def has_vector(self) -> bool: def has_vector(self) -> bool:
return self._pageinfo['has_vector'] return self._has_vector
@property @property
def width_inches(self) -> Decimal: def width_inches(self) -> Decimal:
return self._pageinfo['width_inches'] return self._width_inches
@property @property
def height_inches(self) -> Decimal: def height_inches(self) -> Decimal:
return self._pageinfo['height_inches'] return self._height_inches
@property @property
def width_pixels(self) -> int: def width_pixels(self) -> int:
@@ -737,18 +753,18 @@ class PageInfo:
@property @property
def rotation(self) -> int: def rotation(self) -> int:
return self._pageinfo.get('rotate', None) return self._rotate
@rotation.setter @rotation.setter
def rotation(self, value): def rotation(self, value):
if value in (0, 90, 180, 270, 360, -90, -180, -270): if value in (0, 90, 180, 270, 360, -90, -180, -270):
self._pageinfo['rotate'] = value self._rotate = value
else: else:
raise ValueError("rotation must be a cardinal angle") raise ValueError("rotation must be a cardinal angle")
@property @property
def images(self): def images(self):
return self._pageinfo['images'] return self._images
def get_textareas( def get_textareas(
self, visible: Optional[bool] = None, corrupt: Optional[bool] = None self, visible: Optional[bool] = None, corrupt: Optional[bool] = None
@@ -763,24 +779,22 @@ class PageInfo:
result = False result = False
return result return result
if 'textboxes' not in self._pageinfo: if not self._textboxes:
if visible is not None and corrupt is not None: if visible is not None and corrupt is not None:
raise NotImplementedError('Incomplete information on textboxes') raise NotImplementedError('Incomplete information on textboxes')
return self._pageinfo['bboxes'] return self._textboxes
return ( return (obj.bbox for obj in self._textboxes if predicate(obj, visible, corrupt))
obj.bbox
for obj in self._pageinfo['textboxes']
if predicate(obj, visible, corrupt)
)
@property @property
def dpi(self) -> Resolution: def dpi(self) -> Resolution:
return self._pageinfo.get('dpi', Resolution(0.0, 0.0)) if self._dpi is None:
return Resolution(0.0, 0.0)
return self._dpi
@property @property
def userunit(self) -> Decimal: def userunit(self) -> Decimal:
return self._pageinfo.get('userunit', None) return self._userunit
@property @property
def min_version(self) -> str: def min_version(self) -> str:
+4 -3
View File
@@ -223,6 +223,7 @@ def get_page_analysis(infile, pageno, pscript5_mode):
) )
interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev) interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev)
patcher = None
if pscript5_mode: if pscript5_mode:
patcher = patch.multiple( patcher = patch.multiple(
'pdfminer.pdffont.PDFType3Font', 'pdfminer.pdffont.PDFType3Font',
@@ -237,10 +238,10 @@ def get_page_analysis(infile, pageno, pscript5_mode):
with Path(infile).open('rb') as f: with Path(infile).open('rb') as f:
page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0) page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0)
interp.process_page(next(page)) interp.process_page(next(page))
except PDFTextExtractionNotAllowed: except PDFTextExtractionNotAllowed as e:
raise EncryptedPdfError() raise EncryptedPdfError() from e
finally: finally:
if pscript5_mode: if patcher is not None:
patcher.stop() patcher.stop()
return dev.get_result() return dev.get_result()
+2 -2
View File
@@ -92,10 +92,10 @@ def test_skip_ocr(resources, outpdf):
def test_redo_ocr(resources, outpdf): def test_redo_ocr(resources, outpdf):
in_ = resources / 'graph_ocred.pdf' in_ = resources / 'graph_ocred.pdf'
before = PdfInfo(in_) before = PdfInfo(in_, detailed_analysis=True)
out = outpdf out = outpdf
out = check_ocrmypdf(in_, out, '--redo-ocr') out = check_ocrmypdf(in_, out, '--redo-ocr')
after = PdfInfo(out) after = PdfInfo(out, detailed_analysis=True)
assert before[0].has_text and after[0].has_text assert before[0].has_text and after[0].has_text
assert ( assert (
before[0].get_textareas() != after[0].get_textareas() before[0].get_textareas() != after[0].get_textareas()