Compare commits

..
13 Commits
Author SHA1 Message Date
James R. Barlow 607e2d7e81 v11.4.2 release notes 2020-12-27 03:29:35 -08:00
James R. Barlow b01d9e07e8 Deal with missing pthread_sigmask on Cygwin
Closes #701
2020-12-27 02:24:00 -08:00
James R. Barlow 91db94cf2e watcher: fix OCR_LOGLEVEL env var not processed
Closes #702
2020-12-27 02:02:44 -08:00
James R. Barlow 416df803d4 pdfinfo: stricter typing 2020-12-24 22:39:00 -08:00
James R. Barlow 037b96ca16 pdfinfo: refactor to eliminate RawPageInfo 2020-12-24 02:57:44 -08:00
James R. Barlow bb258fc99c pdfinfo: Refactor pageinfo dictionary into a class 2020-12-24 01:47:53 -08:00
James R. Barlow 4b8ccbe8cb v11.4.1 release notes 2020-12-22 01:41:15 -08:00
James R. Barlow ab1ff3331b misc: synology fix
Accept user-contributed fix. Not testable.

Close #690.
2020-12-22 01:38:41 -08:00
James R. Barlow 3675ae918c Fix certain invalid page ranges causing exception
Closes #686
2020-12-22 01:22:14 -08:00
James R. Barlow 0ba32b96b7 Revert "v11.4.0 release notes - remove change not actually implemented"
This reverts commit ad202693b3.
Temporary folder prefix was actually changed in commit f11bb53e.
2020-12-22 00:47:25 -08:00
James R. Barlow add64e4fa2 docs: com.github.ocrmypdf -> ocrmypdf.io 2020-12-22 00:46:42 -08:00
James R. Barlow 7fe2954ede Change wheel tag to py36, update package_data to include py.typed 2020-12-12 16:49:04 -08:00
James R. Barlow ad202693b3 v11.4.0 release notes - remove change not actually implemented
Remove a change that was pushed back to a future release.
2020-12-12 16:27:38 -08:00
13 changed files with 178 additions and 123 deletions
+1 -1
View File
@@ -314,7 +314,7 @@ message is:
.. code-block:: none .. code-block:: none
Temporary working files retained at: Temporary working files retained at:
/tmp/com.github.ocrmypdf.u20wpz07 /tmp/ocrmypdf.io.u20wpz07
The organization of this folder is an implementation detail and subject The organization of this folder is an implementation detail and subject
to change between releases. However the general organization is that to change between releases. However the general organization is that
+16
View File
@@ -12,6 +12,22 @@ may be unreliable. Use the API to depend on precise behavior.
The public API may be useful in scripts that launch OCRmyPDF processes or that The public API may be useful in scripts that launch OCRmyPDF processes or that
wish to use some of its features for working with PDFs. wish to use some of its features for working with PDFs.
v11.4.2
=======
- Fixed support for Cygwin, hopefully.
- watcher.py: Fixed an issue with the OCR_LOGLEVEL not being interpreted.
v11.4.1
=======
- Fixed an issue where invalid pages ranges passed using the ``pages`` argument,
such as "1-0" would cause unhandled exceptions.
- Accepted a user-contributed to the Synology demo script in misc/synology.py.
- Clarified documentation about change of temporary file location ``ocrmypdf.io``.
- Fixed Python wheel tag which was incorrectly set to py35 even though we long
since dropped support for Python 3.5.
v11.4.0 v11.4.0
======= =======
+3 -1
View File
@@ -79,8 +79,10 @@ for dir_name, subdirs, file_list in os.walk(start_dir):
stdout=output_file, stdout=output_file,
stderr=subprocess.PIPE, stderr=subprocess.PIPE,
check=False, check=False,
text=True,
errors='ignore',
) )
logging.info(proc.stderr.read()) logging.info(proc.stderr)
os.chmod(full_path_ocr, 0o664) os.chmod(full_path_ocr, 0o664)
os.chmod(full_path, 0o664) os.chmod(full_path, 0o664)
full_path_ocr_archive = sys.argv[2] full_path_ocr_archive = sys.argv[2]
+8 -3
View File
@@ -44,7 +44,7 @@ DESKEW = bool(os.getenv('OCR_DESKEW', ''))
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1')) POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
USE_POLLING = bool(os.getenv('OCR_USE_POLLING', '')) USE_POLLING = bool(os.getenv('OCR_USE_POLLING', ''))
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper() LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
PATTERNS = ['*.pdf', '*.PDF'] PATTERNS = ['*.pdf', '*.PDF']
log = logging.getLogger('ocrmypdf-watcher') log = logging.getLogger('ocrmypdf-watcher')
@@ -117,7 +117,12 @@ class HandleObserverEvent(PatternMatchingEventHandler):
def main(): def main():
ocrmypdf.configure_logging( ocrmypdf.configure_logging(
verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True verbosity=(
ocrmypdf.Verbosity.default
if LOGLEVEL != 'DEBUG'
else ocrmypdf.Verbosity.debug
),
manage_root_logger=True,
) )
log.setLevel(LOGLEVEL) log.setLevel(LOGLEVEL)
log.info( log.info(
@@ -135,7 +140,7 @@ def main():
f"ARGS: {OCR_JSON_SETTINGS}\n" f"ARGS: {OCR_JSON_SETTINGS}\n"
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
f"USE_POLLING: {USE_POLLING}\n" f"USE_POLLING: {USE_POLLING}\n"
f"LOGLEVEL: {LOGLEVEL}\n" f"LOGLEVEL: {LOGLEVEL}"
) )
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS: if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS:
+1 -1
View File
@@ -1,5 +1,5 @@
[bdist_wheel] [bdist_wheel]
python-tag = py35 python-tag = py36
[aliases] [aliases]
test=pytest test=pytest
+1 -1
View File
@@ -82,7 +82,7 @@ setup(
], ],
tests_require=tests_require, tests_require=tests_require,
entry_points={'console_scripts': ['ocrmypdf = ocrmypdf.__main__:run']}, entry_points={'console_scripts': ['ocrmypdf = ocrmypdf.__main__:run']},
package_data={'ocrmypdf': ['data/sRGB.icc']}, package_data={'ocrmypdf': ['data/sRGB.icc', 'py.typed']},
include_package_data=True, include_package_data=True,
zip_safe=False, zip_safe=False,
project_urls={ project_urls={
+4 -2
View File
@@ -12,6 +12,7 @@ import os
import signal import signal
import sys import sys
import threading import threading
from contextlib import suppress
from multiprocessing import Pool as ProcessPool from multiprocessing import Pool as ProcessPool
from multiprocessing.dummy import Pool as ThreadPool from multiprocessing.dummy import Pool as ThreadPool
from typing import Callable, Iterable, Optional from typing import Callable, Iterable, Optional
@@ -56,7 +57,7 @@ def process_init(queue, user_init, loglevel):
signal.signal(signal.SIGINT, signal.SIG_IGN) signal.signal(signal.SIGINT, signal.SIG_IGN)
# Install SIGBUS handler (so our parent process can abort somewhat gracefully) # Install SIGBUS handler (so our parent process can abort somewhat gracefully)
if hasattr(signal, 'SIGBUS'): with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS
signal.signal(signal.SIGBUS, process_sigbus) signal.signal(signal.SIGBUS, process_sigbus)
# Reconfigure the root logger for this process to send all messages to a queue # Reconfigure the root logger for this process to send all messages to a queue
@@ -72,7 +73,8 @@ def process_init(queue, user_init, loglevel):
def thread_init(_queue, user_init, _loglevel): def thread_init(_queue, user_init, _loglevel):
# As a thread, block SIGBUS so the main thread deals with it... # As a thread, block SIGBUS so the main thread deals with it...
if hasattr(signal, 'SIGBUS'): with suppress(AttributeError):
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS}) signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS})
if user_init: if user_init:
user_init() user_init()
+1 -1
View File
@@ -150,7 +150,7 @@ def get_pdfinfo(
progbar=False, progbar=False,
max_workers=None, max_workers=None,
check_pages=None, check_pages=None,
): ) -> PdfInfo:
try: try:
return PdfInfo( return PdfInfo(
input_file, input_file,
+14 -5
View File
@@ -13,7 +13,7 @@ import sys
import unicodedata import unicodedata
from pathlib import Path from pathlib import Path
from shutil import copyfileobj from shutil import copyfileobj
from typing import Tuple from typing import List, Set, Tuple, Union
import pikepdf import pikepdf
import PIL import PIL
@@ -136,10 +136,10 @@ def check_options_preprocessing(options):
raise BadArgsError(str(e)) raise BadArgsError(str(e))
def _pages_from_ranges(ranges): def _pages_from_ranges(ranges: str) -> Set[int]:
if is_iterable_notstr(ranges): if is_iterable_notstr(ranges):
return set(ranges) return set(ranges)
pages = [] pages: List[int] = []
page_groups = ranges.replace(' ', '').split(',') page_groups = ranges.replace(' ', '').split(',')
for g in page_groups: for g in page_groups:
if not g: if not g:
@@ -150,9 +150,18 @@ def _pages_from_ranges(ranges):
pages.append(int(g) - 1) pages.append(int(g) - 1)
else: else:
try: try:
pages.extend(range(int(start) - 1, int(end))) new_pages = list(range(int(start) - 1, int(end)))
if not new_pages:
raise BadArgsError(f"invalid page subrange '{start}-{end}'")
pages.extend(new_pages)
except ValueError: except ValueError:
raise BadArgsError("invalid page range") raise BadArgsError("invalid page range") from None
if not pages:
raise BadArgsError(
f"The string of page ranges '{ranges}' did not contain any recognizable "
f"page ranges."
)
if not monotonic(pages): if not monotonic(pages):
log.warning( log.warning(
+117 -103
View File
@@ -15,11 +15,11 @@ from functools import partial
from math import hypot, isclose from math import hypot, isclose
from os import PathLike from os import PathLike
from pathlib import Path from pathlib import Path
from typing import Any, Dict, List, Optional, Union from typing import Any, Container, Dict, Iterator, List, Optional, Tuple, Union
from warnings import warn from warnings import warn
import pikepdf import pikepdf
from pikepdf import PdfMatrix from pikepdf import Object, Pdf, PdfMatrix
from ocrmypdf._concurrent import exec_progress_pool from ocrmypdf._concurrent import exec_progress_pool
from ocrmypdf.exceptions import EncryptedPdfError from ocrmypdf.exceptions import EncryptedPdfError
@@ -115,7 +115,7 @@ def _normalize_stack(graphobjs):
yield (operands, operator) yield (operands, operator)
def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE): def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
"""Interpret the PDF content stream. """Interpret the PDF content stream.
The stack represents the state of the PDF graphics stack. We are only The stack represents the state of the PDF graphics stack. We are only
@@ -204,7 +204,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
) )
def _get_dpi(ctm_shorthand, image_size): def _get_dpi(ctm_shorthand, image_size) -> Resolution:
"""Given the transformation matrix and image size, find the image DPI. """Given the transformation matrix and image size, find the image DPI.
PDFs do not include image resolution information within image data. PDFs do not include image resolution information within image data.
@@ -271,8 +271,14 @@ def _get_dpi(ctm_shorthand, image_size):
class ImageInfo: class ImageInfo:
DPI_PREC = Decimal('1.000') DPI_PREC = Decimal('1.000')
def __init__(self, *, name='', pdfimage=None, inline=None, shorthand=None): def __init__(
self,
*,
name='',
pdfimage: Optional[Object] = None,
inline: Optional[Object] = None,
shorthand=None,
):
self._name = str(name) self._name = str(name)
self._shorthand = shorthand self._shorthand = shorthand
@@ -282,6 +288,8 @@ class ImageInfo:
elif pdfimage is not None: elif pdfimage is not None:
self._origin = 'xobject' self._origin = 'xobject'
pim = pikepdf.PdfImage(pdfimage) pim = pikepdf.PdfImage(pdfimage)
else:
raise ValueError("Either pdfimage or inline must be set")
self._width = pim.width self._width = pim.width
self._height = pim.height self._height = pim.height
@@ -371,7 +379,7 @@ class ImageInfo:
).format(**class_locals) ).format(**class_locals)
def _find_inline_images(contentsinfo): def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
"Find inline images in the contentstream" "Find inline images in the contentstream"
for n, inline in enumerate(contentsinfo.inline_images): for n, inline in enumerate(contentsinfo.inline_images):
@@ -380,7 +388,7 @@ def _find_inline_images(contentsinfo):
) )
def _image_xobjects(container): def _image_xobjects(container) -> Iterator[Tuple[Object, str]]:
"""Search for all XObject-based images in the container """Search for all XObject-based images in the container
Usually the container is a page, but it could also be a Form XObject Usually the container is a page, but it could also be a Form XObject
@@ -400,7 +408,7 @@ def _image_xobjects(container):
return return
xobjs = resources['/XObject'].as_dict() xobjs = resources['/XObject'].as_dict()
for xobj in xobjs: for xobj in xobjs:
candidate = xobjs[xobj] candidate: Object = xobjs[xobj]
if not '/Subtype' in candidate: if not '/Subtype' in candidate:
continue continue
if candidate['/Subtype'] == '/Image': if candidate['/Subtype'] == '/Image':
@@ -408,7 +416,9 @@ def _image_xobjects(container):
yield (pdfimage, xobj) yield (pdfimage, xobj)
def _find_regular_images(container, contentsinfo): def _find_regular_images(
container: Object, contentsinfo: ContentsInfo
) -> Iterator[ImageInfo]:
"""Find images stored in the container's /Resources /XObject """Find images stored in the container's /Resources /XObject
Usually the container is a page, but it could also be a Form XObject Usually the container is a page, but it could also be a Form XObject
@@ -432,7 +442,7 @@ def _find_regular_images(container, contentsinfo):
yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand) yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand)
def _find_form_xobject_images(pdf, container, contentsinfo): def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo):
"""Find any images that are in Form XObjects in the container """Find any images that are in Form XObjects in the container
The container may be a page, or a parent Form XObject. The container may be a page, or a parent Form XObject.
@@ -464,7 +474,9 @@ def _find_form_xobject_images(pdf, container, contentsinfo):
) )
def _process_content_streams(*, pdf, container, shorthand=None): def _process_content_streams(
*, pdf: Pdf, container: Object, shorthand=None
) -> Iterator[Union[VectorMarker, TextMarker, ImageInfo]]:
"""Find all individual instances of images drawn in the container """Find all individual instances of images drawn in the container
Usually the container is a page, but it may also be a Form XObject. Usually the container is a page, but it may also be a Form XObject.
@@ -526,7 +538,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
margin_ratio * ph, # bottom (first quadrant: bottom < top) margin_ratio * ph, # bottom (first quadrant: bottom < top)
) )
def rects_intersect(a, b): def rects_intersect(a, b) -> bool:
""" """
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3) Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
@@ -542,7 +554,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
return has_text return has_text
def simplify_textboxes(miner, textbox_getter): def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]:
"""Extract only limited content from text boxes """Extract only limited content from text boxes
We do this to save memory and ensure that our objects are pickleable. We do this to save memory and ensure that our objects are pickleable.
@@ -556,74 +568,6 @@ def simplify_textboxes(miner, textbox_getter):
yield TextboxInfo(box.bbox, visible, corrupt) yield TextboxInfo(box.bbox, visible, corrupt)
def _pdf_get_pageinfo(
pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool
):
pageinfo: Dict[str, Any] = {}
pageinfo['pageno'] = pageno
pageinfo['images'] = []
page = pdf.pages[pageno]
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
width_pt = mediabox[2] - mediabox[0]
height_pt = mediabox[3] - mediabox[1]
check_this_page = pageno in check_pages
if check_this_page and detailed_analysis:
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
miner = get_page_analysis(infile, pageno, pscript5_mode)
pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes))
bboxes = (box.bbox for box in pageinfo['textboxes'])
pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt)
else:
pageinfo['textboxes'] = []
pageinfo['has_text'] = None # i.e. "no information"
userunit = page.get('/UserUnit', Decimal(1.0))
if not isinstance(userunit, Decimal):
userunit = Decimal(userunit)
pageinfo['userunit'] = userunit
pageinfo['width_inches'] = width_pt * userunit / Decimal(72.0)
pageinfo['height_inches'] = height_pt * userunit / Decimal(72.0)
try:
pageinfo['rotate'] = int(page['/Rotate'])
except KeyError:
pageinfo['rotate'] = 0
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
if check_this_page:
pageinfo['has_vector'] = False
pageinfo['has_text'] = False
pageinfo['images'] = []
for ci in _process_content_streams(
pdf=pdf, container=page, shorthand=userunit_shorthand
):
if isinstance(ci, VectorMarker):
pageinfo['has_vector'] = True
elif isinstance(ci, TextMarker):
pageinfo['has_text'] = True
elif isinstance(ci, ImageInfo):
pageinfo['images'].append(ci)
else:
raise NotImplementedError()
else:
pageinfo['has_vector'] = None # i.e. "no information"
pageinfo['has_text'] = None
pageinfo['images'] = None
if pageinfo['images']:
dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo['images'])
pageinfo['dpi'] = dpi
pageinfo['width_pixels'] = int(round(dpi.x * float(pageinfo['width_inches'])))
pageinfo['height_pixels'] = int(round(dpi.y * float(pageinfo['height_inches'])))
return pageinfo
worker_pdf = None worker_pdf = None
@@ -693,13 +637,85 @@ def _pdf_pageinfo_concurrent(
class PageInfo: class PageInfo:
def __init__(self, pdf, pageno, infile, check_pages, detailed_analysis=False): def __init__(
self,
pdf: Pdf,
pageno: int,
infile: PathLike,
check_pages: Container[int],
detailed_analysis: bool = False,
):
self._pageno = pageno self._pageno = pageno
self._infile = infile self._infile = infile
self._detailed_analysis = detailed_analysis self._detailed_analysis = detailed_analysis
self._pageinfo = _pdf_get_pageinfo( self._gather_pageinfo(pdf, pageno, infile, check_pages, detailed_analysis)
pdf, pageno, infile, check_pages, detailed_analysis
) def _gather_pageinfo(
self,
pdf: Pdf,
pageno: int,
infile: PathLike,
check_pages: Container[int],
detailed_analysis: bool,
):
page = pdf.pages[pageno]
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
width_pt = mediabox[2] - mediabox[0]
height_pt = mediabox[3] - mediabox[1]
check_this_page = pageno in check_pages
if check_this_page and detailed_analysis:
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
miner = get_page_analysis(infile, pageno, pscript5_mode)
self._textboxes = list(simplify_textboxes(miner, get_text_boxes))
bboxes = (box.bbox for box in self._textboxes)
self._has_text = _page_has_text(bboxes, width_pt, height_pt)
else:
self._textboxes = []
self._has_text = None # i.e. "no information"
userunit = page.get('/UserUnit', Decimal(1.0))
if not isinstance(userunit, Decimal):
userunit = Decimal(userunit)
self._userunit = userunit
self._width_inches = width_pt * userunit / Decimal(72.0)
self._height_inches = height_pt * userunit / Decimal(72.0)
try:
self._rotate = int(page['/Rotate'])
except KeyError:
self._rotate = 0
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
if check_this_page:
self._has_vector = False
self._has_text = False
self._images = []
for ci in _process_content_streams(
pdf=pdf, container=page, shorthand=userunit_shorthand
):
if isinstance(ci, VectorMarker):
self._has_vector = True
elif isinstance(ci, TextMarker):
self._has_text = True
elif isinstance(ci, ImageInfo):
self._images.append(ci)
else:
raise NotImplementedError()
else:
self._has_vector = None # i.e. "no information"
self._has_text = None
self._images = None
self._dpi = None
if self._images:
dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in self._images)
self._dpi = dpi
self._width_pixels = int(round(dpi.x * float(self._width_inches)))
self._height_pixels = int(round(dpi.y * float(self._height_inches)))
@property @property
def pageno(self) -> int: def pageno(self) -> int:
@@ -707,25 +723,25 @@ class PageInfo:
@property @property
def has_text(self) -> bool: def has_text(self) -> bool:
return self._pageinfo['has_text'] return self._has_text
@property @property
def has_corrupt_text(self) -> bool: def has_corrupt_text(self) -> bool:
if not self._detailed_analysis: if not self._detailed_analysis:
raise NotImplementedError('Did not do detailed analysis') raise NotImplementedError('Did not do detailed analysis')
return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes']) return any(tbox.is_corrupt for tbox in self._textboxes)
@property @property
def has_vector(self) -> bool: def has_vector(self) -> bool:
return self._pageinfo['has_vector'] return self._has_vector
@property @property
def width_inches(self) -> Decimal: def width_inches(self) -> Decimal:
return self._pageinfo['width_inches'] return self._width_inches
@property @property
def height_inches(self) -> Decimal: def height_inches(self) -> Decimal:
return self._pageinfo['height_inches'] return self._height_inches
@property @property
def width_pixels(self) -> int: def width_pixels(self) -> int:
@@ -737,18 +753,18 @@ class PageInfo:
@property @property
def rotation(self) -> int: def rotation(self) -> int:
return self._pageinfo.get('rotate', None) return self._rotate
@rotation.setter @rotation.setter
def rotation(self, value): def rotation(self, value):
if value in (0, 90, 180, 270, 360, -90, -180, -270): if value in (0, 90, 180, 270, 360, -90, -180, -270):
self._pageinfo['rotate'] = value self._rotate = value
else: else:
raise ValueError("rotation must be a cardinal angle") raise ValueError("rotation must be a cardinal angle")
@property @property
def images(self): def images(self):
return self._pageinfo['images'] return self._images
def get_textareas( def get_textareas(
self, visible: Optional[bool] = None, corrupt: Optional[bool] = None self, visible: Optional[bool] = None, corrupt: Optional[bool] = None
@@ -763,24 +779,22 @@ class PageInfo:
result = False result = False
return result return result
if 'textboxes' not in self._pageinfo: if not self._textboxes:
if visible is not None and corrupt is not None: if visible is not None and corrupt is not None:
raise NotImplementedError('Incomplete information on textboxes') raise NotImplementedError('Incomplete information on textboxes')
return self._pageinfo['bboxes'] return self._textboxes
return ( return (obj.bbox for obj in self._textboxes if predicate(obj, visible, corrupt))
obj.bbox
for obj in self._pageinfo['textboxes']
if predicate(obj, visible, corrupt)
)
@property @property
def dpi(self) -> Resolution: def dpi(self) -> Resolution:
return self._pageinfo.get('dpi', Resolution(0.0, 0.0)) if self._dpi is None:
return Resolution(0.0, 0.0)
return self._dpi
@property @property
def userunit(self) -> Decimal: def userunit(self) -> Decimal:
return self._pageinfo.get('userunit', None) return self._userunit
@property @property
def min_version(self) -> str: def min_version(self) -> str:
+4 -3
View File
@@ -223,6 +223,7 @@ def get_page_analysis(infile, pageno, pscript5_mode):
) )
interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev) interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev)
patcher = None
if pscript5_mode: if pscript5_mode:
patcher = patch.multiple( patcher = patch.multiple(
'pdfminer.pdffont.PDFType3Font', 'pdfminer.pdffont.PDFType3Font',
@@ -237,10 +238,10 @@ def get_page_analysis(infile, pageno, pscript5_mode):
with Path(infile).open('rb') as f: with Path(infile).open('rb') as f:
page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0) page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0)
interp.process_page(next(page)) interp.process_page(next(page))
except PDFTextExtractionNotAllowed: except PDFTextExtractionNotAllowed as e:
raise EncryptedPdfError() raise EncryptedPdfError() from e
finally: finally:
if pscript5_mode: if patcher is not None:
patcher.stop() patcher.stop()
return dev.get_result() return dev.get_result()
+2 -2
View File
@@ -92,10 +92,10 @@ def test_skip_ocr(resources, outpdf):
def test_redo_ocr(resources, outpdf): def test_redo_ocr(resources, outpdf):
in_ = resources / 'graph_ocred.pdf' in_ = resources / 'graph_ocred.pdf'
before = PdfInfo(in_) before = PdfInfo(in_, detailed_analysis=True)
out = outpdf out = outpdf
out = check_ocrmypdf(in_, out, '--redo-ocr') out = check_ocrmypdf(in_, out, '--redo-ocr')
after = PdfInfo(out) after = PdfInfo(out, detailed_analysis=True)
assert before[0].has_text and after[0].has_text assert before[0].has_text and after[0].has_text
assert ( assert (
before[0].get_textareas() != after[0].get_textareas() before[0].get_textareas() != after[0].get_textareas()
+6
View File
@@ -28,6 +28,12 @@ from ocrmypdf.pdfinfo import PdfInfo
['1,3,-11', BadArgsError], ['1,3,-11', BadArgsError],
['1-,', BadArgsError], ['1-,', BadArgsError],
['start-end', BadArgsError], ['start-end', BadArgsError],
['1-0', BadArgsError],
['99-98', BadArgsError],
['0-0', BadArgsError],
['1-0,3-4', BadArgsError],
[',', BadArgsError],
['', BadArgsError],
], ],
) )
def test_pages(pages, result): def test_pages(pages, result):