Compare commits

...
9 Commits
13 changed files with 342 additions and 93 deletions
+3 -3
View File
@@ -32,7 +32,7 @@ jobs:
python: "3.9" python: "3.9"
tesseract5: true tesseract5: true
- os: ubuntu-latest - os: ubuntu-latest
python: "3.12-dev" python: "3.12"
tesseract5: true tesseract5: true
#- os: ubuntu-latest #- os: ubuntu-latest
# python: "pypy3.9" # python: "pypy3.9"
@@ -113,7 +113,7 @@ jobs:
strategy: strategy:
matrix: matrix:
os: [macos-latest] os: [macos-latest]
python: ["3.10", "3.11", "3.12-dev"] python: ["3.10", "3.11", "3.12"]
env: env:
OS: ${{ matrix.os }} OS: ${{ matrix.os }}
@@ -170,7 +170,7 @@ jobs:
strategy: strategy:
matrix: matrix:
os: [windows-latest] os: [windows-latest]
python: ["3.10", "3.11", "3.12-dev"] python: ["3.10", "3.11", "3.12"]
env: env:
OS: ${{ matrix.os }} OS: ${{ matrix.os }}
+3 -3
View File
@@ -152,7 +152,7 @@ The watcher service is included in the OCRmyPDF Docker image. To run it:
docker run \ docker run \
--volume <path to files to convert>:/input \ --volume <path to files to convert>:/input \
--volume <path to store results>:/output \ --volume <path to store results>:/output \
--volume <path to store processed originals>:/archive \ --volume <path to store processed originals>:/processed \
--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \ --env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
--env OCR_ON_SUCCESS_ARCHIVE=1 \ --env OCR_ON_SUCCESS_ARCHIVE=1 \
--env OCR_DESKEW=1 \ --env OCR_DESKEW=1 \
@@ -163,7 +163,7 @@ The watcher service is included in the OCRmyPDF Docker image. To run it:
This service will watch for a file that matches ``/input/\*.pdf``, This service will watch for a file that matches ``/input/\*.pdf``,
convert it to a OCRed PDF in ``/output/``, and move the processed convert it to a OCRed PDF in ``/output/``, and move the processed
original to ``/archive``. The parameters to this image are: original to ``/processed``. The parameters to this image are:
.. csv-table:: watcher.py parameters for Docker .. csv-table:: watcher.py parameters for Docker
:header: "Parameter", "Description" :header: "Parameter", "Description"
@@ -171,7 +171,7 @@ original to ``/archive``. The parameters to this image are:
"``--volume <path to files to convert>:/input``", "Files placed in this location will be OCRed" "``--volume <path to files to convert>:/input``", "Files placed in this location will be OCRed"
"``--volume <path to store results>:/output``", "This is where OCRed files will be stored" "``--volume <path to store results>:/output``", "This is where OCRed files will be stored"
"``--volume <path to store processed originals>:/archive``", "Archive processed originals here" "``--volume <path to store processed originals>:/processed``", "Archive processed originals here"
"``--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``" "``--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``"
"``--env OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals" "``--env OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals"
"``--env OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs" "``--env OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs"
+9
View File
@@ -28,6 +28,15 @@ tagged yet.
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg .. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
v15.3.0
=======
- Update misc/watcher.py to improve command line interface using Typer, and
support ``.env`` specification of environment variables. Improved error
messages. Thanks to @mflagg2814 for the PR that prompted this improvement.
- Improved error message when a file cannot be read because we are running in
a snap container.
v15.2.0 v15.2.0
======= =======
+231 -75
View File
@@ -5,107 +5,122 @@
"""Watch a directory for new PDFs and OCR them.""" """Watch a directory for new PDFs and OCR them."""
from __future__ import annotations # Do not enable annotations!
# https://github.com/tiangolo/typer/discussions/598
# from __future__ import annotations
import json import json
import logging import logging
import os
import shutil import shutil
import sys import sys
import time import time
from datetime import datetime from datetime import datetime
from enum import Enum
from pathlib import Path from pathlib import Path
from typing import Annotated, Any
import pikepdf import pikepdf
import typer
from dotenv import load_dotenv
from watchdog.events import PatternMatchingEventHandler from watchdog.events import PatternMatchingEventHandler
from watchdog.observers import Observer from watchdog.observers import Observer
from watchdog.observers.polling import PollingObserver from watchdog.observers.polling import PollingObserver
import ocrmypdf import ocrmypdf
load_dotenv()
# pylint: disable=logging-format-interpolation # pylint: disable=logging-format-interpolation
app = typer.Typer(name="ocrmypdf-watcher")
def getenv_bool(name: str, default: str = 'False'):
return os.getenv(name, default).lower() in ('true', 'yes', 'y', '1')
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
ARCHIVE_DIRECTORY = os.getenv('OCR_ARCHIVE_DIRECTORY', '/processed')
OUTPUT_DIRECTORY_YEAR_MONTH = getenv_bool('OCR_OUTPUT_DIRECTORY_YEAR_MONTH')
ON_SUCCESS_DELETE = getenv_bool('OCR_ON_SUCCESS_DELETE')
ON_SUCCESS_ARCHIVE = getenv_bool('OCR_ON_SUCCESS_ARCHIVE')
DESKEW = getenv_bool('OCR_DESKEW')
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
USE_POLLING = getenv_bool('OCR_USE_POLLING')
RETRIES_LOADING_FILE = int(os.getenv('OCR_RETRIES_LOADING_FILE', '5'))
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
PATTERNS = ['*.pdf', '*.PDF']
log = logging.getLogger('ocrmypdf-watcher') log = logging.getLogger('ocrmypdf-watcher')
def get_output_dir(root, basename): class LoggingLevelEnum(str, Enum):
if OUTPUT_DIRECTORY_YEAR_MONTH: """Enum for logging levels."""
DEBUG = "DEBUG"
INFO = "INFO"
WARNING = "WARNING"
ERROR = "ERROR"
CRITICAL = "CRITICAL"
def get_output_dir(root: Path, basename: str, output_dir_year_month: bool) -> Path:
if output_dir_year_month:
today = datetime.today() today = datetime.today()
output_directory_year_month = ( output_directory_year_month = root / str(today.year) / f'{today.month:02d}'
Path(root) / str(today.year) / f'{today.month:02d}'
)
if not output_directory_year_month.exists(): if not output_directory_year_month.exists():
output_directory_year_month.mkdir(parents=True, exist_ok=True) output_directory_year_month.mkdir(parents=True, exist_ok=True)
output_path = Path(output_directory_year_month) / basename output_path = Path(output_directory_year_month) / basename
else: else:
output_path = Path(OUTPUT_DIRECTORY) / basename output_path = root / basename
return output_path return output_path
def wait_for_file_ready(file_path): def wait_for_file_ready(
file_path: Path, poll_new_file_seconds: int, retries_loading_file: int
):
# This loop waits to make sure that the file is completely loaded on # This loop waits to make sure that the file is completely loaded on
# disk before attempting to read. Docker sometimes will publish the # disk before attempting to read. Docker sometimes will publish the
# watchdog event before the file is actually fully on disk, causing # watchdog event before the file is actually fully on disk, causing
# pikepdf to fail. # pikepdf to fail.
retries = RETRIES_LOADING_FILE tries = retries_loading_file + 1
while retries: while tries:
try: try:
pdf = pikepdf.open(file_path) with pikepdf.Pdf.open(file_path) as pdf:
except (FileNotFoundError, pikepdf.PdfError) as e: log.debug(f"{file_path} ready with {pdf.pages} pages")
return True
except (FileNotFoundError, OSError) as e:
log.info(f"File {file_path} is not ready yet") log.info(f"File {file_path} is not ready yet")
log.debug("Exception was", exc_info=e) log.debug("Exception was", exc_info=e)
time.sleep(POLL_NEW_FILE_SECONDS) time.sleep(poll_new_file_seconds)
retries -= 1 tries -= 1
else: except pikepdf.PdfError as e:
pdf.close() log.info(f"File {file_path} is not full written yet")
return True log.debug("Exception was", exc_info=e)
time.sleep(poll_new_file_seconds)
tries -= 1
return False return False
def execute_ocrmypdf(file_path): def execute_ocrmypdf(
file_path = Path(file_path) *,
output_path = get_output_dir(OUTPUT_DIRECTORY, file_path.name) file_path: Path,
archive_dir: Path,
output_dir: Path,
deskew: bool,
ocrmypdf_kwargs: dict[str, Any],
on_success_delete: bool,
on_success_archive: bool,
poll_new_file_seconds: int,
retries_loading_file: int,
output_dir_year_month: bool,
):
output_path = get_output_dir(output_dir, file_path.name, output_dir_year_month)
log.info("-" * 20) log.info("-" * 20)
log.info(f'New file: {file_path}. Waiting until fully loaded...') log.info(f'New file: {file_path}. Waiting until fully written...')
if not wait_for_file_ready(file_path): if not wait_for_file_ready(file_path, poll_new_file_seconds, retries_loading_file):
log.info(f"Gave up waiting for {file_path} to become ready") log.info(f"Gave up waiting for {file_path} to become ready")
return return
log.info(f'Attempting to OCRmyPDF to: {output_path}') log.info(f'Attempting to OCRmyPDF to: {output_path}')
exit_code = ocrmypdf.ocr( exit_code = ocrmypdf.ocr(
input_file=file_path, input_file=file_path,
output_file=output_path, output_file=output_path,
deskew=DESKEW, deskew=deskew,
**OCR_JSON_SETTINGS, **ocrmypdf_kwargs,
) )
if exit_code == 0: if exit_code == 0:
if ON_SUCCESS_DELETE: if on_success_delete:
log.info(f'OCR is done. Deleting: {file_path}') log.info(f'OCR is done. Deleting: {file_path}')
file_path.unlink() file_path.unlink()
elif ON_SUCCESS_ARCHIVE: elif on_success_archive:
log.info(f'OCR is done. Archiving {file_path.name} to {ARCHIVE_DIRECTORY}') log.info(f'OCR is done. Archiving {file_path.name} to {archive_dir}')
shutil.move(file_path, f'{ARCHIVE_DIRECTORY}/{file_path.name}') shutil.move(file_path, f'{archive_dir}/{file_path.name}')
else: else:
log.info('OCR is done') log.info('OCR is done')
else: else:
@@ -113,61 +128,202 @@ def execute_ocrmypdf(file_path):
class HandleObserverEvent(PatternMatchingEventHandler): class HandleObserverEvent(PatternMatchingEventHandler):
def __init__(
self,
patterns=None,
ignore_patterns=None,
ignore_directories=False,
case_sensitive=False,
settings={},
):
super().__init__(
patterns=patterns,
ignore_patterns=ignore_patterns,
ignore_directories=ignore_directories,
case_sensitive=case_sensitive,
)
self._settings = settings
def on_any_event(self, event): def on_any_event(self, event):
if event.event_type in ['created']: if event.event_type in ['created']:
execute_ocrmypdf(event.src_path) execute_ocrmypdf(event.src_path, **self._settings)
def main(): @app.command()
def main(
input_dir: Annotated[
Path,
typer.Argument(
envvar='OCR_INPUT_DIRECTORY',
exists=True,
file_okay=False,
dir_okay=True,
readable=True,
resolve_path=True,
),
] = '/input',
output_dir: Annotated[
Path,
typer.Argument(
envvar='OCR_OUTPUT_DIRECTORY',
exists=True,
file_okay=False,
dir_okay=True,
writable=True,
resolve_path=True,
),
] = '/output',
archive_dir: Annotated[
Path,
typer.Argument(
envvar='OCR_ARCHIVE_DIRECTORY',
exists=True,
file_okay=False,
dir_okay=True,
writable=True,
resolve_path=True,
),
] = '/processed',
output_dir_year_month: Annotated[
bool,
typer.Option(
envvar='OCR_OUTPUT_DIRECTORY_YEAR_MONTH',
help='Create a subdirectory in the output directory for each year and month',
),
] = False,
on_success_delete: Annotated[
bool,
typer.Option(
envvar='OCR_ON_SUCCESS_DELETE',
help='Delete the input file after successful OCR',
),
] = False,
on_success_archive: Annotated[
bool,
typer.Option(
envvar='OCR_ON_SUCCESS_ARCHIVE',
help='Archive the input file after successful OCR',
),
] = False,
deskew: Annotated[
bool,
typer.Option(
envvar='OCR_DESKEW',
help='Deskew the input file before OCR',
),
] = False,
ocr_json_settings: Annotated[
typer.FileText,
typer.Option(
envvar='OCR_JSON_SETTINGS',
help='JSON settings to pass to OCRmyPDF',
),
] = None,
poll_new_file_seconds: Annotated[
int,
typer.Option(
envvar='OCR_POLL_NEW_FILE_SECONDS',
help='Seconds to wait before polling a new file',
min=0,
),
] = 1,
use_polling: Annotated[
bool,
typer.Option(
envvar='OCR_USE_POLLING',
help='Use polling instead of filesystem events',
),
] = False,
retries_loading_file: Annotated[
int,
typer.Option(
envvar='OCR_RETRIES_LOADING_FILE',
help='Number of times to retry loading a file before giving up',
min=0,
),
] = 5,
loglevel: Annotated[
LoggingLevelEnum,
typer.Option(
envvar='OCR_LOGLEVEL',
help='Logging level',
),
] = LoggingLevelEnum.INFO,
patterns: Annotated[
str,
typer.Option(
envvar='OCR_PATTERNS',
help='File patterns to watch',
),
] = '*.pdf,*.PDF',
):
ocrmypdf.configure_logging( ocrmypdf.configure_logging(
verbosity=( verbosity=(
ocrmypdf.Verbosity.default ocrmypdf.Verbosity.default
if LOGLEVEL != 'DEBUG' if loglevel != 'DEBUG'
else ocrmypdf.Verbosity.debug else ocrmypdf.Verbosity.debug
), ),
manage_root_logger=True, manage_root_logger=True,
) )
log.setLevel(LOGLEVEL) log.setLevel(loglevel)
log.info( log.info(
f"Starting OCRmyPDF watcher with config:\n" f"Starting OCRmyPDF watcher with config:\n"
f"Input Directory: {INPUT_DIRECTORY}\n" f"Input Directory: {input_dir}\n"
f"Output Directory: {OUTPUT_DIRECTORY}\n" f"Output Directory: {output_dir}\n"
f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}\n" f"Output Directory Year & Month: {output_dir_year_month}\n"
f"Archive Directory: {ARCHIVE_DIRECTORY}" f"Archive Directory: {archive_dir}"
) )
log.debug( log.debug(
f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n" f"INPUT_DIRECTORY: {input_dir}\n"
f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n" f"OUTPUT_DIRECTORY: {output_dir}\n"
f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n" f"OUTPUT_DIRECTORY_YEAR_MONTH: {output_dir_year_month}\n"
f"ARCHIVE_DIRECTORY: {ARCHIVE_DIRECTORY}\n" f"ARCHIVE_DIRECTORY: {archive_dir}\n"
f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n" f"ON_SUCCESS_DELETE: {on_success_delete}\n"
f"ON_SUCCESS_ARCHIVE: {ON_SUCCESS_ARCHIVE}\n" f"ON_SUCCESS_ARCHIVE: {on_success_archive}\n"
f"DESKEW: {DESKEW}\n" f"DESKEW: {deskew}\n"
f"ARGS: {OCR_JSON_SETTINGS}\n" f"ARGS: {ocr_json_settings}\n"
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" f"POLL_NEW_FILE_SECONDS: {poll_new_file_seconds}\n"
f"RETRIES_LOADING_FILE: {RETRIES_LOADING_FILE}\n" f"RETRIES_LOADING_FILE: {retries_loading_file}\n"
f"USE_POLLING: {USE_POLLING}\n" f"USE_POLLING: {use_polling}\n"
f"LOGLEVEL: {LOGLEVEL}" f"LOGLEVEL: {loglevel}"
) )
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS: json_settings = json.loads(ocr_json_settings.read() if ocr_json_settings else '{}')
log.error('OCR_JSON_SETTINGS should not specify input file or output file')
if 'input_file' in json_settings or 'output_file' in json_settings:
log.error(
'OCR_JSON_SETTINGS (--ocr-json-settings) may not specify input/output file'
)
sys.exit(1) sys.exit(1)
handler = HandleObserverEvent(patterns=PATTERNS) handler = HandleObserverEvent(
if USE_POLLING: patterns=patterns.split(','),
settings={
'archive_dir': archive_dir,
'output_dir': output_dir,
'deskew': deskew,
'ocrmypdf_kwargs': json_settings,
'on_success_delete': on_success_delete,
'on_success_archive': on_success_archive,
'poll_new_file_seconds': poll_new_file_seconds,
'retries_loading_file': retries_loading_file,
'output_dir_year_month': output_dir_year_month,
},
)
if use_polling:
observer = PollingObserver() observer = PollingObserver()
else: else:
observer = Observer() observer = Observer()
observer.schedule(handler, INPUT_DIRECTORY, recursive=True) observer.schedule(handler, input_dir, recursive=True)
observer.start() observer.start()
typer.echo(f"Watching {input_dir} for new PDFs. Press Ctrl+C to exit.")
try: try:
while True: while True:
time.sleep(1) time.sleep(30)
except KeyboardInterrupt: except KeyboardInterrupt:
observer.stop() observer.stop()
observer.join() observer.join()
if __name__ == "__main__": if __name__ == "__main__":
main() app()
+1 -1
View File
@@ -61,7 +61,7 @@ test = [
"types-Pillow", "types-Pillow",
"types-humanfriendly", "types-humanfriendly",
] ]
watcher = ["watchdog>=1.0.2"] watcher = ["watchdog>=1.0.2", "typer[all]", "python-dotenv"]
webservice = ["Flask>=2.0.1"] webservice = ["Flask>=2.0.1"]
[project.scripts] [project.scripts]
+11
View File
@@ -33,6 +33,7 @@ from ocrmypdf.exceptions import (
EncryptedPdfError, EncryptedPdfError,
InputFileError, InputFileError,
PriorOcrFoundError, PriorOcrFoundError,
TaggedPDFError,
UnsupportedImageFormatError, UnsupportedImageFormatError,
) )
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
@@ -218,6 +219,16 @@ def validate_pdfinfo_options(context: PdfContext) -> None:
"form and all filled form fields. The output PDF will be " "form and all filled form fields. The output PDF will be "
"'flattened' and will no longer be fillable." "'flattened' and will no longer be fillable."
) )
if pdfinfo.is_tagged:
if options.force_ocr or options.skip_text or options.redo_ocr:
log.warning(
"This PDF is marked as a Tagged PDF. This often indicates "
"that the PDF was generated from an office document and does "
"not need OCR. PDF pages processed by OCRmyPDF may not be "
"tagged correctly."
)
else:
raise TaggedPDFError()
context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options) context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options)
+22 -1
View File
@@ -245,6 +245,18 @@ def check_options(options: Namespace, plugin_manager: PluginManager) -> None:
_check_plugin_options(options, plugin_manager) _check_plugin_options(options, plugin_manager)
def _in_docker():
return Path('/.dockerenv').exists()
def _in_snap():
try:
cgroup_text = Path('/proc/self/cgroup').read_text()
return 'snap.ocrmypdf' in cgroup_text
except FileNotFoundError:
return False
def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]: def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]:
if options.input_file == '-': if options.input_file == '-':
# stdin # stdin
@@ -268,7 +280,7 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
return target, os.fspath(options.input_file) return target, os.fspath(options.input_file)
except FileNotFoundError as e: except FileNotFoundError as e:
msg = f"File not found - {options.input_file}" msg = f"File not found - {options.input_file}"
if Path('/.dockerenv').exists(): # pragma: no cover if _in_docker(): # pragma: no cover
msg += ( msg += (
"\nDocker cannot your working directory unless you " "\nDocker cannot your working directory unless you "
"explicitly share it with the Docker container and set up" "explicitly share it with the Docker container and set up"
@@ -278,6 +290,15 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf" "\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf"
"\n" "\n"
) )
elif _in_snap(): # pragma: no cover
msg += (
"\nSnap applications cannot access files outside of "
"your home directory unless you explicitly allow it. "
"You may find it easier to use stdin/stdout:"
"\n"
"\tsnap run ocrmypdf - - <input.pdf >output.pdf"
"\n"
)
raise InputFileError(msg) from e raise InputFileError(msg) from e
+18 -6
View File
@@ -108,10 +108,16 @@ class EncryptedPdfError(ExitCodeException):
) )
class DigitalSignatureError(ExitCodeException): class TesseractConfigError(ExitCodeException):
"""Tesseract config can't be parsed."""
exit_code = ExitCode.invalid_config
message = "Error occurred while parsing a Tesseract configuration file"
class DigitalSignatureError(InputFileError):
"""PDF has a digital signature.""" """PDF has a digital signature."""
exit_code = ExitCode.input_file
message = dedent( message = dedent(
"""\ """\
Input PDF has a digital signature. OCR would alter the document, Input PDF has a digital signature. OCR would alter the document,
@@ -120,8 +126,14 @@ class DigitalSignatureError(ExitCodeException):
) )
class TesseractConfigError(ExitCodeException): class TaggedPDFError(InputFileError):
"""Tesseract config can't be parsed.""" """PDF is tagged."""
exit_code = ExitCode.invalid_config message = dedent(
message = "Error occurred while parsing a Tesseract configuration file" """\
This PDF is marked as a Tagged PDF. This often indicates
that the PDF was generated from an office document and does
not need OCR. Use --force-ocr, --skip-text or --redo-ocr to
override this error.
"""
)
+5 -3
View File
@@ -321,8 +321,10 @@ def remove_all_log_handlers(logger: logging.Logger) -> None:
def pikepdf_enable_mmap() -> None: def pikepdf_enable_mmap() -> None:
"""Enable pikepdf mmap.""" """Enable pikepdf mmap."""
try: try:
if pikepdf._core.set_access_default_mmap(True): pikepdf._core.set_access_default_mmap(True)
log.debug("pikepdf mmap enabled") log.debug(
"pikepdf mmap "
+ ('enabled' if pikepdf._core.get_access_default_mmap() else 'disabled')
)
except AttributeError: except AttributeError:
log.debug("pikepdf mmap not available") log.debug("pikepdf mmap not available")
log.debug("pikepdf mmap disabled")
+13 -1
View File
@@ -1038,7 +1038,11 @@ DEFAULT_EXECUTOR = SerialExecutor()
class PdfInfo: class PdfInfo:
"""Get summary information about a PDF.""" """Extract summary information about a PDF without retaining the PDF itself.
Crucially this lets us get the information in a pure Python format so that
it can be pickled and passed to a worker process.
"""
_has_acroform: bool = False _has_acroform: bool = False
_has_signature: bool = False _has_signature: bool = False
@@ -1078,6 +1082,9 @@ class PdfInfo:
elif Name.XFA in pdf.Root.AcroForm: elif Name.XFA in pdf.Root.AcroForm:
self._has_acroform = True self._has_acroform = True
self._has_signature = bool(pdf.Root.AcroForm.get(Name.SigFlags, 0) & 1) self._has_signature = bool(pdf.Root.AcroForm.get(Name.SigFlags, 0) & 1)
self._is_tagged = bool(
pdf.Root.get(Name.MarkInfo, {}).get(Name.Marked, False)
)
@property @property
def pages(self) -> Sequence[PageInfo | None]: def pages(self) -> Sequence[PageInfo | None]:
@@ -1105,6 +1112,11 @@ class PdfInfo:
"""Return True if the document annotations has a digital signature.""" """Return True if the document annotations has a digital signature."""
return self._has_signature return self._has_signature
@property
def is_tagged(self) -> bool:
"""Return True if the document catalog indicates this is a Tagged PDF."""
return self._is_tagged
@property @property
def filename(self) -> str | Path: def filename(self) -> str | Path:
"""Return filename of PDF.""" """Return filename of PDF."""
Binary file not shown.
Binary file not shown.
+26
View File
@@ -0,0 +1,26 @@
# SPDX-FileCopyrightText: 2023 James R. Barlow
# SPDX-License-Identifier: MPL-2.0
from __future__ import annotations
import argparse
import pytest
import ocrmypdf
def test_block_tagged(resources):
with pytest.raises(ocrmypdf.exceptions.TaggedPDFError):
ocrmypdf.ocr(resources / 'tagged.pdf', '_.pdf')
def test_force_tagged_warns(resources, outpdf, caplog):
caplog.set_level('WARNING')
ocrmypdf.ocr(
resources / 'tagged.pdf',
outpdf,
force_ocr=True,
plugins=['tests/plugins/tesseract_noop.py'],
)
assert 'marked as a Tagged PDF' in caplog.text