Compare commits

...
16 Commits
16 changed files with 372 additions and 109 deletions
+4 -3
View File
@@ -32,7 +32,7 @@ jobs:
python: "3.9" python: "3.9"
tesseract5: true tesseract5: true
- os: ubuntu-latest - os: ubuntu-latest
python: "3.12-dev" python: "3.12"
tesseract5: true tesseract5: true
#- os: ubuntu-latest #- os: ubuntu-latest
# python: "pypy3.9" # python: "pypy3.9"
@@ -113,7 +113,7 @@ jobs:
strategy: strategy:
matrix: matrix:
os: [macos-latest] os: [macos-latest]
python: ["3.10", "3.11", "3.12-dev"] python: ["3.10", "3.11", "3.12"]
env: env:
OS: ${{ matrix.os }} OS: ${{ matrix.os }}
@@ -133,6 +133,7 @@ jobs:
ghostscript \ ghostscript \
jbig2enc \ jbig2enc \
openjpeg \ openjpeg \
openssl \
pngquant \ pngquant \
tesseract tesseract
@@ -170,7 +171,7 @@ jobs:
strategy: strategy:
matrix: matrix:
os: [windows-latest] os: [windows-latest]
python: ["3.10", "3.11", "3.12-dev"] python: ["3.10", "3.11", "3.12"]
env: env:
OS: ${{ matrix.os }} OS: ${{ matrix.os }}
+3 -3
View File
@@ -152,7 +152,7 @@ The watcher service is included in the OCRmyPDF Docker image. To run it:
docker run \ docker run \
--volume <path to files to convert>:/input \ --volume <path to files to convert>:/input \
--volume <path to store results>:/output \ --volume <path to store results>:/output \
--volume <path to store processed originals>:/archive \ --volume <path to store processed originals>:/processed \
--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \ --env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
--env OCR_ON_SUCCESS_ARCHIVE=1 \ --env OCR_ON_SUCCESS_ARCHIVE=1 \
--env OCR_DESKEW=1 \ --env OCR_DESKEW=1 \
@@ -163,7 +163,7 @@ The watcher service is included in the OCRmyPDF Docker image. To run it:
This service will watch for a file that matches ``/input/\*.pdf``, This service will watch for a file that matches ``/input/\*.pdf``,
convert it to a OCRed PDF in ``/output/``, and move the processed convert it to a OCRed PDF in ``/output/``, and move the processed
original to ``/archive``. The parameters to this image are: original to ``/processed``. The parameters to this image are:
.. csv-table:: watcher.py parameters for Docker .. csv-table:: watcher.py parameters for Docker
:header: "Parameter", "Description" :header: "Parameter", "Description"
@@ -171,7 +171,7 @@ original to ``/archive``. The parameters to this image are:
"``--volume <path to files to convert>:/input``", "Files placed in this location will be OCRed" "``--volume <path to files to convert>:/input``", "Files placed in this location will be OCRed"
"``--volume <path to store results>:/output``", "This is where OCRed files will be stored" "``--volume <path to store results>:/output``", "This is where OCRed files will be stored"
"``--volume <path to store processed originals>:/archive``", "Archive processed originals here" "``--volume <path to store processed originals>:/processed``", "Archive processed originals here"
"``--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``" "``--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``"
"``--env OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals" "``--env OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals"
"``--env OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs" "``--env OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs"
+6 -6
View File
@@ -12,12 +12,12 @@ OCRmyPDF is also available in Docker images that packages recent
versions of all dependencies. versions of all dependencies.
For users who already have Docker installed this may be an easy and For users who already have Docker installed this may be an easy and
convenient option. However, it is less performant than a system convenient option.
installation and may require Docker engine configuration. OCRmyPDF
needs a generous amount of RAM, CPU cores, temporary storage On platforms other than Linux, Docker runs in a virtual machine, and so may
space, whether running in a Docker container or on its own. It may be be less performant. You may also want to adjust the Docker virtual machine's
necessary to ensure the container is provisioned with additional memory and CPU allocation. On Linux, the Docker image runs natively and
resources. performance is comparable to a system installation.
.. _docker-install: .. _docker-install:
+16
View File
@@ -28,6 +28,22 @@ tagged yet.
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg .. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
v15.3.1
=======
- Fixed an issue with logging settings for misc/watcher.py introduced in the
previous release. :issue:`1180`
- Updated documentation on Docker performance concerns.
v15.3.0
=======
- Update misc/watcher.py to improve command line interface using Typer, and
support ``.env`` specification of environment variables. Improved error
messages. Thanks to @mflagg2814 for the PR that prompted this improvement.
- Improved error message when a file cannot be read because we are running in
a snap container.
v15.2.0 v15.2.0
======= =======
+231 -75
View File
@@ -5,107 +5,122 @@
"""Watch a directory for new PDFs and OCR them.""" """Watch a directory for new PDFs and OCR them."""
from __future__ import annotations # Do not enable annotations!
# https://github.com/tiangolo/typer/discussions/598
# from __future__ import annotations
import json import json
import logging import logging
import os
import shutil import shutil
import sys import sys
import time import time
from datetime import datetime from datetime import datetime
from enum import Enum
from pathlib import Path from pathlib import Path
from typing import Annotated, Any
import pikepdf import pikepdf
import typer
from dotenv import load_dotenv
from watchdog.events import PatternMatchingEventHandler from watchdog.events import PatternMatchingEventHandler
from watchdog.observers import Observer from watchdog.observers import Observer
from watchdog.observers.polling import PollingObserver from watchdog.observers.polling import PollingObserver
import ocrmypdf import ocrmypdf
load_dotenv()
# pylint: disable=logging-format-interpolation # pylint: disable=logging-format-interpolation
app = typer.Typer(name="ocrmypdf-watcher")
def getenv_bool(name: str, default: str = 'False'):
return os.getenv(name, default).lower() in ('true', 'yes', 'y', '1')
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
ARCHIVE_DIRECTORY = os.getenv('OCR_ARCHIVE_DIRECTORY', '/processed')
OUTPUT_DIRECTORY_YEAR_MONTH = getenv_bool('OCR_OUTPUT_DIRECTORY_YEAR_MONTH')
ON_SUCCESS_DELETE = getenv_bool('OCR_ON_SUCCESS_DELETE')
ON_SUCCESS_ARCHIVE = getenv_bool('OCR_ON_SUCCESS_ARCHIVE')
DESKEW = getenv_bool('OCR_DESKEW')
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
USE_POLLING = getenv_bool('OCR_USE_POLLING')
RETRIES_LOADING_FILE = int(os.getenv('OCR_RETRIES_LOADING_FILE', '5'))
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
PATTERNS = ['*.pdf', '*.PDF']
log = logging.getLogger('ocrmypdf-watcher') log = logging.getLogger('ocrmypdf-watcher')
def get_output_dir(root, basename): class LoggingLevelEnum(str, Enum):
if OUTPUT_DIRECTORY_YEAR_MONTH: """Enum for logging levels."""
DEBUG = "DEBUG"
INFO = "INFO"
WARNING = "WARNING"
ERROR = "ERROR"
CRITICAL = "CRITICAL"
def get_output_dir(root: Path, basename: str, output_dir_year_month: bool) -> Path:
if output_dir_year_month:
today = datetime.today() today = datetime.today()
output_directory_year_month = ( output_directory_year_month = root / str(today.year) / f'{today.month:02d}'
Path(root) / str(today.year) / f'{today.month:02d}'
)
if not output_directory_year_month.exists(): if not output_directory_year_month.exists():
output_directory_year_month.mkdir(parents=True, exist_ok=True) output_directory_year_month.mkdir(parents=True, exist_ok=True)
output_path = Path(output_directory_year_month) / basename output_path = Path(output_directory_year_month) / basename
else: else:
output_path = Path(OUTPUT_DIRECTORY) / basename output_path = root / basename
return output_path return output_path
def wait_for_file_ready(file_path): def wait_for_file_ready(
file_path: Path, poll_new_file_seconds: int, retries_loading_file: int
):
# This loop waits to make sure that the file is completely loaded on # This loop waits to make sure that the file is completely loaded on
# disk before attempting to read. Docker sometimes will publish the # disk before attempting to read. Docker sometimes will publish the
# watchdog event before the file is actually fully on disk, causing # watchdog event before the file is actually fully on disk, causing
# pikepdf to fail. # pikepdf to fail.
retries = RETRIES_LOADING_FILE tries = retries_loading_file + 1
while retries: while tries:
try: try:
pdf = pikepdf.open(file_path) with pikepdf.Pdf.open(file_path) as pdf:
except (FileNotFoundError, pikepdf.PdfError) as e: log.debug(f"{file_path} ready with {pdf.pages} pages")
return True
except (FileNotFoundError, OSError) as e:
log.info(f"File {file_path} is not ready yet") log.info(f"File {file_path} is not ready yet")
log.debug("Exception was", exc_info=e) log.debug("Exception was", exc_info=e)
time.sleep(POLL_NEW_FILE_SECONDS) time.sleep(poll_new_file_seconds)
retries -= 1 tries -= 1
else: except pikepdf.PdfError as e:
pdf.close() log.info(f"File {file_path} is not full written yet")
return True log.debug("Exception was", exc_info=e)
time.sleep(poll_new_file_seconds)
tries -= 1
return False return False
def execute_ocrmypdf(file_path): def execute_ocrmypdf(
file_path = Path(file_path) *,
output_path = get_output_dir(OUTPUT_DIRECTORY, file_path.name) file_path: Path,
archive_dir: Path,
output_dir: Path,
deskew: bool,
ocrmypdf_kwargs: dict[str, Any],
on_success_delete: bool,
on_success_archive: bool,
poll_new_file_seconds: int,
retries_loading_file: int,
output_dir_year_month: bool,
):
output_path = get_output_dir(output_dir, file_path.name, output_dir_year_month)
log.info("-" * 20) log.info("-" * 20)
log.info(f'New file: {file_path}. Waiting until fully loaded...') log.info(f'New file: {file_path}. Waiting until fully written...')
if not wait_for_file_ready(file_path): if not wait_for_file_ready(file_path, poll_new_file_seconds, retries_loading_file):
log.info(f"Gave up waiting for {file_path} to become ready") log.info(f"Gave up waiting for {file_path} to become ready")
return return
log.info(f'Attempting to OCRmyPDF to: {output_path}') log.info(f'Attempting to OCRmyPDF to: {output_path}')
exit_code = ocrmypdf.ocr( exit_code = ocrmypdf.ocr(
input_file=file_path, input_file=file_path,
output_file=output_path, output_file=output_path,
deskew=DESKEW, deskew=deskew,
**OCR_JSON_SETTINGS, **ocrmypdf_kwargs,
) )
if exit_code == 0: if exit_code == 0:
if ON_SUCCESS_DELETE: if on_success_delete:
log.info(f'OCR is done. Deleting: {file_path}') log.info(f'OCR is done. Deleting: {file_path}')
file_path.unlink() file_path.unlink()
elif ON_SUCCESS_ARCHIVE: elif on_success_archive:
log.info(f'OCR is done. Archiving {file_path.name} to {ARCHIVE_DIRECTORY}') log.info(f'OCR is done. Archiving {file_path.name} to {archive_dir}')
shutil.move(file_path, f'{ARCHIVE_DIRECTORY}/{file_path.name}') shutil.move(file_path, f'{archive_dir}/{file_path.name}')
else: else:
log.info('OCR is done') log.info('OCR is done')
else: else:
@@ -113,61 +128,202 @@ def execute_ocrmypdf(file_path):
class HandleObserverEvent(PatternMatchingEventHandler): class HandleObserverEvent(PatternMatchingEventHandler):
def __init__(
self,
patterns=None,
ignore_patterns=None,
ignore_directories=False,
case_sensitive=False,
settings={},
):
super().__init__(
patterns=patterns,
ignore_patterns=ignore_patterns,
ignore_directories=ignore_directories,
case_sensitive=case_sensitive,
)
self._settings = settings
def on_any_event(self, event): def on_any_event(self, event):
if event.event_type in ['created']: if event.event_type in ['created']:
execute_ocrmypdf(event.src_path) execute_ocrmypdf(event.src_path, **self._settings)
def main(): @app.command()
def main(
input_dir: Annotated[
Path,
typer.Argument(
envvar='OCR_INPUT_DIRECTORY',
exists=True,
file_okay=False,
dir_okay=True,
readable=True,
resolve_path=True,
),
] = '/input',
output_dir: Annotated[
Path,
typer.Argument(
envvar='OCR_OUTPUT_DIRECTORY',
exists=True,
file_okay=False,
dir_okay=True,
writable=True,
resolve_path=True,
),
] = '/output',
archive_dir: Annotated[
Path,
typer.Argument(
envvar='OCR_ARCHIVE_DIRECTORY',
exists=True,
file_okay=False,
dir_okay=True,
writable=True,
resolve_path=True,
),
] = '/processed',
output_dir_year_month: Annotated[
bool,
typer.Option(
envvar='OCR_OUTPUT_DIRECTORY_YEAR_MONTH',
help='Create a subdirectory in the output directory for each year and month',
),
] = False,
on_success_delete: Annotated[
bool,
typer.Option(
envvar='OCR_ON_SUCCESS_DELETE',
help='Delete the input file after successful OCR',
),
] = False,
on_success_archive: Annotated[
bool,
typer.Option(
envvar='OCR_ON_SUCCESS_ARCHIVE',
help='Archive the input file after successful OCR',
),
] = False,
deskew: Annotated[
bool,
typer.Option(
envvar='OCR_DESKEW',
help='Deskew the input file before OCR',
),
] = False,
ocr_json_settings: Annotated[
typer.FileText,
typer.Option(
envvar='OCR_JSON_SETTINGS',
help='JSON settings to pass to OCRmyPDF',
),
] = None,
poll_new_file_seconds: Annotated[
int,
typer.Option(
envvar='OCR_POLL_NEW_FILE_SECONDS',
help='Seconds to wait before polling a new file',
min=0,
),
] = 1,
use_polling: Annotated[
bool,
typer.Option(
envvar='OCR_USE_POLLING',
help='Use polling instead of filesystem events',
),
] = False,
retries_loading_file: Annotated[
int,
typer.Option(
envvar='OCR_RETRIES_LOADING_FILE',
help='Number of times to retry loading a file before giving up',
min=0,
),
] = 5,
loglevel: Annotated[
LoggingLevelEnum,
typer.Option(
envvar='OCR_LOGLEVEL',
help='Logging level',
),
] = LoggingLevelEnum.INFO,
patterns: Annotated[
str,
typer.Option(
envvar='OCR_PATTERNS',
help='File patterns to watch',
),
] = '*.pdf,*.PDF',
):
ocrmypdf.configure_logging( ocrmypdf.configure_logging(
verbosity=( verbosity=(
ocrmypdf.Verbosity.default ocrmypdf.Verbosity.default
if LOGLEVEL != 'DEBUG' if loglevel != LoggingLevelEnum.DEBUG
else ocrmypdf.Verbosity.debug else ocrmypdf.Verbosity.debug
), ),
manage_root_logger=True, manage_root_logger=True,
) )
log.setLevel(LOGLEVEL) log.setLevel(loglevel.value)
log.info( log.info(
f"Starting OCRmyPDF watcher with config:\n" f"Starting OCRmyPDF watcher with config:\n"
f"Input Directory: {INPUT_DIRECTORY}\n" f"Input Directory: {input_dir}\n"
f"Output Directory: {OUTPUT_DIRECTORY}\n" f"Output Directory: {output_dir}\n"
f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}\n" f"Output Directory Year & Month: {output_dir_year_month}\n"
f"Archive Directory: {ARCHIVE_DIRECTORY}" f"Archive Directory: {archive_dir}"
) )
log.debug( log.debug(
f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n" f"INPUT_DIRECTORY: {input_dir}\n"
f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n" f"OUTPUT_DIRECTORY: {output_dir}\n"
f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n" f"OUTPUT_DIRECTORY_YEAR_MONTH: {output_dir_year_month}\n"
f"ARCHIVE_DIRECTORY: {ARCHIVE_DIRECTORY}\n" f"ARCHIVE_DIRECTORY: {archive_dir}\n"
f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n" f"ON_SUCCESS_DELETE: {on_success_delete}\n"
f"ON_SUCCESS_ARCHIVE: {ON_SUCCESS_ARCHIVE}\n" f"ON_SUCCESS_ARCHIVE: {on_success_archive}\n"
f"DESKEW: {DESKEW}\n" f"DESKEW: {deskew}\n"
f"ARGS: {OCR_JSON_SETTINGS}\n" f"ARGS: {ocr_json_settings}\n"
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" f"POLL_NEW_FILE_SECONDS: {poll_new_file_seconds}\n"
f"RETRIES_LOADING_FILE: {RETRIES_LOADING_FILE}\n" f"RETRIES_LOADING_FILE: {retries_loading_file}\n"
f"USE_POLLING: {USE_POLLING}\n" f"USE_POLLING: {use_polling}\n"
f"LOGLEVEL: {LOGLEVEL}" f"LOGLEVEL: {loglevel.value}"
) )
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS: json_settings = json.loads(ocr_json_settings.read() if ocr_json_settings else '{}')
log.error('OCR_JSON_SETTINGS should not specify input file or output file')
if 'input_file' in json_settings or 'output_file' in json_settings:
log.error(
'OCR_JSON_SETTINGS (--ocr-json-settings) may not specify input/output file'
)
sys.exit(1) sys.exit(1)
handler = HandleObserverEvent(patterns=PATTERNS) handler = HandleObserverEvent(
if USE_POLLING: patterns=patterns.split(','),
settings={
'archive_dir': archive_dir,
'output_dir': output_dir,
'deskew': deskew,
'ocrmypdf_kwargs': json_settings,
'on_success_delete': on_success_delete,
'on_success_archive': on_success_archive,
'poll_new_file_seconds': poll_new_file_seconds,
'retries_loading_file': retries_loading_file,
'output_dir_year_month': output_dir_year_month,
},
)
if use_polling:
observer = PollingObserver() observer = PollingObserver()
else: else:
observer = Observer() observer = Observer()
observer.schedule(handler, INPUT_DIRECTORY, recursive=True) observer.schedule(handler, input_dir, recursive=True)
observer.start() observer.start()
typer.echo(f"Watching {input_dir} for new PDFs. Press Ctrl+C to exit.")
try: try:
while True: while True:
time.sleep(1) time.sleep(30)
except KeyboardInterrupt: except KeyboardInterrupt:
observer.stop() observer.stop()
observer.join() observer.join()
if __name__ == "__main__": if __name__ == "__main__":
main() app()
+1 -1
View File
@@ -61,7 +61,7 @@ test = [
"types-Pillow", "types-Pillow",
"types-humanfriendly", "types-humanfriendly",
] ]
watcher = ["watchdog>=1.0.2"] watcher = ["watchdog>=1.0.2", "typer[all]", "python-dotenv"]
webservice = ["Flask>=2.0.1"] webservice = ["Flask>=2.0.1"]
[project.scripts] [project.scripts]
+3 -2
View File
@@ -7,6 +7,7 @@ from __future__ import annotations
import logging import logging
import re import re
from contextlib import suppress
from math import pi from math import pi
from os import fspath from os import fspath
from pathlib import Path from pathlib import Path
@@ -350,7 +351,7 @@ def generate_hocr(
tesseract_log_output(stdout) tesseract_log_output(stdout)
# The sidecar text file will get the suffix .txt; rename it to # The sidecar text file will get the suffix .txt; rename it to
# whatever caller wants it named # whatever caller wants it named
if prefix.with_suffix('.txt').exists(): with suppress(FileNotFoundError):
prefix.with_suffix('.txt').replace(output_text) prefix.with_suffix('.txt').replace(output_text)
@@ -406,7 +407,7 @@ def generate_pdf(
try: try:
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True) p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
stdout = p.stdout stdout = p.stdout
if prefix.with_suffix('.txt').exists(): with suppress(FileNotFoundError):
prefix.with_suffix('.txt').replace(output_text) prefix.with_suffix('.txt').replace(output_text)
except TimeoutExpired: except TimeoutExpired:
page_timedout(timeout) page_timedout(timeout)
+22 -5
View File
@@ -14,7 +14,7 @@ from collections.abc import Iterable, Iterator, Sequence
from contextlib import suppress from contextlib import suppress
from datetime import datetime, timezone from datetime import datetime, timezone
from pathlib import Path from pathlib import Path
from shutil import copyfileobj from shutil import copyfileobj, copystat
from typing import Any, BinaryIO, TypeVar, cast from typing import Any, BinaryIO, TypeVar, cast
import img2pdf import img2pdf
@@ -33,6 +33,7 @@ from ocrmypdf.exceptions import (
EncryptedPdfError, EncryptedPdfError,
InputFileError, InputFileError,
PriorOcrFoundError, PriorOcrFoundError,
TaggedPDFError,
UnsupportedImageFormatError, UnsupportedImageFormatError,
) )
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
@@ -218,6 +219,16 @@ def validate_pdfinfo_options(context: PdfContext) -> None:
"form and all filled form fields. The output PDF will be " "form and all filled form fields. The output PDF will be "
"'flattened' and will no longer be fillable." "'flattened' and will no longer be fillable."
) )
if pdfinfo.is_tagged:
if options.force_ocr or options.skip_text or options.redo_ocr:
log.warning(
"This PDF is marked as a Tagged PDF. This often indicates "
"that the PDF was generated from an office document and does "
"not need OCR. PDF pages processed by OCRmyPDF may not be "
"tagged correctly."
)
else:
raise TaggedPDFError()
context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options) context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options)
@@ -1079,14 +1090,14 @@ def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext) -> Pat
def copy_final( def copy_final(
input_file: Path, output_file: str | Path | BinaryIO, _context: PdfContext input_file: Path, output_file: str | Path | BinaryIO, context: PdfContext
) -> None: ) -> None:
"""Copy the final temporary file to the output destination. """Copy the final temporary file to the output destination.
Args: Args:
input_file (Path): The input file to copy. input_file (Path): The intermediate input file to copy.
output_file (str | Path | BinaryIO): The output file to copy to. output_file (str | Path | BinaryIO): The output file to copy to.
_context (PdfContext): The PDF context. context (PdfContext): The PDF context.
Returns: Returns:
None None
@@ -1105,5 +1116,11 @@ def copy_final(
# At this point we overwrite the output_file specified by the user # At this point we overwrite the output_file specified by the user
# use copyfileobj because then we use open() to create the file and # use copyfileobj because then we use open() to create the file and
# get the appropriate umask, ownership, etc. # get the appropriate umask, ownership, etc.
with open(output_file, 'wb') as output_stream: with open(output_file, 'w+b') as output_stream:
copyfileobj(input_stream, output_stream) copyfileobj(input_stream, output_stream)
# Attempt to copy file attributes from input to output
with suppress(OSError):
# Copy original file's permissions, ownership, etc. if possible
copystat(context.options.input_file, output_file)
# Set output file's modification time to now
Path(output_file).touch(exist_ok=True)
+22 -1
View File
@@ -245,6 +245,18 @@ def check_options(options: Namespace, plugin_manager: PluginManager) -> None:
_check_plugin_options(options, plugin_manager) _check_plugin_options(options, plugin_manager)
def _in_docker():
return Path('/.dockerenv').exists()
def _in_snap():
try:
cgroup_text = Path('/proc/self/cgroup').read_text()
return 'snap.ocrmypdf' in cgroup_text
except FileNotFoundError:
return False
def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]: def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]:
if options.input_file == '-': if options.input_file == '-':
# stdin # stdin
@@ -268,7 +280,7 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
return target, os.fspath(options.input_file) return target, os.fspath(options.input_file)
except FileNotFoundError as e: except FileNotFoundError as e:
msg = f"File not found - {options.input_file}" msg = f"File not found - {options.input_file}"
if Path('/.dockerenv').exists(): # pragma: no cover if _in_docker(): # pragma: no cover
msg += ( msg += (
"\nDocker cannot your working directory unless you " "\nDocker cannot your working directory unless you "
"explicitly share it with the Docker container and set up" "explicitly share it with the Docker container and set up"
@@ -278,6 +290,15 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf" "\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf"
"\n" "\n"
) )
elif _in_snap(): # pragma: no cover
msg += (
"\nSnap applications cannot access files outside of "
"your home directory unless you explicitly allow it. "
"You may find it easier to use stdin/stdout:"
"\n"
"\tsnap run ocrmypdf - - <input.pdf >output.pdf"
"\n"
)
raise InputFileError(msg) from e raise InputFileError(msg) from e
+18 -6
View File
@@ -108,10 +108,16 @@ class EncryptedPdfError(ExitCodeException):
) )
class DigitalSignatureError(ExitCodeException): class TesseractConfigError(ExitCodeException):
"""Tesseract config can't be parsed."""
exit_code = ExitCode.invalid_config
message = "Error occurred while parsing a Tesseract configuration file"
class DigitalSignatureError(InputFileError):
"""PDF has a digital signature.""" """PDF has a digital signature."""
exit_code = ExitCode.input_file
message = dedent( message = dedent(
"""\ """\
Input PDF has a digital signature. OCR would alter the document, Input PDF has a digital signature. OCR would alter the document,
@@ -120,8 +126,14 @@ class DigitalSignatureError(ExitCodeException):
) )
class TesseractConfigError(ExitCodeException): class TaggedPDFError(InputFileError):
"""Tesseract config can't be parsed.""" """PDF is tagged."""
exit_code = ExitCode.invalid_config message = dedent(
message = "Error occurred while parsing a Tesseract configuration file" """\
This PDF is marked as a Tagged PDF. This often indicates
that the PDF was generated from an office document and does
not need OCR. Use --force-ocr, --skip-text or --redo-ocr to
override this error.
"""
)
+5 -3
View File
@@ -321,8 +321,10 @@ def remove_all_log_handlers(logger: logging.Logger) -> None:
def pikepdf_enable_mmap() -> None: def pikepdf_enable_mmap() -> None:
"""Enable pikepdf mmap.""" """Enable pikepdf mmap."""
try: try:
if pikepdf._core.set_access_default_mmap(True): pikepdf._core.set_access_default_mmap(True)
log.debug("pikepdf mmap enabled") log.debug(
"pikepdf mmap "
+ ('enabled' if pikepdf._core.get_access_default_mmap() else 'disabled')
)
except AttributeError: except AttributeError:
log.debug("pikepdf mmap not available") log.debug("pikepdf mmap not available")
log.debug("pikepdf mmap disabled")
+13 -1
View File
@@ -1038,7 +1038,11 @@ DEFAULT_EXECUTOR = SerialExecutor()
class PdfInfo: class PdfInfo:
"""Get summary information about a PDF.""" """Extract summary information about a PDF without retaining the PDF itself.
Crucially this lets us get the information in a pure Python format so that
it can be pickled and passed to a worker process.
"""
_has_acroform: bool = False _has_acroform: bool = False
_has_signature: bool = False _has_signature: bool = False
@@ -1078,6 +1082,9 @@ class PdfInfo:
elif Name.XFA in pdf.Root.AcroForm: elif Name.XFA in pdf.Root.AcroForm:
self._has_acroform = True self._has_acroform = True
self._has_signature = bool(pdf.Root.AcroForm.get(Name.SigFlags, 0) & 1) self._has_signature = bool(pdf.Root.AcroForm.get(Name.SigFlags, 0) & 1)
self._is_tagged = bool(
pdf.Root.get(Name.MarkInfo, {}).get(Name.Marked, False)
)
@property @property
def pages(self) -> Sequence[PageInfo | None]: def pages(self) -> Sequence[PageInfo | None]:
@@ -1105,6 +1112,11 @@ class PdfInfo:
"""Return True if the document annotations has a digital signature.""" """Return True if the document annotations has a digital signature."""
return self._has_signature return self._has_signature
@property
def is_tagged(self) -> bool:
"""Return True if the document catalog indicates this is a Tagged PDF."""
return self._is_tagged
@property @property
def filename(self) -> str | Path: def filename(self) -> str | Path:
"""Return filename of PDF.""" """Return filename of PDF."""
+2 -3
View File
@@ -25,9 +25,8 @@ def is_macos():
def running_in_docker(): def running_in_docker():
# Docker creates a file named /.dockerenv (newer versions) or # Docker creates a file named /.dockerenv in all supported versions
# /.dockerinit (older) -- this is undocumented, not an offical test return Path('/.dockerenv').exists()
return Path('/.dockerenv').exists() or Path('/.dockerinit').exists()
def have_unpaper(): def have_unpaper():
Binary file not shown.
Binary file not shown.
+26
View File
@@ -0,0 +1,26 @@
# SPDX-FileCopyrightText: 2023 James R. Barlow
# SPDX-License-Identifier: MPL-2.0
from __future__ import annotations
import argparse
import pytest
import ocrmypdf
def test_block_tagged(resources):
with pytest.raises(ocrmypdf.exceptions.TaggedPDFError):
ocrmypdf.ocr(resources / 'tagged.pdf', '_.pdf')
def test_force_tagged_warns(resources, outpdf, caplog):
caplog.set_level('WARNING')
ocrmypdf.ocr(
resources / 'tagged.pdf',
outpdf,
force_ocr=True,
plugins=['tests/plugins/tesseract_noop.py'],
)
assert 'marked as a Tagged PDF' in caplog.text