Writing the output PDF to stdout (ocrmypdf in.pdf -) previously relied on an honor system: no in-process code -- third-party libraries, plugins, or stray print() calls -- was supposed to write to stdout, enforced only indirectly. A single accidental write to fd 1 would silently corrupt the output PDF. Enforce this at the OS level. At CLI startup, before plugins load or any worker process/thread starts, save the real stdout via os.dup() and point fd 1 at stderr, so stray writes are diverted to stderr while only the final "produce the PDF" step writes to the preserved descriptor. Exposed as the opt-in public API function configure_stdout_protection(), mirroring configure_logging(); it is not enabled inside ocr() so in-process library users keep their own stdout. Also fix check_requested_output_file() to test the preserved real stdout for tty-ness, since after the redirect sys.stdout reports stderr's status. Fold unreleased v17.7.2 notes into v17.8.0.
88 lines
2.0 KiB
Python
88 lines
2.0 KiB
Python
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
|
# SPDX-License-Identifier: MPL-2.0
|
|
|
|
"""Adds OCR layer to PDFs."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pluggy import HookimplMarker as _HookimplMarker
|
|
|
|
from ocrmypdf import helpers, hocrtransform, pdfa, pdfinfo
|
|
from ocrmypdf._concurrent import Executor
|
|
from ocrmypdf._defaults import PROGRAM_NAME
|
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
|
from ocrmypdf._options import OcrOptions, TaggedPdfMode
|
|
from ocrmypdf._pipelines._common import (
|
|
configure_debug_logging,
|
|
)
|
|
from ocrmypdf._version import __version__
|
|
from ocrmypdf.api import (
|
|
Verbosity,
|
|
configure_logging,
|
|
configure_stdout_protection,
|
|
ocr,
|
|
)
|
|
from ocrmypdf.exceptions import (
|
|
BadArgsError,
|
|
DpiError,
|
|
EncryptedPdfError,
|
|
ExitCode,
|
|
ExitCodeException,
|
|
InputFileError,
|
|
MissingDependencyError,
|
|
OutputFileAccessError,
|
|
PriorOcrFoundError,
|
|
SubprocessOutputError,
|
|
TesseractConfigError,
|
|
UnsupportedImageFormatError,
|
|
)
|
|
from ocrmypdf.models.ocr_element import (
|
|
Baseline,
|
|
BoundingBox,
|
|
FontInfo,
|
|
OcrClass,
|
|
OcrElement,
|
|
)
|
|
from ocrmypdf.pluginspec import OcrEngine, OrientationConfidence
|
|
|
|
hookimpl = _HookimplMarker('ocrmypdf')
|
|
|
|
__all__ = [
|
|
'__version__',
|
|
'BadArgsError',
|
|
'Baseline',
|
|
'BoundingBox',
|
|
'configure_debug_logging',
|
|
'configure_logging',
|
|
'configure_stdout_protection',
|
|
'DpiError',
|
|
'EncryptedPdfError',
|
|
'Executor',
|
|
'ExitCode',
|
|
'ExitCodeException',
|
|
'FontInfo',
|
|
'helpers',
|
|
'hocrtransform',
|
|
'hookimpl',
|
|
'InputFileError',
|
|
'MissingDependencyError',
|
|
'ocr',
|
|
'OcrClass',
|
|
'OcrElement',
|
|
'OcrEngine',
|
|
'OcrOptions',
|
|
'OrientationConfidence',
|
|
'OutputFileAccessError',
|
|
'PageContext',
|
|
'pdfa',
|
|
'PdfContext',
|
|
'pdfinfo',
|
|
'PriorOcrFoundError',
|
|
'PROGRAM_NAME',
|
|
'SubprocessOutputError',
|
|
'TaggedPdfMode',
|
|
'TesseractConfigError',
|
|
'UnsupportedImageFormatError',
|
|
'Verbosity',
|
|
]
|