diff --git a/src/ocrmypdf/_jobcontext.py b/src/ocrmypdf/_jobcontext.py index b1c775d9..798650ff 100644 --- a/src/ocrmypdf/_jobcontext.py +++ b/src/ocrmypdf/_jobcontext.py @@ -10,9 +10,11 @@ from argparse import Namespace from collections.abc import Iterator from copy import copy from pathlib import Path +from typing import Union from pluggy import PluginManager +from ocrmypdf._options import OCROptions from ocrmypdf.pdfinfo import PdfInfo from ocrmypdf.pdfinfo.info import PageInfo @@ -20,25 +22,38 @@ from ocrmypdf.pdfinfo.info import PageInfo class PdfContext: """Holds the context for a particular run of the pipeline.""" - options: Namespace #: The specified options for processing this PDF. + options: Union[Namespace, OCROptions] #: The specified options for processing this PDF. origin: Path #: The filename of the original input file. pdfinfo: PdfInfo #: Detailed data for this PDF. plugin_manager: PluginManager #: PluginManager for processing the current PDF. def __init__( self, - options: Namespace, + options: Union[Namespace, OCROptions], work_folder: Path, origin: Path, pdfinfo: PdfInfo, plugin_manager, ): - self.options = options + # Accept both types during transition + if isinstance(options, Namespace): + self.options = OCROptions.from_namespace(options) + self._legacy_options = options + else: + self.options = options + self._legacy_options = None self.work_folder = work_folder self.origin = origin self.pdfinfo = pdfinfo self.plugin_manager = plugin_manager + @property + def legacy_options(self) -> Namespace: + """Provide Namespace for plugin compatibility.""" + if self._legacy_options is None: + self._legacy_options = self.options.to_namespace() + return self._legacy_options + def get_path(self, name: str) -> Path: """Generate a ``Path`` for an intermediate file involved in processing. @@ -67,7 +82,7 @@ class PageContext: capable of their serializing themselves via ``__getstate__``. """ - options: Namespace #: The specified options for processing this PDF. + options: Union[Namespace, OCROptions] #: The specified options for processing this PDF. origin: Path #: The filename of the original input file. pageno: int #: This page number (zero-based). pageinfo: PageInfo #: Information on this page. @@ -93,8 +108,11 @@ class PageContext: state = self.__dict__.copy() state['options'] = copy(self.options) - if not isinstance(state['options'].input_file, str | bytes | os.PathLike): - state['options'].input_file = 'stream' - if not isinstance(state['options'].output_file, str | bytes | os.PathLike): - state['options'].output_file = 'stream' + # Handle both OCROptions and Namespace + if hasattr(state['options'], 'input_file'): + if not isinstance(state['options'].input_file, str | bytes | os.PathLike): + state['options'].input_file = 'stream' + if hasattr(state['options'], 'output_file'): + if not isinstance(state['options'].output_file, str | bytes | os.PathLike): + state['options'].output_file = 'stream' return state diff --git a/src/ocrmypdf/_options.py b/src/ocrmypdf/_options.py new file mode 100644 index 00000000..5de82f1c --- /dev/null +++ b/src/ocrmypdf/_options.py @@ -0,0 +1,228 @@ +# SPDX-FileCopyrightText: 2024 James R. Barlow +# SPDX-License-Identifier: MPL-2.0 + +"""Internal options model for OCRmyPDF.""" + +from __future__ import annotations + +import os +from argparse import Namespace +from collections.abc import Iterable, Sequence +from copy import copy +from pathlib import Path +from typing import Any, BinaryIO, Union + +from pydantic import BaseModel, Field, validator + +from ocrmypdf._defaults import DEFAULT_LANGUAGE, DEFAULT_ROTATE_PAGES_THRESHOLD + +PathOrIO = Union[BinaryIO, Path, str, bytes] + + +class OCROptions(BaseModel): + """Internal options model that can masquerade as argparse.Namespace. + + This model provides proper typing and validation while maintaining + compatibility with existing code that expects argparse.Namespace behavior. + """ + + # I/O options + input_file: PathOrIO + output_file: PathOrIO + sidecar: PathOrIO | None = None + + # Core OCR options + languages: list[str] = Field(default_factory=lambda: [DEFAULT_LANGUAGE]) + output_type: str = 'pdfa' + force_ocr: bool = False + skip_text: bool = False + redo_ocr: bool = False + + # Job control + jobs: int | None = None + use_threads: bool = True + progress_bar: bool = True + quiet: bool = False + verbose: int = 0 + keep_temporary_files: bool = False + + # Image processing + image_dpi: int | None = None + deskew: bool = False + clean: bool = False + clean_final: bool = False + rotate_pages: bool = False + remove_background: bool = False + remove_vectors: bool = False + oversample: int = 0 + unpaper_args: str | None = None + + # OCR behavior + skip_big: float | None = None + pages: str | None = None + invalidate_digital_signatures: bool = False + + # Metadata + title: str | None = None + author: str | None = None + subject: str | None = None + keywords: str | None = None + + # Optimization + optimize: int | None = None + jpg_quality: int | None = None + png_quality: int | None = None + jbig2_lossy: bool | None = None + jbig2_page_group_size: int | None = None + jbig2_threshold: float | None = None + + # Advanced options + max_image_mpixels: float = 250.0 + pdf_renderer: str = 'auto' + tesseract_config: Iterable[str] | None = None + tesseract_pagesegmode: int | None = None + tesseract_oem: int | None = None + tesseract_thresholding: int | None = None + tesseract_timeout: float | None = None + tesseract_non_ocr_timeout: float | None = None + tesseract_downsample_above: int | None = None + tesseract_downsample_large_images: bool | None = None + rotate_pages_threshold: float = DEFAULT_ROTATE_PAGES_THRESHOLD + pdfa_image_compression: str | None = None + color_conversion_strategy: str | None = None + user_words: os.PathLike | None = None + user_patterns: os.PathLike | None = None + fast_web_view: float | None = None + continue_on_soft_render_error: bool | None = None + + # Plugin system + plugins: Sequence[Path | str] | None = None + + # Store any extra attributes (for plugins and dynamic options) + _extra_attrs: dict[str, Any] = Field(default_factory=dict, exclude=True) + + def __getattr__(self, name: str) -> Any: + """Allow attribute access like argparse.Namespace.""" + if name in self._extra_attrs: + return self._extra_attrs[name] + raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'") + + def __setattr__(self, name: str, value: Any) -> None: + """Allow attribute setting like argparse.Namespace.""" + if name.startswith('_') or name in self.__fields__: + super().__setattr__(name, value) + else: + if not hasattr(self, '_extra_attrs'): + super().__setattr__('_extra_attrs', {}) + self._extra_attrs[name] = value + + def __delattr__(self, name: str) -> None: + """Allow attribute deletion like argparse.Namespace.""" + if name in self.__fields__: + super().__delattr__(name) + elif name in self._extra_attrs: + del self._extra_attrs[name] + else: + raise AttributeError(f"'{type(self).__name__}' object has no attribute '{name}'") + + @classmethod + def from_namespace(cls, ns: Namespace) -> OCROptions: + """Convert argparse.Namespace to OCROptions.""" + # Extract known fields + known_fields = {} + extra_attrs = {} + + for key, value in vars(ns).items(): + if key in cls.__fields__: + known_fields[key] = value + else: + extra_attrs[key] = value + + instance = cls(**known_fields) + instance._extra_attrs = extra_attrs + return instance + + def to_namespace(self) -> Namespace: + """Convert back to argparse.Namespace for compatibility.""" + ns = Namespace() + + # Add pydantic fields + for field_name in self.__fields__: + field_value = getattr(self, field_name) + setattr(ns, field_name, field_value) + + # Add extra attributes + for key, value in self._extra_attrs.items(): + setattr(ns, key, value) + + return ns + + @validator('languages') + def validate_languages(cls, v): + """Ensure languages list is not empty.""" + if not v: + return [DEFAULT_LANGUAGE] + return v + + @validator('output_type') + def validate_output_type(cls, v): + """Validate output type is one of the allowed values.""" + valid_types = {'pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3', 'none'} + if v not in valid_types: + raise ValueError(f"output_type must be one of {valid_types}") + return v + + @validator('pdf_renderer') + def validate_pdf_renderer(cls, v): + """Validate PDF renderer is one of the allowed values.""" + valid_renderers = {'auto', 'hocr', 'sandwich', 'hocrdebug'} + if v not in valid_renderers: + raise ValueError(f"pdf_renderer must be one of {valid_renderers}") + return v + + @validator('clean_final') + def validate_clean_final(cls, v, values): + """If clean_final is True, also set clean to True.""" + if v and 'clean' in values: + values['clean'] = True + return v + + @validator('jobs') + def validate_jobs(cls, v): + """Validate jobs is a reasonable number.""" + if v is not None and (v < 0 or v > 256): + raise ValueError("jobs must be between 0 and 256") + return v + + @validator('verbose') + def validate_verbose(cls, v): + """Validate verbose level.""" + if v < 0 or v > 2: + raise ValueError("verbose must be between 0 and 2") + return v + + @validator('oversample') + def validate_oversample(cls, v): + """Validate oversample DPI.""" + if v < 0 or v > 5000: + raise ValueError("oversample must be between 0 and 5000") + return v + + @validator('max_image_mpixels') + def validate_max_image_mpixels(cls, v): + """Validate max image megapixels.""" + if v < 0: + raise ValueError("max_image_mpixels must be non-negative") + return v + + @validator('rotate_pages_threshold') + def validate_rotate_pages_threshold(cls, v): + """Validate rotate pages threshold.""" + if v < 0 or v > 1000: + raise ValueError("rotate_pages_threshold must be between 0 and 1000") + return v + + class Config: + extra = "forbid" # Force use of _extra_attrs for unknown fields + arbitrary_types_allowed = True # Allow BinaryIO, Path, etc. + validate_assignment = True # Validate on attribute assignment diff --git a/src/ocrmypdf/_validation.py b/src/ocrmypdf/_validation.py index 76f9b7a4..fa16f516 100644 --- a/src/ocrmypdf/_validation.py +++ b/src/ocrmypdf/_validation.py @@ -15,6 +15,7 @@ from argparse import Namespace from collections.abc import Sequence from pathlib import Path from shutil import copyfileobj +from typing import Union import pikepdf import PIL @@ -22,6 +23,7 @@ from pluggy import PluginManager from ocrmypdf._defaults import DEFAULT_LANGUAGE, DEFAULT_ROTATE_PAGES_THRESHOLD from ocrmypdf._exec import unpaper +from ocrmypdf._options import OCROptions from ocrmypdf.exceptions import ( BadArgsError, InputFileError, @@ -51,7 +53,7 @@ def check_platform() -> None: def check_options_languages( - options: Namespace, ocr_engine_languages: list[str] + options: Union[Namespace, OCROptions], ocr_engine_languages: list[str] ) -> None: if not options.languages: options.languages = [DEFAULT_LANGUAGE] @@ -81,7 +83,7 @@ def check_options_languages( raise MissingDependencyError(msg) -def check_options_output(options: Namespace) -> None: +def check_options_output(options: Union[Namespace, OCROptions]) -> None: if options.output_type == 'none' and options.output_file not in (os.devnull, '-'): raise BadArgsError( "Since you specified `--output-type none`, the output file " @@ -90,7 +92,7 @@ def check_options_output(options: Namespace) -> None: ) -def set_lossless_reconstruction(options: Namespace) -> None: +def set_lossless_reconstruction(options: Union[Namespace, OCROptions]) -> None: lossless_reconstruction = False if not any( ( @@ -110,7 +112,7 @@ def set_lossless_reconstruction(options: Namespace) -> None: ) -def check_options_sidecar(options: Namespace) -> None: +def check_options_sidecar(options: Union[Namespace, OCROptions]) -> None: if options.sidecar == '\0': if options.output_file == '-': raise BadArgsError("--sidecar filename needed when output file is stdout.") @@ -125,7 +127,7 @@ def check_options_sidecar(options: Namespace) -> None: ) -def check_options_preprocessing(options: Namespace) -> None: +def check_options_preprocessing(options: Union[Namespace, OCROptions]) -> None: if options.clean_final: options.clean = True if options.unpaper_args and not options.clean: @@ -191,7 +193,7 @@ def _pages_from_ranges(ranges: str) -> set[int]: return set(pages) -def check_options_ocr_behavior(options: Namespace) -> None: +def check_options_ocr_behavior(options: Union[Namespace, OCROptions]) -> None: exclusive_options = sum( (1 if opt else 0) for opt in (options.force_ocr, options.skip_text, options.redo_ocr) @@ -202,7 +204,7 @@ def check_options_ocr_behavior(options: Namespace) -> None: options.pages = _pages_from_ranges(options.pages) -def check_options_metadata(options: Namespace) -> None: +def check_options_metadata(options: Union[Namespace, OCROptions]) -> None: docinfo = [options.title, options.author, options.keywords, options.subject] for s in (m for m in docinfo if m): for char in s: @@ -215,13 +217,13 @@ def check_options_metadata(options: Namespace) -> None: ) -def check_options_pillow(options: Namespace) -> None: +def check_options_pillow(options: Union[Namespace, OCROptions]) -> None: PIL.Image.MAX_IMAGE_PIXELS = int(options.max_image_mpixels * 1_000_000) if PIL.Image.MAX_IMAGE_PIXELS == 0: PIL.Image.MAX_IMAGE_PIXELS = None # type: ignore -def _check_plugin_invariant_options(options: Namespace) -> None: +def _check_plugin_invariant_options(options: Union[Namespace, OCROptions]) -> None: check_platform() check_options_metadata(options) check_options_output(options) @@ -232,18 +234,23 @@ def _check_plugin_invariant_options(options: Namespace) -> None: check_options_pillow(options) -def _check_plugin_options(options: Namespace, plugin_manager: PluginManager) -> None: - plugin_manager.hook.check_options(options=options) - ocr_engine_languages = plugin_manager.hook.get_ocr_engine().languages(options) +def _check_plugin_options(options: Union[Namespace, OCROptions], plugin_manager: PluginManager) -> None: + # Convert to Namespace for plugin compatibility during transition + if isinstance(options, OCROptions): + legacy_options = options.to_namespace() + else: + legacy_options = options + plugin_manager.hook.check_options(options=legacy_options) + ocr_engine_languages = plugin_manager.hook.get_ocr_engine().languages(legacy_options) check_options_languages(options, ocr_engine_languages) -def check_options(options: Namespace, plugin_manager: PluginManager) -> None: +def check_options(options: Union[Namespace, OCROptions], plugin_manager: PluginManager) -> None: _check_plugin_invariant_options(options) _check_plugin_options(options, plugin_manager) -def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]: +def create_input_file(options: Union[Namespace, OCROptions], work_folder: Path) -> tuple[Path, str]: if options.input_file == '-': # stdin log.info('reading file from standard input') @@ -288,7 +295,7 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str] raise InputFileError(msg) from e -def check_requested_output_file(options: Namespace) -> None: +def check_requested_output_file(options: Union[Namespace, OCROptions]) -> None: if options.output_file == '-': if sys.stdout.isatty(): raise BadArgsError( @@ -306,7 +313,7 @@ def check_requested_output_file(options: Namespace) -> None: def report_output_file_size( - options: Namespace, + options: Union[Namespace, OCROptions], input_file: Path, output_file: Path, optimize_messages: Sequence[str] | None = None, diff --git a/src/ocrmypdf/api.py b/src/ocrmypdf/api.py index 0d737aef..87037073 100644 --- a/src/ocrmypdf/api.py +++ b/src/ocrmypdf/api.py @@ -54,6 +54,7 @@ from warnings import warn import pluggy from ocrmypdf._logging import PageNumberFilter +from ocrmypdf._options import OCROptions from ocrmypdf._pipelines.hocr_to_ocr_pdf import run_hocr_to_ocr_pdf_pipeline from ocrmypdf._pipelines.ocr import run_pipeline, run_pipeline_cli from ocrmypdf._pipelines.pdf_to_hocr import run_hocr_pipeline @@ -216,7 +217,7 @@ def _kwargs_to_cmdline( def create_options( *, input_file: PathOrIO, output_file: PathOrIO, parser: ArgumentParser, **kwargs -) -> Namespace: +) -> OCROptions: """Construct an options object from the input/output files and keyword arguments. Args: @@ -226,7 +227,7 @@ def create_options( **kwargs: Keyword arguments. Returns: - argparse.Namespace: A Namespace object containing the parsed arguments. + OCROptions: An options object containing the parsed arguments. Raises: TypeError: If the type of a keyword argument is not supported. @@ -248,17 +249,19 @@ def create_options( cmdline.append('stream://sidecar') parser.enable_api_mode() - options = parser.parse_args(cmdline) + namespace_options = parser.parse_args(cmdline) for keyword, val in deferred.items(): - setattr(options, keyword, val) + setattr(namespace_options, keyword, val) - if options.input_file == 'stream://input_file': - options.input_file = input_file - if options.output_file == 'stream://output_file': - options.output_file = output_file - if options.sidecar == 'stream://sidecar': - options.sidecar = kwargs['sidecar'] + if namespace_options.input_file == 'stream://input_file': + namespace_options.input_file = input_file + if namespace_options.output_file == 'stream://output_file': + namespace_options.output_file = output_file + if namespace_options.sidecar == 'stream://sidecar': + namespace_options.sidecar = kwargs['sidecar'] + # Convert to OCROptions + options = OCROptions.from_namespace(namespace_options) return options