Add --tagged-pdf-mode option to control Tagged PDF handling

Allow users to bypass the TaggedPDFError when processing Tagged PDFs
by setting --tagged-pdf-mode=ignore. This is useful when users know
they want to OCR a Tagged PDF despite the warning.

- 'default': Error if --mode is default, otherwise warn (current behavior)
- 'ignore': Always warn but continue processing (never error)
This commit is contained in:
James R. Barlow
2026-01-30 16:15:43 -08:00
parent 0a980fb11b
commit e036a902ae
8 changed files with 89 additions and 11 deletions
+17
View File
@@ -32,6 +32,7 @@ __ocrmypdf_arguments()
--skip-text (skip OCR on any pages that already contain text)
--redo-ocr (redo OCR on any pages that seem to have OCR already)
--invalidate-digital-signatures (remove digital signatures from PDF)
--tagged-pdf-mode (control behavior for Tagged PDFs)
--skip-big (skip OCR on pages larger than this many MPixels)
--optimize (select optimization level)
--jpeg-quality (JPEG quality [0..100])
@@ -232,6 +233,18 @@ redo (re-OCR pages, replacing old invisible text)"
fi
}
__ocrmypdf_tagged-pdf-mode()
{
local choices="default (error if --mode is default, otherwise warn)
ignore (always warn but continue processing)"
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
# Remove description if only one completion exists
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
COMPREPLY=( ${COMPREPLY[0]%% *} )
fi
}
__ocrmypdf_ocr-engine()
{
local choices="auto (select best available engine)
@@ -293,6 +306,10 @@ __ocrmypdf_check_previous()
__ocrmypdf_mode
return 0
;;
--tagged-pdf-mode)
__ocrmypdf_tagged-pdf-mode
return 0
;;
--ocr-engine)
__ocrmypdf_ocr-engine
return 0
+6
View File
@@ -26,6 +26,12 @@ complete -c ocrmypdf -s s -l skip-text -d "skip OCR on any pages that already co
complete -c ocrmypdf -l redo-ocr -d "redo OCR on any pages that seem to have OCR already"
complete -c ocrmypdf -l invalidate-digital-signatures -d "invalidate digital signatures and allow OCR to proceed"
function __fish_ocrmypdf_tagged_pdf_mode
echo -e "default\t"(_ "error if --mode is default, otherwise warn")
echo -e "ignore\t"(_ "always warn but continue processing")
end
complete -c ocrmypdf -x -l tagged-pdf-mode -a '(__fish_ocrmypdf_tagged_pdf_mode)' -d "control behavior for Tagged PDFs"
complete -c ocrmypdf -s k -l keep-temporary-files -d "keep temporary files (debug)"
function __fish_ocrmypdf_languages
+2 -1
View File
@@ -11,7 +11,7 @@ from ocrmypdf import helpers, hocrtransform, pdfa, pdfinfo
from ocrmypdf._concurrent import Executor
from ocrmypdf._defaults import PROGRAM_NAME
from ocrmypdf._jobcontext import PageContext, PdfContext
from ocrmypdf._options import OcrOptions
from ocrmypdf._options import OcrOptions, TaggedPdfMode
from ocrmypdf._pipelines._common import (
configure_debug_logging,
)
@@ -78,6 +78,7 @@ __all__ = [
'PriorOcrFoundError',
'PROGRAM_NAME',
'SubprocessOutputError',
'TaggedPdfMode',
'TesseractConfigError',
'UnsupportedImageFormatError',
'Verbosity',
+15
View File
@@ -50,6 +50,20 @@ class ProcessingMode(StrEnum):
redo = 'redo'
class TaggedPdfMode(StrEnum):
"""Control behavior when encountering a Tagged PDF.
Tagged PDFs often indicate documents generated from office applications
that may not need OCR. This enum controls how OCRmyPDF handles them:
- ``default``: Error if ProcessingMode is default, otherwise warn
- ``ignore``: Always warn but continue processing (never error)
"""
default = 'default'
ignore = 'ignore'
def _pages_from_ranges(ranges: str) -> set[int]:
"""Convert page range string to set of page numbers."""
pages: list[int] = []
@@ -150,6 +164,7 @@ class OcrOptions(BaseModel):
skip_big: float | None = None
pages: str | set[int] | None = None # Can be string or set after validation
invalidate_digital_signatures: bool = False
tagged_pdf_mode: TaggedPdfMode = TaggedPdfMode.default
# Metadata
title: str | None = None
+12 -9
View File
@@ -28,7 +28,7 @@ from ocrmypdf._concurrent import Executor
from ocrmypdf._exec import unpaper
from ocrmypdf._jobcontext import PageContext, PdfContext
from ocrmypdf._metadata import repair_docinfo_nuls
from ocrmypdf._options import OcrOptions, ProcessingMode
from ocrmypdf._options import OcrOptions, ProcessingMode, TaggedPdfMode
from ocrmypdf.exceptions import (
DigitalSignatureError,
DpiError,
@@ -251,14 +251,17 @@ def validate_pdfinfo_options(context: PdfContext) -> None:
"will be 'flattened' and will no longer be fillable."
)
if pdfinfo.is_tagged:
if options.mode != ProcessingMode.default:
log.warning(
"This PDF is marked as a Tagged PDF. This often indicates "
"that the PDF was generated from an office document and does "
"not need OCR. PDF pages processed by OCRmyPDF may not be "
"tagged correctly."
)
else:
log.warning(
"This PDF is marked as a Tagged PDF. This often indicates "
"that the PDF was generated from an office document and does "
"not need OCR. PDF pages processed by OCRmyPDF may not be "
"tagged correctly."
)
if (
options.tagged_pdf_mode == TaggedPdfMode.default
and options.mode == ProcessingMode.default
):
log.info("Use --tagged-pdf-mode ignore to ignore Tagged PDFs.")
raise TaggedPDFError()
context.plugin_manager.validate(pdfinfo=pdfinfo, options=options)
+2
View File
@@ -408,6 +408,7 @@ def ocr(
fast_web_view: float | None = None,
continue_on_soft_render_error: bool | None = None,
invalidate_digital_signatures: bool | None = None,
tagged_pdf_mode: str | None = None,
plugins: Iterable[Path | str] | None = None,
plugin_manager: OcrmypdfPluginManager | None = None,
keep_temporary_files: bool | None = None,
@@ -469,6 +470,7 @@ def ocr( # noqa: D417
fast_web_view: float | None = None,
continue_on_soft_render_error: bool | None = None,
invalidate_digital_signatures: bool | None = None,
tagged_pdf_mode: str | None = None,
plugins: Iterable[Path | str] | None = None,
plugin_manager: OcrmypdfPluginManager | None = None,
keep_temporary_files: bool | None = None,
+9 -1
View File
@@ -12,7 +12,7 @@ from typing import Any, TypeVar
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
from ocrmypdf._defaults import PROGRAM_NAME as _PROGRAM_NAME
from ocrmypdf._options import OcrOptions, ProcessingMode
from ocrmypdf._options import OcrOptions, ProcessingMode, TaggedPdfMode
from ocrmypdf._plugin_manager import OcrmypdfPluginManager
from ocrmypdf._version import __version__ as _VERSION
@@ -360,6 +360,14 @@ Online documentation is located at:
"signature. This option allows OCR to proceed, but the digital signature "
"will be invalidated.",
)
ocrsettings.add_argument(
'--tagged-pdf-mode',
choices=[mode.value for mode in TaggedPdfMode],
default=TaggedPdfMode.default.value,
help="Control behavior when a Tagged PDF is encountered. "
"'default' errors if --mode is default, otherwise warns. "
"'ignore' always warns but continues processing.",
)
advanced = parser.add_argument_group(
"Advanced", "Advanced options to control OCRmyPDF"
+26
View File
@@ -22,3 +22,29 @@ def test_force_tagged_warns(resources, outpdf, caplog):
plugins=['tests/plugins/tesseract_noop.py'],
)
assert 'marked as a Tagged PDF' in caplog.text
def test_tagged_pdf_mode_ignore_with_skip_text(resources, outpdf, caplog):
"""Ignore tagged_pdf_mode should warn but not error."""
caplog.set_level('WARNING')
ocrmypdf.ocr(
resources / 'tagged.pdf',
outpdf,
tagged_pdf_mode='ignore',
skip_text=True, # Tagged PDF has text, so skip pages with text
plugins=['tests/plugins/tesseract_noop.py'],
)
assert 'marked as a Tagged PDF' in caplog.text
def test_tagged_pdf_mode_ignore_with_force(resources, outpdf, caplog):
"""Ignore tagged_pdf_mode with force mode should warn."""
caplog.set_level('WARNING')
ocrmypdf.ocr(
resources / 'tagged.pdf',
outpdf,
tagged_pdf_mode='ignore',
force_ocr=True,
plugins=['tests/plugins/tesseract_noop.py'],
)
assert 'marked as a Tagged PDF' in caplog.text