Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
aa6a32e7d1 | ||
|
|
ea99758747 | ||
|
|
4942751a1b | ||
|
|
be06e3184a | ||
|
|
39bf09f1eb | ||
|
|
aaffc46f73 | ||
|
|
0277b3b3ba | ||
|
|
0817542883 | ||
|
|
6f4744dd20 | ||
|
|
5d49f75c56 | ||
|
|
5a824ddd8c | ||
|
|
54bf03a454 | ||
|
|
009754d137 | ||
|
|
f0a3a74374 | ||
|
|
178d339c8e | ||
|
|
d3f8d01227 | ||
|
|
b60df59c62 | ||
|
|
640b3062b2 | ||
|
|
ef903db360 | ||
|
|
9cda02317b | ||
|
|
92a2fe880a | ||
|
|
089f46690a | ||
|
|
e45c40b063 | ||
|
|
bbac5307f2 | ||
|
|
6167783696 | ||
|
|
3d291e72c0 | ||
|
|
efebe9ca2e | ||
|
|
12ec97f732 | ||
|
|
3f1aceade2 | ||
|
|
212b28e602 | ||
|
|
dfdb32995e | ||
|
|
273826377e | ||
|
|
5569d4db07 | ||
|
|
8de7b05fb9 | ||
|
|
72ce05768e | ||
|
|
3dc68778fc | ||
|
|
1aec92b919 | ||
|
|
43d3448709 | ||
|
|
7512b1042a | ||
|
|
efe83e8c54 | ||
|
|
a13d27bfb5 | ||
|
|
ea7ad7d683 | ||
|
|
8b20bb3c5b | ||
|
|
320876a6d1 | ||
|
|
dfbb4c9275 | ||
|
|
d4f5c2d160 | ||
|
|
263d6034be | ||
|
|
de403f6d5e | ||
|
|
86b6f2c907 | ||
|
|
e6fab76918 | ||
|
|
334918d0f7 | ||
|
|
d6329489ce | ||
|
|
e6d240ee93 | ||
|
|
ff45e54c07 | ||
|
|
e0ee0882ef | ||
|
|
3d17419a6c | ||
|
|
476ec12383 | ||
|
|
e99177ada7 | ||
|
|
e95ec9c497 | ||
|
|
82f30bfbec | ||
|
|
d1437e6bbc | ||
|
|
c669d30642 | ||
|
|
3613b30ca8 | ||
|
|
0d4c3bcdcf | ||
|
|
8a8d515933 | ||
|
|
11de13ecfe | ||
|
|
58642d8411 | ||
|
|
7e42d3c771 | ||
|
|
5cb5d7a682 | ||
|
|
37e71dece6 | ||
|
|
df84945773 | ||
|
|
b5a6a9f9f1 | ||
|
|
ed36aefe48 | ||
|
|
32013f4294 | ||
|
|
8f2bcc2c64 | ||
|
|
015b53ae30 | ||
|
|
164cf2dc8a | ||
|
|
98d6d02704 | ||
|
|
5efb98931d | ||
|
|
2f4e47213f | ||
|
|
94c8123bd7 | ||
|
|
0db130e1c3 | ||
|
|
91b6a818f5 | ||
|
|
6bc9499e68 | ||
|
|
09f2d6c386 | ||
|
|
87f918f58c | ||
|
|
80e77fb021 | ||
|
|
fa9c5b3fae | ||
|
|
3d17a60a54 | ||
|
|
c33f073d4f | ||
|
|
5d7b5742e4 | ||
|
|
c391b2b7d0 | ||
|
|
0250929150 | ||
|
|
9748208e68 | ||
|
|
e4b0c04be4 | ||
|
|
efb83ad64f | ||
|
|
08e40f96e8 | ||
|
|
3f6feb1dcc | ||
|
|
ab6553f4ff | ||
|
|
cedca9fa1f | ||
|
|
3f40118022 | ||
|
|
b18b1da6d0 | ||
|
|
14fb9f56e8 | ||
|
|
8709cf506b | ||
|
|
9a92eb40df | ||
|
|
0a59c210f9 | ||
|
|
0b370fdd15 | ||
|
|
1c16dd26f7 | ||
|
|
c355d927ba | ||
|
|
c993857752 | ||
|
|
84f5fe9ee0 |
+29
-7
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM ubuntu:25.04 AS base
|
||||
FROM ubuntu:26.04 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -40,7 +40,7 @@ RUN \
|
||||
WORKDIR /app
|
||||
|
||||
# Copy uv from ghcr
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9.8 /uv /uvx /bin/
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.21 /uv /uvx /bin/
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
@@ -60,10 +60,8 @@ RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
|
||||
FROM base
|
||||
|
||||
RUN apt-get update && apt-get install -y software-properties-common
|
||||
|
||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr5
|
||||
|
||||
# Tesseract 5 ships in the Ubuntu archive as of 24.04, so no third-party PPA is
|
||||
# needed. (Previously this used ppa:alex-p/tesseract-ocr5.)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ghostscript \
|
||||
fonts-droid-fallback \
|
||||
@@ -81,6 +79,18 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
unpaper \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Create a non-root user to run the application (defense in depth). The build
|
||||
# stages above need root to install packages, but the entrypoint should not.
|
||||
# A fixed uid/gid of 1000 keeps `--user`/`--userns keep-id` mappings predictable
|
||||
# and matches the --chown below. See docs/docker.md for the volume/permissions
|
||||
# implications under rootless vs rootful Docker.
|
||||
# The Ubuntu base ships a default "ubuntu" user at uid/gid 1000; remove it so
|
||||
# "app" can claim that uid for parity with the Alpine image.
|
||||
RUN userdel -r ubuntu 2>/dev/null; groupdel ubuntu 2>/dev/null; \
|
||||
groupadd -g 1000 app \
|
||||
&& useradd -u 1000 -g app -m -d /home/app app
|
||||
ENV HOME=/home/app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
||||
@@ -90,9 +100,21 @@ COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
ln -s /app/misc/watcher.py /app/watcher.py && \
|
||||
chown app:app /app
|
||||
|
||||
# Default working directory for bind-mounted data, so relative input/output
|
||||
# paths work without passing --workdir (e.g. `-v "$PWD:/data" in.pdf out.pdf`).
|
||||
# The webservice/watcher are run by absolute path (/app/*.py), unaffected by this.
|
||||
RUN mkdir -p /data && chown app:app /data
|
||||
WORKDIR /data
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
# Drop privileges: run the entrypoint (ocrmypdf, or the webservice/watcher when
|
||||
# overridden) as the unprivileged app user. Override with `--user root` if you
|
||||
# need root inside a running container (e.g. to apt install extra packages).
|
||||
USER app
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM alpine:3.23 AS base
|
||||
FROM alpine:3.24 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -22,7 +22,7 @@ RUN apk add --no-cache \
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9.8 /uv /uvx /bin/
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.21 /uv /uvx /bin/
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
@@ -62,14 +62,35 @@ RUN apk add --no-cache \
|
||||
unpaper \
|
||||
&& rm -rf /var/cache/apk/*
|
||||
|
||||
# Create a non-root user to run the application (defense in depth). The build
|
||||
# stages above need root to install packages, but the entrypoint should not.
|
||||
# A fixed uid/gid of 1000 keeps `--user`/`--userns keep-id` mappings predictable
|
||||
# and matches the --chown below. See docs/docker.md for the volume/permissions
|
||||
# implications under rootless vs rootful Docker.
|
||||
RUN addgroup -g 1000 app \
|
||||
&& adduser -u 1000 -G app -D -h /home/app app
|
||||
ENV HOME=/home/app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
ln -s /app/misc/watcher.py /app/watcher.py && \
|
||||
chown app:app /app
|
||||
|
||||
# Default working directory for bind-mounted data, so relative input/output
|
||||
# paths work without passing --workdir (e.g. `-v "$PWD:/data" in.pdf out.pdf`).
|
||||
# The webservice/watcher are run by absolute path (/app/*.py), unaffected by this.
|
||||
RUN mkdir -p /data && chown app:app /data
|
||||
WORKDIR /data
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
# Drop privileges: run the entrypoint (ocrmypdf, or the webservice/watcher when
|
||||
# overridden) as the unprivileged app user. Override with `--user root` if you
|
||||
# need root inside a running container (e.g. to apk add extra packages).
|
||||
USER app
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
|
||||
+44
-13
@@ -14,8 +14,29 @@ on:
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
lint:
|
||||
name: Lint (prek)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Run prek
|
||||
run: |
|
||||
uv run prek run --all-files
|
||||
|
||||
test_linux:
|
||||
name: Test ${{ matrix.os }} with Python ${{ matrix.python }}
|
||||
needs: lint
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -31,7 +52,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -39,7 +60,7 @@ jobs:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
@@ -87,7 +108,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v6
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -96,6 +117,7 @@ jobs:
|
||||
|
||||
test_macos:
|
||||
name: Test macOS
|
||||
needs: lint
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -107,7 +129,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install Homebrew deps
|
||||
continue-on-error: true
|
||||
@@ -129,7 +151,7 @@ jobs:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
@@ -149,7 +171,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v6
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -158,6 +180,7 @@ jobs:
|
||||
|
||||
test_windows:
|
||||
name: Test Windows
|
||||
needs: lint
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -169,7 +192,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -177,7 +200,7 @@ jobs:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
@@ -196,7 +219,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v6
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -207,7 +230,7 @@ jobs:
|
||||
name: Build sdist and wheels
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -233,7 +256,7 @@ jobs:
|
||||
permissions:
|
||||
contents: write
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/download-artifact@v8
|
||||
with:
|
||||
@@ -252,6 +275,14 @@ jobs:
|
||||
run: |
|
||||
TAG="v${{ steps.version.outputs.version }}"
|
||||
|
||||
# If release.yml already published this version, _version.py may
|
||||
# still reflect it until the next version bump commit. Don't
|
||||
# re-draft an already-published release on later pushes to main.
|
||||
if [[ "$(gh release view "$TAG" --json isDraft --jq .isDraft 2>/dev/null)" == "false" ]]; then
|
||||
echo "Release $TAG is already published; skipping."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Delete existing draft release if it exists (ignore errors)
|
||||
gh release delete "$TAG" --yes 2>/dev/null || true
|
||||
|
||||
@@ -281,7 +312,7 @@ jobs:
|
||||
- name: Set image name
|
||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
||||
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4
|
||||
@@ -327,7 +358,7 @@ jobs:
|
||||
- name: Set image name
|
||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf-alpine" >> $GITHUB_ENV
|
||||
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4
|
||||
|
||||
@@ -19,7 +19,7 @@ jobs:
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Download artifacts from draft release
|
||||
env:
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
|
||||
# PyPI doesn't support sigstore publishing, so generate after publishing to PyPI
|
||||
- name: Sign the dists with Sigstore
|
||||
uses: sigstore/gh-action-sigstore-python@v3.3.0
|
||||
uses: sigstore/gh-action-sigstore-python@v3.4.0
|
||||
with:
|
||||
inputs: |
|
||||
./dist/*.tar.gz
|
||||
|
||||
@@ -1,27 +0,0 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v4.4.0
|
||||
hooks:
|
||||
- id: check-case-conflict
|
||||
- id: check-merge-conflict
|
||||
- id: check-toml
|
||||
- id: check-yaml
|
||||
- id: debug-statements
|
||||
- repo: https://github.com/astral-sh/ruff-pre-commit
|
||||
rev: "v0.14.11"
|
||||
hooks:
|
||||
- id: ruff-check
|
||||
args: [--fix]
|
||||
- id: ruff-format
|
||||
- repo: https://github.com/pre-commit/mirrors-mypy
|
||||
rev: v1.2.0
|
||||
hooks:
|
||||
- id: mypy
|
||||
additional_dependencies:
|
||||
- types-toml
|
||||
- types-setuptools
|
||||
- types-requests
|
||||
- types-Pillow
|
||||
+90
-76
@@ -1,8 +1,3 @@
|
||||
# OCRmyPDF 中文说明
|
||||
|
||||
我将为您创建一个中文版的 README.md 文件。以下是基于原始英文版本的中文翻译:
|
||||
|
||||
```markdown:/Users/huapai/PycharmProjects/OCRmyPDF/README.md
|
||||
<!-- SPDX-FileCopyrightText: 2014 Julien Pfefferkorn -->
|
||||
<!-- SPDX-FileCopyrightText: 2015 James R. Barlow -->
|
||||
<!-- SPDX-License-Identifier: CC-BY-SA-4.0 -->
|
||||
@@ -16,98 +11,109 @@
|
||||
[docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD"
|
||||
[pyversions]: https://img.shields.io/pypi/pyversions/ocrmypdf "支持的 Python 版本"
|
||||
|
||||
OCRmyPDF 为扫描的 PDF 文件添加 OCR 文本层,使其可以被搜索或复制粘贴。
|
||||
OCRmyPDF 会为扫描版 PDF 文件添加 OCR 文本层,使其可以搜索或复制粘贴。
|
||||
|
||||
```bash
|
||||
ocrmypdf # 这是一个可脚本化的命令行程序
|
||||
-l eng+fra # 支持多种语言
|
||||
--rotate-pages # 可以修正旋转错误的页面
|
||||
--deskew # 可以校正倾斜的 PDF!
|
||||
--title "My PDF" # 可以更改输出元数据
|
||||
--jobs 4 # 默认使用多核心处理
|
||||
--output-type pdfa # 默认生成 PDF/A 格式
|
||||
ocrmypdf # 它是一个可脚本化的命令行程序
|
||||
-l eng+fra # 它支持多种语言
|
||||
--rotate-pages # 它可以修正旋转方向错误的页面
|
||||
--deskew # 它可以校正歪斜的 PDF!
|
||||
--title "My PDF" # 它可以更改输出元数据
|
||||
--jobs 4 # 它默认使用多个 CPU 核心
|
||||
--output-type pdfa # 它默认生成 PDF/A
|
||||
input_scanned.pdf # 接受 PDF 输入(或图像)
|
||||
output_searchable.pdf # 生成经过验证的 PDF 输出
|
||||
```
|
||||
|
||||
[查看发布说明了解最新变更的详情](https://ocrmypdf.readthedocs.io/en/latest/release_notes.html)。
|
||||
[查看发布说明,了解最新变更详情](https://ocrmypdf.readthedocs.io/en/latest/release_notes.html)。
|
||||
|
||||
## 主要特点
|
||||
## 主要功能
|
||||
|
||||
- 从普通 PDF 生成可搜索的 [PDF/A](https://en.wikipedia.org/?title=PDF/A) 文件
|
||||
- 准确地将 OCR 文本放置在图像下方,便于复制/粘贴
|
||||
- 将 OCR 文本准确放置在图像下方,便于复制/粘贴
|
||||
- 保持原始嵌入图像的精确分辨率
|
||||
- 在可能的情况下,以"无损"操作方式插入 OCR 信息,不破坏任何其他内容
|
||||
- 在可能时,以“无损”操作插入 OCR 信息,不干扰任何其他内容
|
||||
- 优化 PDF 图像,通常生成比输入文件更小的文件
|
||||
- 如果需要,在执行 OCR 前对图像进行校正和/或清理
|
||||
- 按需在执行 OCR 前校正和/或清理图像
|
||||
- 验证输入和输出文件
|
||||
- 在所有可用的 CPU 核心上分配工作
|
||||
- 在所有可用 CPU 核心间分配工作
|
||||
- 使用 [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) 引擎识别超过 [100 种语言](https://github.com/tesseract-ocr/tessdata)
|
||||
- 保护您的私人数据安全
|
||||
- 适当扩展以处理包含数千页的文件
|
||||
- 在数百万 PDF 上经过实战测试
|
||||
- 保护你的私有数据。
|
||||
- 可以妥善扩展,处理包含数千页的文件。
|
||||
- 已在数百万份 PDF 上经过实战检验。
|
||||
|
||||
<img src="misc/screencast/demo.svg" alt="终端会话中的 OCRmyPDF 演示">
|
||||
<img src="misc/screencast/demo.svg" alt="OCRmyPDF 在终端会话中的演示">
|
||||
|
||||
详情请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/)。
|
||||
|
||||
## 开发动机
|
||||
## 动机
|
||||
|
||||
我在网上搜索免费的命令行工具来对 PDF 文件进行 OCR:我找到了很多,但没有一个真正令人满意:
|
||||
我曾在网上寻找一款免费的命令行工具来对 PDF 文件执行 OCR:我找到了很多,但没有一个真正令人满意:
|
||||
|
||||
- 要么它们生成的 PDF 文件中文本位置错误(使复制/粘贴变得不可能)
|
||||
- 要么它们不处理重音和多语言字符
|
||||
- 要么它们改变了嵌入图像的分辨率
|
||||
- 要么它们生成了体积巨大的 PDF 文件
|
||||
- 要么它们在尝试 OCR 时崩溃
|
||||
- 要么它们不生成有效的 PDF 文件
|
||||
- 最重要的是,它们都不生成 PDF/A 文件(专为长期存储设计的格式)
|
||||
- 要么生成的 PDF 文件中文本位于图像下方的错误位置(导致无法复制/粘贴)
|
||||
- 要么无法处理重音字符和多语言字符
|
||||
- 要么会改变嵌入图像的分辨率
|
||||
- 要么生成的 PDF 文件大得离谱
|
||||
- 要么在尝试 OCR 时崩溃
|
||||
- 要么无法生成有效的 PDF 文件
|
||||
- 除此之外,它们都不能生成 PDF/A 文件(专为长期存储设计的格式)
|
||||
|
||||
...所以我决定开发自己的工具。
|
||||
……所以我决定开发自己的工具。
|
||||
|
||||
## 安装
|
||||
|
||||
支持 Linux、Windows、macOS 和 FreeBSD。Docker 镜像也可用,同时支持 x64 和 ARM。
|
||||
支持 Linux、Windows、macOS 和 FreeBSD。也提供 Docker 镜像,同时支持 x64 和 ARM。
|
||||
|
||||
| 操作系统 | 安装命令 |
|
||||
| --------------------------- | ----------------------------- |
|
||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||
| 操作系统 | 安装命令 |
|
||||
| ----------------------------- | ------------------------------ |
|
||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| OpenBSD | ``pkg_add ocrmypdf`` |
|
||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||
|
||||
对于其他用户,[请参阅我们的文档](https://ocrmypdf.readthedocs.io/en/latest/installation.html)了解安装步骤。
|
||||
其他用户请[参阅我们的文档](https://ocrmypdf.readthedocs.io/en/latest/installation.html)了解安装步骤。
|
||||
|
||||
## 语言
|
||||
|
||||
OCRmyPDF 使用 Tesseract 进行 OCR,并依赖其语言包。对于 Linux 用户,您通常可以找到提供语言包的软件包:
|
||||
OCRmyPDF 使用 Tesseract 执行 OCR,并依赖其语言包。对于 Linux 用户,通常可以找到提供语言包的软件包:
|
||||
|
||||
```bash
|
||||
# 显示所有 Tesseract 语言包的列表
|
||||
apt-cache search tesseract-ocr
|
||||
|
||||
# Debian/Ubuntu 用户
|
||||
apt-get install tesseract-ocr-chi-sim # 示例:安装中文简体语言包
|
||||
apt-cache search tesseract-ocr # 显示所有 Tesseract 语言包列表
|
||||
apt-get install tesseract-ocr-chi-sim # 示例:安装简体中文语言包
|
||||
|
||||
|
||||
# Arch Linux 用户
|
||||
pacman -S tesseract-data-eng tesseract-data-deu # 示例:安装英语和德语语言包
|
||||
|
||||
# OpenBSD 用户
|
||||
pkg_info -aQ tesseract # 显示所有 Tesseract 语言包列表
|
||||
pkg_add tesseract-cym # 示例:安装威尔士语语言包
|
||||
|
||||
# brew macOS 用户
|
||||
brew install tesseract-lang
|
||||
|
||||
# Fedora 用户
|
||||
dnf search tesseract-langpack # 显示所有 Tesseract 语言包列表
|
||||
dnf install tesseract-langpack-ita # 示例:安装意大利语语言包
|
||||
|
||||
|
||||
```
|
||||
|
||||
然后,您可以向 OCRmyPDF 传递 `-l LANG` 参数,提示它应该搜索哪些语言。可以请求多种语言。
|
||||
随后可以向 OCRmyPDF 传递 `-l LANG` 参数,提示它应搜索哪些语言。可以同时请求多种语言。
|
||||
|
||||
OCRmyPDF 支持 Tesseract 4.1.1+。它会自动使用在 `PATH` 环境变量中首先找到的版本。在 Windows 上,如果 `PATH` 不提供 Tesseract 二进制文件,我们会根据 Windows 注册表使用已安装的最高版本号。
|
||||
OCRmyPDF 支持 Tesseract 4.1.1+。它会自动使用 `PATH` 环境变量中首先找到的版本。在 Windows 上,如果 `PATH` 中没有 Tesseract 二进制文件,我们会根据 Windows 注册表使用已安装的最高版本号。
|
||||
|
||||
## 文档和支持
|
||||
|
||||
安装 OCRmyPDF 后,可以通过以下方式访问内置帮助,解释命令语法和选项:
|
||||
安装 OCRmyPDF 后,可以通过以下命令访问内置帮助,了解命令语法和选项:
|
||||
|
||||
```bash
|
||||
ocrmypdf --help
|
||||
@@ -115,13 +121,13 @@ ocrmypdf --help
|
||||
|
||||
我们的[文档托管在 Read the Docs 上](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
|
||||
请在我们的 [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) 页面上报告问题,并遵循问题模板以获得快速响应。
|
||||
请在我们的 [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) 页面报告问题,并遵循 issue 模板以便快速获得响应。
|
||||
|
||||
## 功能演示
|
||||
|
||||
```bash
|
||||
# 添加 OCR 层并转换为 PDF/A
|
||||
ocrmypdf input.pdf output.pdf
|
||||
# 添加 OCR 层并要求输出 PDF/A
|
||||
ocrmypdf --output-type pdfa input.pdf output.pdf
|
||||
|
||||
# 将图像转换为单页 PDF
|
||||
ocrmypdf input.jpg output.pdf
|
||||
@@ -129,45 +135,53 @@ ocrmypdf input.jpg output.pdf
|
||||
# 就地为文件添加 OCR(仅在成功时修改文件)
|
||||
ocrmypdf myfile.pdf myfile.pdf
|
||||
|
||||
# 使用非英语语言进行 OCR(查找您语言的 ISO 639-3 代码)
|
||||
# 使用非英语语言执行 OCR(请查找对应语言的 ISO 639-3 代码)
|
||||
ocrmypdf -l fra LeParisien.pdf LeParisien.pdf
|
||||
|
||||
# OCR 多语言文档
|
||||
ocrmypdf -l eng+fra Bilingual-English-French.pdf Bilingual-English-French.pdf
|
||||
|
||||
# 校正(矫正倾斜的页面)
|
||||
# 校正歪斜页面
|
||||
ocrmypdf --deskew input.pdf output.pdf
|
||||
```
|
||||
|
||||
更多功能,请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
更多功能请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
|
||||
## 要求
|
||||
|
||||
除了所需的 Python 版本外,OCRmyPDF 还需要外部程序安装 Ghostscript 和 Tesseract OCR。OCRmyPDF 是纯 Python 编写的,几乎可以在所有平台上运行:Linux、macOS、Windows 和 FreeBSD。
|
||||
除所需的 Python 版本外,OCRmyPDF 还需要安装 Ghostscript 和 Tesseract OCR 这两个外部程序。OCRmyPDF 是纯 Python 项目,几乎可以在所有平台上运行:Linux、macOS、Windows 和 FreeBSD。
|
||||
|
||||
## 媒体报道
|
||||
## 插件
|
||||
|
||||
- [使用 OCRmyPDF 实现无纸化](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a)
|
||||
- [将扫描文档转换为带有编辑的压缩可搜索 PDF](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c)
|
||||
- [c't 1-2014, 第 59 页](https://heise.de/-2279695):在德国领先的 IT 杂志 c't 中详细介绍 OCRmyPDF v1.0
|
||||
- [heise Open Source, 09/2014: 使用 OCRmyPDF 进行文本识别](https://heise.de/-2356670)
|
||||
- [heise 使用 OCRmyPDF 创建可搜索的 PDF 文档](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [优秀实用工具:OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser 使用 OCRmyPDF 和 Scanbd 自动化文本识别](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [Y Combinator 讨论](https://news.ycombinator.com/item?id=32028752)
|
||||
OCRmyPDF 提供插件接口,允许扩展或替换其能力。以下是我们知道的一些插件:
|
||||
|
||||
## 商业咨询
|
||||
- [OCRmyPDF-AppleOCR](https://github.com/mkyt/ocrmypdf-AppleOCR):用 Apple Vision Framework 替换标准 Tesseract OCR 引擎。需要 macOS。
|
||||
- [OCRmyPDF-EasyOCR](https://github.com/ocrmypdf/OCRmyPDF-EasyOCR):用 EasyOCR 替换标准 Tesseract OCR 引擎;EasyOCR 是基于 PyTorch 的较新 OCR 引擎。强烈建议使用 GPU。
|
||||
- [OCRmyPDF-PaddleOCR](https://github.com/clefru/ocrmypdf-paddleocr):用 PaddleOCR 替换标准 Tesseract OCR 引擎;PaddleOCR 是功能强大的 GPU 加速 OCR 引擎。
|
||||
|
||||
如果没有公司和用户选择为功能开发和咨询提供支持,OCRmyPDF 就不会成为今天的软件。我们很乐意讨论所有咨询,无论是扩展现有功能集,还是将 OCRmyPDF 集成到更大的系统中。
|
||||
[paperless-ngx](https://docs.paperless-ngx.com/) 将 OCRmyPDF 集成到可搜索的文档管理系统中。
|
||||
|
||||
## 新闻与媒体
|
||||
|
||||
- [Going paperless with OCRmyPDF](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a)
|
||||
- [Converting a scanned document into a compressed searchable PDF with redactions](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c)
|
||||
- [c't 1-2014,第 59 页](https://heise.de/-2279695):德国领先 IT 杂志 c't 对 OCRmyPDF v1.0 的详细介绍
|
||||
- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670)
|
||||
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [Y Combinator discussion](https://news.ycombinator.com/item?id=32028752)
|
||||
|
||||
## 商务咨询
|
||||
|
||||
如果没有公司和用户选择支持功能开发与咨询服务,OCRmyPDF 不会成为今天的软件。无论是扩展现有功能集,还是将 OCRmyPDF 集成到更大的系统中,我们都很乐意讨论各类咨询需求。
|
||||
|
||||
## 许可证
|
||||
|
||||
OCRmyPDF 软件根据 Mozilla 公共许可证 2.0 (MPL-2.0) 授权。此许可证允许将 OCRmyPDF 与其他代码集成,包括商业和闭源代码,但要求您发布对 OCRmyPDF 所做的源代码级修改。
|
||||
OCRmyPDF 软件采用 Mozilla Public License 2.0 (MPL-2.0) 授权。该许可证允许将 OCRmyPDF 与其他代码集成,包括商业代码和闭源代码,但要求你发布对 OCRmyPDF 所做的源代码级修改。
|
||||
|
||||
OCRmyPDF 的某些组件有其他许可证,如标准 SPDX 许可证标识符或 DEP5 版权和许可信息文件所示。一般来说,非核心代码根据 MIT 许可,文档和测试文件根据 Creative Commons ShareAlike 4.0 (CC-BY-SA 4.0) 许可。
|
||||
OCRmyPDF 的某些组件采用其他许可证,具体由标准 SPDX 许可证标识符或 DEP5 版权与许可信息文件标明。一般来说,非核心代码采用 MIT 许可证,文档和测试文件采用 Creative Commons ShareAlike 4.0 (CC-BY-SA 4.0) 许可证。
|
||||
|
||||
## 免责声明
|
||||
|
||||
本软件按"原样"分发,不提供任何明示或暗示的保证或条件。
|
||||
|
||||
这份中文版 README.md 保留了原始文档的所有重要信息,包括功能介绍、安装说明、语言支持、使用示例等内容,同时保持了原始格式和结构。
|
||||
本软件按“原样”分发,不提供任何明示或暗示的保证或条件。
|
||||
|
||||
+17
-6
@@ -6,7 +6,6 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import glob
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -18,8 +17,9 @@ import cyclopts
|
||||
from packaging.version import InvalidVersion, Version
|
||||
|
||||
try:
|
||||
from github import Github, GithubException
|
||||
from github import Auth, Github, GithubException
|
||||
except ImportError:
|
||||
Auth = None # type: ignore
|
||||
Github = None # type: ignore
|
||||
GithubException = Exception # type: ignore
|
||||
|
||||
@@ -68,7 +68,7 @@ def validate_release_notes(new_version: str) -> bool:
|
||||
|
||||
def get_github_client():
|
||||
"""Get an authenticated GitHub client."""
|
||||
if Github is None:
|
||||
if Github is None or Auth is None:
|
||||
print(f"{RED}error:{OFF} PyGithub is not installed")
|
||||
print(" Install with: pip install PyGithub")
|
||||
return None
|
||||
@@ -92,7 +92,7 @@ def get_github_client():
|
||||
return None
|
||||
|
||||
try:
|
||||
return Github(token)
|
||||
return Github(auth=Auth.Token(token))
|
||||
except GithubException as e:
|
||||
print(f"{RED}error:{OFF} Failed to authenticate with GitHub: {e}")
|
||||
return None
|
||||
@@ -281,7 +281,7 @@ def bump_version() -> None:
|
||||
actions = []
|
||||
|
||||
for path_pattern, version_pattern in config:
|
||||
paths = [Path(p) for p in glob.glob(path_pattern)]
|
||||
paths = list(Path().glob(path_pattern))
|
||||
|
||||
if not paths:
|
||||
print(f"error: Pattern {path_pattern} didn't match any files")
|
||||
@@ -305,7 +305,8 @@ def bump_version() -> None:
|
||||
|
||||
if not found_at_least_one_file_needing_update:
|
||||
print(
|
||||
f'''error: Didn't find any occurrences of "{find_pattern}" in "{path_pattern}"'''
|
||||
f'''error: Didn't find any occurrences of "{find_pattern}" '''
|
||||
f'''in "{path_pattern}"'''
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
@@ -330,6 +331,16 @@ def bump_version() -> None:
|
||||
contents = contents.replace(find, replace)
|
||||
path.write_text(contents, encoding="utf8")
|
||||
|
||||
# Format only after every file (including pyproject.toml) reflects the new
|
||||
# version. Running `uv run` while pyproject.toml still had the old version
|
||||
# would leave its post-bump environment/lockfile resync to happen for the
|
||||
# first time during the commit's pre-commit hooks instead of here, which
|
||||
# then aborts the commit with a spurious "files were modified by this
|
||||
# hook" error.
|
||||
for path, _find, _replace in actions:
|
||||
if path.suffix == ".py":
|
||||
subprocess.run(["uv", "run", "ruff", "format", str(path)], check=True)
|
||||
|
||||
print("Files updated.")
|
||||
print()
|
||||
|
||||
|
||||
+131
-3
@@ -121,6 +121,30 @@ representation. This is useful for redoing OCR, for fixing OCR text
|
||||
with a damaged character map (text is selectable but not searchable),
|
||||
and destroying redacted information.
|
||||
|
||||
### Tagged PDFs and structural markup
|
||||
|
||||
Some PDFs carry a logical structure tree (`/StructTreeRoot`), the markup that
|
||||
makes a "Tagged PDF" — typically the result of layout analysis or a born-digital
|
||||
export. By default OCRmyPDF treats this as a signal that the document may not need
|
||||
OCR and exits, in the same way it stops on PDFs that already contain text. Use
|
||||
`--tagged-pdf-mode ignore`, or one of `--mode skip`/`redo`/`force`, to process
|
||||
such a file anyway.
|
||||
|
||||
OCRmyPDF cannot rebuild a structure tree to match newly recognized text. When
|
||||
`--force-ocr` rasterizes pages, or `--redo-ocr` strips and rewrites the text layer,
|
||||
the structure tree no longer corresponds to the page content, so it is discarded.
|
||||
`--mode skip` leaves text pages untouched, so their structural markup is preserved.
|
||||
|
||||
:::{note}
|
||||
Preservation under `--mode skip` only holds when the output is not converted to
|
||||
PDF/A. PDF/A conversion is performed by Ghostscript, and Ghostscript 10.x discards
|
||||
the structure tree during conversion (Ghostscript 9.x preserved it). Because the
|
||||
default `--output-type auto` may fall back to Ghostscript, use
|
||||
`--output-type pdf` if you need to guarantee that a Tagged PDF's structural markup
|
||||
survives. For best results, install veraPDF so that speculative PDF/A
|
||||
conversion can sidestep this issue entirely in most real cases.
|
||||
:::
|
||||
|
||||
### Time and image size limits
|
||||
|
||||
By default, OCRmyPDF permits tesseract to run for three minutes (180
|
||||
@@ -187,6 +211,13 @@ include:
|
||||
Overrides the path to Tesseract's data files. This can allow
|
||||
simultaneous installation of the "best" and "fast" training data
|
||||
sets. OCRmyPDF does not manage this environment variable.
|
||||
|
||||
If you point ``TESSDATA_PREFIX`` at a hand-assembled ``tessdata``
|
||||
folder (for example, individual ``.traineddata`` files downloaded
|
||||
from tessdata_best), make sure it also contains the ``configs/``
|
||||
subdirectory with the ``hocr`` and ``txt`` files. OCRmyPDF requires
|
||||
these; without them Tesseract produces no output. See
|
||||
:ref:`Tesseract cannot open its config file <tesseract-config-missing>`.
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
@@ -419,6 +450,70 @@ curves. In this case, you may want to use a different color conversion
|
||||
strategy. The `--color-conversion-strategy` option allows you to select a
|
||||
different strategy, such as `RGB`.
|
||||
|
||||
## Advanced Ghostscript tuning
|
||||
|
||||
:::{versionadded} 17.5.0
|
||||
:::
|
||||
|
||||
OCRmyPDF intentionally hides most Ghostscript controls because Ghostscript
|
||||
is a legacy code path. The preferred PDF/A pipeline in v17+ uses pypdfium2
|
||||
as the rasterizer and verapdf to validate speculative PDF/A output, with
|
||||
Ghostscript reserved as a fallback for PDFs that cannot be made compliant
|
||||
without it. OCRmyPDF's separate optimizer (controlled by `--optimize`,
|
||||
`--jpeg-quality`, `--png-quality`, etc.) is the supported way to shrink
|
||||
output PDFs: it gives consistent results across input files, and isolates
|
||||
Ghostscript so it can focus on producing a PDF/A with as few image
|
||||
transformations as possible.
|
||||
|
||||
The two options below are exposed for advanced users who want to tune
|
||||
Ghostscript's intermediate PDF/A output directly. Most users will get
|
||||
more predictable results from the optimizer.
|
||||
|
||||
### `--ghostscript-jpeg-quality Q`
|
||||
|
||||
Sets Ghostscript's `-dJPEGQ` switch for images that Ghostscript chooses
|
||||
to recompress to JPEG while building a PDF/A. `Q=0` requests maximum
|
||||
compression and `Q=100` requests best quality; if the flag is omitted,
|
||||
OCRmyPDF passes `95` (the historical default). This only affects images
|
||||
Ghostscript transcodes — existing JPEGs pass through unchanged on modern
|
||||
Ghostscript releases. For end-to-end JPEG quality tuning, prefer
|
||||
`--jpeg-quality`, which is implemented by the OCRmyPDF optimizer and is
|
||||
applied independently of whatever Ghostscript decides to do.
|
||||
|
||||
Note: setting both `--ghostscript-jpeg-quality` and `--jpeg-quality` can
|
||||
result in double JPEG recompression, since the optimizer may re-encode
|
||||
images that Ghostscript already recompressed. This can degrade quality
|
||||
in subtle ways.
|
||||
|
||||
### `--ghostscript-jpeg-maxdpi DPI`
|
||||
|
||||
Enables Ghostscript's image downsampling and caps color, grayscale, and
|
||||
monochrome image resolution to `DPI`. The downsample threshold is set to
|
||||
`1.0`, so any image whose effective DPI exceeds the cap will be
|
||||
downsampled.
|
||||
|
||||
Reducing JPEG quality is almost always a better trade than downsampling
|
||||
at the same compression budget: a 400 DPI JPEG at modest quality usually
|
||||
looks much better than a 200 DPI JPEG, because the JPEG codec can spend
|
||||
bits where they count. Downsampling is also dangerous for PDFs that
|
||||
combine a low-resolution color image with a high-resolution monochrome
|
||||
mask — capping the mask resolution can produce visible quality loss.
|
||||
For these reasons, prefer `--jpeg-quality` over `--ghostscript-jpeg-maxdpi`
|
||||
unless you specifically want to force a hard DPI cap.
|
||||
|
||||
Example:
|
||||
|
||||
```bash
|
||||
ocrmypdf --output-type pdfa \
|
||||
--ghostscript-jpeg-quality 80 \
|
||||
--ghostscript-jpeg-maxdpi 150 \
|
||||
in.pdf out.pdf
|
||||
```
|
||||
|
||||
These options only take effect when Ghostscript is invoked for PDF/A
|
||||
conversion (`--output-type pdfa`, `pdfa-1`, `pdfa-2`, or `pdfa-3`, or
|
||||
when `--output-type auto` falls back to Ghostscript).
|
||||
|
||||
## PDF/A output modes
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
@@ -438,6 +533,36 @@ OCRmyPDF can produce PDF/A compliant output for long-term archival. The
|
||||
| `pdf` | Standard PDF, no PDF/A conversion |
|
||||
| `none` | No output file (useful with `--sidecar`) |
|
||||
|
||||
### Non-embedded fonts and PDF/A
|
||||
|
||||
:::{versionadded} 17.8.0
|
||||
OCRmyPDF now refuses to corrupt non-embedded CID text layers during PDF/A
|
||||
conversion.
|
||||
:::
|
||||
|
||||
PDF/A requires every font to be embedded. If your input already has a text
|
||||
layer that uses *non-embedded* CID fonts — most commonly a CJK
|
||||
(Chinese-Japanese-Korean) OCR layer
|
||||
produced by Adobe Acrobat, which relies on the reader's system fonts —
|
||||
Ghostscript would have to substitute and re-embed a replacement font to make
|
||||
the file PDF/A. For CID-keyed (CJK) fonts this routinely corrupts the
|
||||
character-to-Unicode mapping, so the text silently becomes garbage or stops
|
||||
being searchable even though the page still *looks* correct.
|
||||
|
||||
Rather than emit corrupted output, OCRmyPDF detects this situation and:
|
||||
|
||||
- with `--output-type auto` (the default), produces a regular PDF instead of
|
||||
PDF/A, preserving the existing text layer exactly;
|
||||
- with an explicit `--output-type pdfa` (or `pdfa-1`/`pdfa-2`/`pdfa-3`), stops
|
||||
with an error.
|
||||
|
||||
This is a Ghostscript limitation that OCRmyPDF cannot repair, because a
|
||||
non-embedded font cannot be made PDF/A-compliant without re-embedding it. To
|
||||
keep the existing text layer, use `--output-type pdf`. To produce PDF/A anyway,
|
||||
re-run OCR with `--force-ocr`, which discards the original text layer and
|
||||
rebuilds it with embedded fonts. Text layers whose fonts are *already embedded*
|
||||
are converted to PDF/A normally.
|
||||
|
||||
### Speculative PDF/A conversion
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
@@ -451,9 +576,12 @@ fast "speculative" PDF/A conversion that avoids Ghostscript when possible:
|
||||
3. If validation passes, Ghostscript is skipped entirely
|
||||
4. If validation fails or verapdf is unavailable, falls back to Ghostscript
|
||||
|
||||
This approach is faster and avoids some Ghostscript limitations (such as
|
||||
image transcoding), but only works for PDFs that are already "mostly"
|
||||
PDF/A compliant.
|
||||
This fast path avoids some Ghostscript limitations (such as image
|
||||
transcoding) and is used whenever it can produce valid PDF/A. When it
|
||||
cannot — for example when veraPDF is not installed, or the input needs real
|
||||
conversion — `auto` falls back to Ghostscript so that it still produces
|
||||
PDF/A by default, matching OCRmyPDF 16 and earlier. If even Ghostscript
|
||||
cannot safely produce PDF/A, `auto` outputs a regular PDF instead of failing.
|
||||
|
||||
### PDF/A conversion flow
|
||||
|
||||
|
||||
@@ -95,10 +95,12 @@ from multiprocessing import Process
|
||||
import ocrmypdf
|
||||
from ocrmypdf import OcrOptions
|
||||
|
||||
|
||||
def ocrmypdf_process():
|
||||
options = OcrOptions(input_file='input.pdf', output_file='output.pdf')
|
||||
ocrmypdf.ocr(options)
|
||||
|
||||
|
||||
def call_ocrmypdf_from_my_app():
|
||||
p = Process(target=ocrmypdf_process)
|
||||
p.start()
|
||||
|
||||
+9
-1
@@ -174,7 +174,15 @@ docker run \
|
||||
--env PYTHONUNBUFFERED=1 \
|
||||
--interactive --tty --entrypoint python3 \
|
||||
jbarlow83/ocrmypdf \
|
||||
watcher.py
|
||||
/app/watcher.py
|
||||
:::
|
||||
|
||||
:::{note}
|
||||
The image runs as the non-root `app` user (uid 1000) by default, so it
|
||||
may not be able to write to the `/output` and `/processed` volumes unless
|
||||
you add a `--user` argument. The correct value depends on whether you use
|
||||
rootful Docker, rootless Docker, or Podman -- see
|
||||
{ref}`Bind-mounted volumes <docker-volumes>` for details.
|
||||
:::
|
||||
|
||||
This service will watch for a file that matches `/input/\*.pdf`, convert
|
||||
|
||||
@@ -178,6 +178,20 @@ html_theme = 'sphinx_rtd_theme'
|
||||
#
|
||||
html_theme_options = {}
|
||||
|
||||
# ReadTheDocs used to inject the "Edit on GitHub" context automatically, but
|
||||
# dropped it when it switched to Addons, so set it explicitly here. This makes
|
||||
# sphinx_rtd_theme add an "Edit on GitHub" link to each page that points at the
|
||||
# corresponding source file in the repository, replacing the static
|
||||
# "View page source" (_sources/*.txt) link. See
|
||||
# https://github.com/ocrmypdf/OCRmyPDF/issues/1490
|
||||
html_context = {
|
||||
'display_github': True,
|
||||
'github_user': 'ocrmypdf',
|
||||
'github_repo': 'OCRmyPDF',
|
||||
'github_version': 'main',
|
||||
'conf_py_path': '/docs/',
|
||||
}
|
||||
|
||||
# Add any paths that contain custom themes here, relative to this directory.
|
||||
# html_theme_path = []
|
||||
|
||||
|
||||
+47
-12
@@ -31,6 +31,16 @@ ocrmypdf --output-type pdf input.pdf output.pdf
|
||||
ocrmypdf --output-type pdfa --pdfa-image-compression jpeg input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Reduce JPEG quality with the optimizer
|
||||
|
||||
This is the recommended way to shrink JPEG content in the output. The
|
||||
optimizer applies regardless of `--output-type`, so it works on both
|
||||
plain PDFs and Ghostscript-produced PDF/A files.
|
||||
|
||||
```bash
|
||||
ocrmypdf --optimize 2 --jpeg-quality 60 input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Modify a file in place
|
||||
|
||||
The file will only be overwritten if OCRmyPDF is successful.
|
||||
@@ -239,19 +249,34 @@ case. Use `--tesseract-non-ocr-timeout` to control the timeout for
|
||||
non-OCR operations, if needed.
|
||||
:::
|
||||
|
||||
### Remove all text or OCR from my PDF
|
||||
### Remove the OCR text layer from my PDF
|
||||
|
||||
This is getting ridiculous, but OCRmyPDF can complete strip all textual
|
||||
information from a PDF and reconstruct it as a \"bag of images\" PDF.
|
||||
To remove the invisible OCR text layer while keeping the original pages
|
||||
exactly as they are -- no rasterizing, no change to images or visible
|
||||
content, and a smaller output file -- use `--mode strip`:
|
||||
|
||||
```bash
|
||||
ocrmypdf --mode strip input.pdf output.pdf
|
||||
```
|
||||
|
||||
Why would you want to do this? Perhaps you have a PDF where OCR failed to
|
||||
produce useful results and you simply want to get rid of it.
|
||||
|
||||
`--mode strip` removes only text drawn as *invisible* (PDF text render
|
||||
mode 3), which is how OCRmyPDF and most OCR tools add a searchable layer
|
||||
over a scanned page. Some OCR products -- and OCRmyPDF v2.2 and earlier --
|
||||
instead draw *visible* text and paint an opaque image on top of it. That
|
||||
text is part of the visible page, so `--mode strip` cannot remove it
|
||||
without altering the page's appearance.
|
||||
|
||||
To strip *all* text, including such visible text, rasterize the whole page
|
||||
into a \"bag of images\" PDF instead (this rebuilds every page as an image,
|
||||
so the file usually grows and vector content is lost):
|
||||
|
||||
```bash
|
||||
ocrmypdf --ocr-engine none --force-ocr input.pdf output.pdf
|
||||
```
|
||||
|
||||
Why would you want to do this? Perhaps you have a PDF where OCR fails to
|
||||
produce useful results, and just want to get rid of all OCR information.
|
||||
This command also removes OCR generated by third party tools.
|
||||
|
||||
### Optimize images without performing OCR
|
||||
|
||||
You can also optimize all images without performing any OCR:
|
||||
@@ -333,12 +358,22 @@ Hyphens denote a range of pages and commas separate page numbers. If you
|
||||
prefer to use spaces, quote all of the page numbers:
|
||||
`--pages '2, 3, 5, 7'`.
|
||||
|
||||
The token `end` (case-insensitive) is an alias for the last page in the
|
||||
document. For example, `--pages 3-end` OCRs from page 3 through the
|
||||
final page, and `--pages end` OCRs only the last page:
|
||||
|
||||
```bash
|
||||
ocrmypdf --pages 3-end input.pdf output.pdf
|
||||
ocrmypdf --pages end input.pdf output.pdf
|
||||
```
|
||||
|
||||
OCRmyPDF will warn if your list of page numbers contains duplicates or
|
||||
overlapping pages. OCRmyPDF does not currently account for document page
|
||||
numbers, such as an introduction section of a book that uses Roman
|
||||
numerals. It simply counts the number of virtual pieces of paper since
|
||||
the start. If your list of pages is out of numerical order, OCRmyPDF
|
||||
will sort it for you.
|
||||
overlapping pages. (Repeated page numbers are de-duplicated automatically,
|
||||
since the underlying set of pages is what matters.) OCRmyPDF does not
|
||||
currently account for document page numbers, such as an introduction
|
||||
section of a book that uses Roman numerals. It simply counts the number
|
||||
of virtual pieces of paper since the start. If your list of pages is out
|
||||
of numerical order, OCRmyPDF will sort it for you.
|
||||
|
||||
Regardless of the argument to `--pages`, OCRmyPDF will optimize all
|
||||
pages/images in the file and convert it to PDF/A, unless you disable
|
||||
|
||||
+84
-23
@@ -71,15 +71,29 @@ application (as opposed to the more conventional case, where a Docker
|
||||
container runs as a server). For that reason we usually use the `--rm`
|
||||
argument to delete the container when it exits.
|
||||
|
||||
:::{note}
|
||||
The image runs as a non-root user (`app`, uid/gid 1000) by default,
|
||||
rather than as root. This is a defense-in-depth measure: a flaw in
|
||||
OCRmyPDF or one of its dependencies cannot trivially act as root inside
|
||||
the container. The examples below assume **rootless Docker** or
|
||||
**Podman**; the differences for traditional *rootful* Docker are
|
||||
described separately under *Special case: rootful Docker* below.
|
||||
:::
|
||||
|
||||
To start a Docker container (instance of the image):
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm -i jbarlow83/ocrmypdf-alpine (... all other arguments here...) - -
|
||||
:::
|
||||
|
||||
For convenience, create a shell alias to hide the Docker command. It is
|
||||
easier to send the input file as stdin and read the output from stdout
|
||||
-- **this avoids the messy permission issues with Docker entirely**.
|
||||
### Recommended: pipe through stdin and stdout
|
||||
|
||||
The easiest and most portable way to use the image is to send the input
|
||||
file on stdin and read the output from stdout. This **avoids file
|
||||
permission issues entirely** -- nothing is written to a mounted
|
||||
directory, so it does not matter which user the container runs as, nor
|
||||
whether you use rootless or rootful Docker. For convenience, create a
|
||||
shell alias to hide the Docker command:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i jbarlow83/ocrmypdf-alpine'
|
||||
@@ -90,28 +104,42 @@ docker_ocrmypdf - - <input.pdf >output.pdf
|
||||
Or in the wonderful [fish shell](https://fishshell.com/):
|
||||
|
||||
:::{code} fish
|
||||
alias docker_ocrmypdf 'docker run --rm jbarlow83/ocrmypdf-alpine'
|
||||
alias docker_ocrmypdf 'docker run --rm -i jbarlow83/ocrmypdf-alpine'
|
||||
funcsave docker_ocrmypdf
|
||||
:::
|
||||
|
||||
Alternately, you could mount the local current working directory as a
|
||||
Docker volume:
|
||||
{#docker-volumes}
|
||||
### Bind-mounted volumes
|
||||
|
||||
If you would rather mount a directory and pass file paths, you need to
|
||||
consider which user owns the files OCRmyPDF writes back into that
|
||||
directory. The image's default working directory is `/data`, so mounting
|
||||
your files there lets you pass plain relative paths without an explicit
|
||||
`--workdir`. Because the container runs as the non-root `app` user, the
|
||||
right invocation otherwise depends on your container runtime.
|
||||
|
||||
**Rootless Docker (the assumed default).** Your own account runs the
|
||||
daemon, so the container's `root` maps back to *your* unprivileged host
|
||||
user, while every other container uid -- including the image's default
|
||||
`app`/1000 -- maps to a *subordinate* uid. A directory you own on the
|
||||
host therefore appears owned by `root` inside the container, so the
|
||||
default `app` user usually **cannot write to it at all**. Run the job as
|
||||
container-`root`, which under rootless Docker is still your ordinary host
|
||||
user, so the write succeeds and the output is owned by you:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" --workdir /data -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias docker_ocrmypdf='docker run --rm -i --user 0:0 -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
## Podman
|
||||
|
||||
Especially if you use [Podman](https://podman.io/) (or use Docker in
|
||||
rootless mode), you may need to add `--userns keep-id` there,
|
||||
otherwise you may get access errors, because the user ID is otherwise not
|
||||
mapped to the same UID as on the host:
|
||||
**Podman.** Podman provides `--userns keep-id`, which maps your host uid
|
||||
straight through into the container. Combined with `--user`, you run as
|
||||
your own uid and own the output directly, otherwise you may get access
|
||||
errors because the user ID is not mapped to the same UID as on the host:
|
||||
|
||||
:::{code} bash
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id --workdir /data -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
If you have SELinux enabled, you may additionally need to add the `:Z` [suffix to
|
||||
@@ -124,10 +152,27 @@ the end of the linked podman documentation for details. This results in
|
||||
the following full command:
|
||||
|
||||
:::{code} bash
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id --workdir /data -v "$PWD:/data" --security-opt label=disable jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id -v "$PWD:/data" --security-opt label=disable jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
{#docker-rootful}
|
||||
### Special case: rootful Docker
|
||||
|
||||
With a traditional root daemon, container uid *N* is the *same* uid *N*
|
||||
on the host. Running the container as root would therefore fill your
|
||||
mounted directory with root-owned files and -- more importantly -- a
|
||||
container escape would run as real host root. Drop to your own uid so the
|
||||
output is owned by you and the process stays unprivileged:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
The non-root default and the `--user` override both reduce the risk here,
|
||||
but rootless Docker or Podman remain the safer choice when available.
|
||||
|
||||
{#docker-lang-packs}
|
||||
## Adding languages to the Docker image
|
||||
|
||||
@@ -139,8 +184,12 @@ creating a new Dockerfile based on the public one.
|
||||
:::{code} dockerfile
|
||||
FROM jbarlow83/ocrmypdf
|
||||
|
||||
# The image runs as the non-root "app" user, so switch back to root for
|
||||
# build steps that install packages, then drop back to "app".
|
||||
USER root
|
||||
# Example: add Italian
|
||||
RUN apt install tesseract-ocr-ita
|
||||
RUN apt-get update && apt-get install -y tesseract-ocr-ita
|
||||
USER app
|
||||
:::
|
||||
|
||||
To install language packs (training data) such as the
|
||||
@@ -179,7 +228,11 @@ Extending the Docker image
|
||||
--------------------------
|
||||
|
||||
You can extend the Docker image with your own customizations, similar to
|
||||
the way it is extended to add language packs.
|
||||
the way it is extended to add language packs. Because the image runs as
|
||||
the non-root `app` user, switch to `USER root` for any build steps that
|
||||
require root (installing packages, writing to system directories) and
|
||||
back to `USER app` afterwards, as shown in the language pack example
|
||||
above.
|
||||
|
||||
Note that the Docker image is subject to change at any time. For
|
||||
example, the base image may be updated to a newer version of Ubuntu or
|
||||
@@ -196,7 +249,7 @@ Executing the test suite
|
||||
The OCRmyPDF test suite is installed with image. To run it:
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm --entrypoint python jbarlow83/ocrmypdf -m pytest
|
||||
docker run --rm --workdir /app --entrypoint python jbarlow83/ocrmypdf -m pytest
|
||||
:::
|
||||
|
||||
Accessing the shell
|
||||
@@ -205,7 +258,15 @@ Accessing the shell
|
||||
To use the shell in the Docker image:
|
||||
|
||||
:::{code} bash
|
||||
docker run -it --entrypoint sh jbarlow83/ocrmypdf
|
||||
docker run -it --entrypoint sh jbarlow83/ocrmypdf-alpine
|
||||
:::
|
||||
|
||||
This shell runs as the non-root `app` user. If you need root inside the
|
||||
container -- for example to install extra packages with `apk` or `apt` --
|
||||
add `--user root`:
|
||||
|
||||
:::{code} bash
|
||||
docker run -it --user root --entrypoint sh jbarlow83/ocrmypdf-alpine
|
||||
:::
|
||||
|
||||
Using the OCRmyPDF web service wrapper
|
||||
@@ -215,7 +276,7 @@ The OCRmyPDF Docker image includes an example, barebones HTTP web
|
||||
service. The webservice may be launched as follows:
|
||||
|
||||
:::{code} bash
|
||||
docker run --entrypoint python -p 5000:5000 jbarlow83/ocrmypdf webservice.py
|
||||
docker run --entrypoint python -p 5000:5000 jbarlow83/ocrmypdf /app/webservice.py
|
||||
:::
|
||||
|
||||
We omit the `--rm` parameter so that the container will not be
|
||||
|
||||
@@ -49,3 +49,31 @@ pdftk input.pdf cat output output.pdf
|
||||
|
||||
Sometimes Acrobat can repair PDFs with its [Preflight
|
||||
tool](https://helpx.adobe.com/acrobat/using/correcting-problem-areas-preflight-tool.html).
|
||||
|
||||
(tesseract-config-missing)=
|
||||
|
||||
## Tesseract cannot open its config file \'hocr\' or \'txt\'
|
||||
|
||||
:::{code}
|
||||
ERROR - Tesseract cannot open its config file 'hocr'.
|
||||
:::
|
||||
|
||||
OCRmyPDF asks Tesseract to produce `hocr` and `txt` output. Tesseract
|
||||
reads the instructions for these output formats from configuration files
|
||||
named `hocr` and `txt` that live in the `configs/` subdirectory of its
|
||||
`tessdata` folder. If those files are missing, Tesseract prints
|
||||
`read_params_file: Can't open hocr`, exits without error, and produces no
|
||||
output.
|
||||
|
||||
This usually happens when a `tessdata` directory was assembled by hand --
|
||||
for example, by downloading individual `.traineddata` files from
|
||||
[tessdata_best](https://github.com/tesseract-ocr/tessdata_best) and
|
||||
pointing `TESSDATA_PREFIX` at them -- because those repositories do not
|
||||
include the `configs/` directory. A complete Tesseract installation from
|
||||
your operating system\'s package manager includes it.
|
||||
|
||||
To fix this, ensure the `configs/hocr` and `configs/txt` files exist in
|
||||
the `tessdata` directory that Tesseract is using. Copying the `configs/`
|
||||
directory from a full Tesseract installation is sufficient. See
|
||||
{envvar}`TESSDATA_PREFIX` for more on selecting an alternate `tessdata`
|
||||
folder.
|
||||
|
||||
+19
-3
@@ -663,9 +663,25 @@ provides text shaping for proper multilingual support. These replace the
|
||||
legacy hOCR-based renderer. Install with: `pip install fpdf2 uharfbuzz`
|
||||
|
||||
**fonts-noto** (or an equivalent comprehensive font package) is recommended
|
||||
for proper text rendering, especially for non-Latin scripts. On Debian/Ubuntu:
|
||||
`apt install fonts-noto`. On Fedora: `dnf install google-noto-fonts-common`.
|
||||
On macOS with Homebrew: `brew install font-noto`.
|
||||
for proper text rendering, especially for non-Latin scripts. OCRmyPDF bundles
|
||||
a Latin font only, and discovers the rest from the fonts installed on your
|
||||
system.
|
||||
|
||||
- Debian/Ubuntu: `apt install fonts-noto`
|
||||
- Fedora: `dnf install google-noto-fonts-all`
|
||||
- macOS with Homebrew: Homebrew has no single Noto package; each family is a
|
||||
separate cask. Install at least
|
||||
`brew install --cask font-noto-sans font-noto-serif`, plus a cask per
|
||||
additional script you OCR, for example
|
||||
`brew install --cask font-noto-sans-arabic font-noto-sans-cjk`. Run
|
||||
`brew search font-noto` to list them all.
|
||||
|
||||
If OCRmyPDF warns that no installed font has glyphs for some of the text, the
|
||||
message names the characters it could not render, for example
|
||||
`'Ꮳ' U+13E3 CHEROKEE LETTER TSA`. Install the Noto font for that script — here,
|
||||
`fonts-noto-core` on Debian or `font-noto-sans-cherokee` on Homebrew. The text
|
||||
layer remains searchable and copyable either way; only its appearance when
|
||||
highlighted in a PDF viewer is affected.
|
||||
|
||||
**pypdfium2**, if present, provides fast PDF page rasterization using
|
||||
the pdfium library (the same library used by Google Chrome). It is
|
||||
|
||||
+16
-4
@@ -178,11 +178,23 @@ v17 addresses through alternative codepaths. When Ghostscript is used:
|
||||
encoding, which may introduce compression artifacts, if Ghostscript
|
||||
PDF/A is enabled.
|
||||
- Ghostscript may transcode grayscale and color images, potentially
|
||||
lossily, based on an internal algorithm. This
|
||||
behavior can be suppressed by setting `--pdfa-image-compression` to
|
||||
`jpeg` or `lossless` to set all images to one type or the other.
|
||||
Ghostscript lacks an option to maintain the input image's format.
|
||||
lossily, based on an internal algorithm. By default
|
||||
(`--pdfa-image-compression=auto`) OCRmyPDF selects lossless image
|
||||
compression at `-O0` so Ghostscript will not transcode lossless images
|
||||
to JPEG. At `-O1` (the default optimization level) and above, `auto`
|
||||
defers to Ghostscript's heuristic instead; `-O1` is a historical
|
||||
exception, kept for backwards compatibility because coercing it to
|
||||
lossless can substantially bloat output. You can override this by
|
||||
setting `--pdfa-image-compression` to `jpeg` or `lossless` to force all
|
||||
images to one type or the other. `lossless` passes existing JPEGs
|
||||
through untouched (re-encoding them losslessly would only inflate them)
|
||||
while encoding non-JPEG images losslessly.
|
||||
(Modern Ghostscript can copy JPEG images without transcoding them.)
|
||||
Advanced users can also tune Ghostscript's image recompression with
|
||||
`--ghostscript-jpeg-quality` and `--ghostscript-jpeg-maxdpi`; see
|
||||
[Advanced Ghostscript tuning](advanced.md#advanced-ghostscript-tuning).
|
||||
Most users should prefer `--jpeg-quality` (applied by the OCRmyPDF
|
||||
optimizer) over those Ghostscript-scoped controls.
|
||||
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
||||
one of the standard XMP metadata namespaces for PDFs. In particular,
|
||||
PRISM Metadata is removed.
|
||||
|
||||
+3
-1
@@ -47,7 +47,9 @@ OCRmyPDF has the following runtime dependencies:
|
||||
**For text rendering** (expressing OCR results in PDF):
|
||||
- `fpdf2` (Python package) - Required for text layer rendering
|
||||
- `uharfbuzz` (Python package) - Required for text layer rendering
|
||||
- `font-noto` (system package) - Recommended for text layer rendering
|
||||
- Noto fonts (system package) - Recommended for text layer rendering.
|
||||
`fonts-noto` on Debian/Ubuntu, `google-noto-fonts-all` on Fedora; Homebrew
|
||||
has no single Noto package, only per-family casks such as `font-noto-sans`.
|
||||
|
||||
**Other dependencies**:
|
||||
- `unpaper` (system binary) - Optional, enables `--clean` and `--clean-final`
|
||||
|
||||
+10
-1
@@ -98,7 +98,16 @@ If `pngquant` is installed, OCRmyPDF will use it to perform quantize
|
||||
paletted images to reduce their size.
|
||||
|
||||
The quality of JPEGs may be lowered, on the assumption that a lower
|
||||
quality image may be suitable for storage after OCR.
|
||||
quality image may be suitable for storage after OCR. Use `--jpeg-quality`
|
||||
to control the optimizer's JPEG quality target. The optimizer is the
|
||||
recommended way to reduce JPEG image sizes: it applies consistently
|
||||
regardless of whether Ghostscript was used to produce a PDF/A.
|
||||
|
||||
If you specifically need to tune Ghostscript's own PDF/A image handling
|
||||
(for example, to force a hard DPI cap), see
|
||||
[Advanced Ghostscript tuning](advanced.md#advanced-ghostscript-tuning)
|
||||
for the separate `--ghostscript-jpeg-quality` and
|
||||
`--ghostscript-jpeg-maxdpi` options.
|
||||
|
||||
It is not possible to optimize all image types. Uncommon image types may
|
||||
be skipped by the optimizer.
|
||||
|
||||
+13
-13
@@ -120,6 +120,7 @@ A plugin may provide the following hooks. Hooks must be decorated with
|
||||
```python
|
||||
from ocrmypdf import hookimpl
|
||||
|
||||
|
||||
@hookimpl
|
||||
def add_options(parser):
|
||||
pass
|
||||
@@ -205,12 +206,11 @@ from ocrmypdf._options import OcrOptions
|
||||
|
||||
```python
|
||||
# Before (v16 and earlier)
|
||||
def check_options(options: argparse.Namespace) -> None:
|
||||
...
|
||||
def check_options(options: argparse.Namespace) -> None: ...
|
||||
|
||||
|
||||
# After (v17+)
|
||||
def check_options(options: OcrOptions) -> None:
|
||||
...
|
||||
def check_options(options: OcrOptions) -> None: ...
|
||||
```
|
||||
|
||||
**Attribute access unchanged:**
|
||||
@@ -229,6 +229,7 @@ options.tesseract_timeout
|
||||
def check_options(options):
|
||||
options.some_computed_value = compute_value(options)
|
||||
|
||||
|
||||
# After (v17 pattern - compute at point of use)
|
||||
def some_function(options):
|
||||
computed = compute_value(options)
|
||||
@@ -336,19 +337,17 @@ from ocrmypdf import OcrElement, OcrClass, BoundingBox
|
||||
|
||||
# OcrElement - represents any OCR structural unit
|
||||
page = OcrElement(
|
||||
ocr_class=OcrClass.PAGE,
|
||||
bbox=BoundingBox(0, 0, 612, 792),
|
||||
children=[...]
|
||||
ocr_class=OcrClass.PAGE, bbox=BoundingBox(0, 0, 612, 792), children=[...]
|
||||
)
|
||||
|
||||
# BoundingBox - axis-aligned bounding box (left, top, right, bottom)
|
||||
bbox = BoundingBox(left=100, top=50, right=300, bottom=80)
|
||||
|
||||
# OcrClass - constants for element types
|
||||
OcrClass.PAGE # "ocr_page"
|
||||
OcrClass.LINE # "ocr_line"
|
||||
OcrClass.WORD # "ocrx_word"
|
||||
OcrClass.PARAGRAPH # "ocr_par"
|
||||
OcrClass.PAGE # "ocr_page"
|
||||
OcrClass.LINE # "ocr_line"
|
||||
OcrClass.WORD # "ocrx_word"
|
||||
OcrClass.PARAGRAPH # "ocr_par"
|
||||
```
|
||||
|
||||
**Navigating the tree:**
|
||||
@@ -378,6 +377,7 @@ from pathlib import Path
|
||||
from ocrmypdf.pluginspec import OcrEngine
|
||||
from ocrmypdf import OcrElement, OcrClass, BoundingBox
|
||||
|
||||
|
||||
class MyOcrEngine(OcrEngine):
|
||||
def generate_ocr(
|
||||
self,
|
||||
@@ -402,10 +402,10 @@ class MyOcrEngine(OcrEngine):
|
||||
text="Hello",
|
||||
),
|
||||
# ... more words
|
||||
]
|
||||
],
|
||||
),
|
||||
# ... more lines
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
def supports_generate_ocr(self) -> bool:
|
||||
|
||||
@@ -3,6 +3,257 @@
|
||||
|
||||
# v17
|
||||
|
||||
## v17.9.0
|
||||
|
||||
- OCRmyPDF now uses any Noto font installed on the system, not just the two
|
||||
dozen script families it knows by name ({issue}`1722`). Previously a document
|
||||
in, say, Cherokee or Vai was rendered with the glyphless fallback font even
|
||||
though the matching font was installed — a common situation on macOS, which
|
||||
ships around a hundred script-specific Noto faces. When the named fonts
|
||||
cannot cover a word, OCRmyPDF now searches the installed fonts for one that
|
||||
can.
|
||||
- The "no installed font has glyphs" warning now names the characters it could
|
||||
not render, with their codepoints and Unicode names, so it is clear which
|
||||
font to install. Text that mixes scripts no single font covers is now
|
||||
reported as such, instead of advising the user to install fonts they may
|
||||
already have.
|
||||
- Fixed the macOS font installation instructions, which recommended a Homebrew
|
||||
package (`font-noto`) that does not exist ({issue}`1722`). Homebrew has no
|
||||
single Noto package; each family is a separate cask. The Fedora package name
|
||||
was also corrected to `google-noto-fonts-all`.
|
||||
- Font providers may now implement the optional `GlyphSearchingFontProvider`
|
||||
protocol to participate in coverage-based font search.
|
||||
- Fixed `--jpeg-quality`/`--jpg-quality` having no effect on the CLI: the
|
||||
value was silently dropped before reaching the optimizer, which then
|
||||
always used its own built-in default JPEG quality regardless of what was
|
||||
requested ({issue}`1723`). The same bug affected the Python API's
|
||||
`jpg_quality` parameter. `ocrmypdf.ocr()` now accepts `jpeg_quality`
|
||||
(matching the CLI flag name) as the canonical parameter; `jpg_quality`
|
||||
still works but is deprecated.
|
||||
- Hardened PDF parsing against malformed (non-dictionary) `/Resources`,
|
||||
`/XObject`, and `/FontDescriptor` entries, which previously crashed
|
||||
`ocrmypdf.ocr()` with `AttributeError`/`TypeError`/`ValueError` on
|
||||
otherwise-processable files, both during PDF/A font scanning and general
|
||||
image scanning ({issue}`1713`). Thanks @mvanhorn for the initial fix.
|
||||
- Release process improvements: fixed a CI bug where every push to main
|
||||
after a release was tagged would incorrectly revert the just-published
|
||||
GitHub release back to draft status.
|
||||
|
||||
## v17.8.1
|
||||
|
||||
- Improved the `--tesseract-pagesegmode` help text to point to
|
||||
`tesseract --help-extra`, since Tesseract 5.5.2 moved the page segmentation
|
||||
mode documentation there from `tesseract --help`. Thanks @sokai.
|
||||
- Internal refactoring: completed a project-wide mypy type-checking pass
|
||||
(`--check-untyped-defs` is now enabled, and the mypy pre-commit hook is now
|
||||
blocking rather than advisory), fixing several latent edge-case bugs
|
||||
surfaced along the way.
|
||||
- Release process improvements: migrated from pre-commit to prek for local
|
||||
git hooks, and added a dedicated lint job to CI.
|
||||
- Improved typing strictness for `Path`.
|
||||
|
||||
## v17.8.0
|
||||
|
||||
- `--output-type auto` (the default) again produces PDF/A whenever it can,
|
||||
matching OCRmyPDF 16's "PDF/A by default" behavior. It first tries the fast
|
||||
Ghostscript-free conversion (validated by veraPDF when available) and now
|
||||
falls back to Ghostscript when that cannot produce PDF/A, only emitting a
|
||||
regular PDF when even Ghostscript cannot safely convert (for example, an
|
||||
input with non-embedded CID/CJK fonts, per {issue}`1561`). A consequence is
|
||||
that the default path may once again invoke Ghostscript, which is slower and
|
||||
may transcode images; use `--output-type pdf` to skip PDF/A conversion
|
||||
entirely.
|
||||
- Fixed detection of veraPDF 1.30.0 and newer: recent builds print JVM
|
||||
warnings before their version string, which caused OCRmyPDF to report
|
||||
veraPDF as unavailable and skip the fast PDF/A path.
|
||||
- OCRmyPDF no longer silently corrupts a non-embedded CID (CJK) text layer when
|
||||
producing PDF/A ({issue}`1561`). PDF/A requires all fonts to be embedded, so
|
||||
Ghostscript substitutes and re-embeds non-embedded CID fonts — such as the OCR
|
||||
text layer Adobe Acrobat adds to scanned CJK documents — which mangles the
|
||||
text and destroys searchability. OCRmyPDF now detects non-embedded CID fonts
|
||||
before conversion: with `--output-type auto` (the default) it produces a
|
||||
regular PDF and preserves the existing text layer, and with an explicit
|
||||
`--output-type pdfa*` it stops with an error rather than emit corrupted
|
||||
output. Use `--output-type pdf` to keep the text layer, or `--force-ocr` to
|
||||
rebuild it with embedded fonts.
|
||||
- Writing the output PDF to standard output (`ocrmypdf input.pdf -`) is now
|
||||
protected against corruption at the operating system level. Previously
|
||||
OCRmyPDF relied on no in-process code — third-party libraries, plugins, or
|
||||
stray `print()` calls — ever writing to stdout; a single accidental write
|
||||
would silently corrupt the PDF. The command line program now saves the real
|
||||
stdout at startup, before plugins are loaded or any worker process/thread is
|
||||
started, and redirects file descriptor 1 to stderr, so that only OCRmyPDF's
|
||||
final PDF output can reach stdout. A consequence is that a plugin which
|
||||
intentionally prints to stdout will have that output redirected to stderr.
|
||||
- Added the public API function {func}`ocrmypdf.configure_stdout_protection`,
|
||||
which installs this same protection. Like {func}`ocrmypdf.configure_logging`,
|
||||
it is optional and intended for callers that want command-line-like behavior;
|
||||
applications that manage their own standard output should not call it.
|
||||
- Fixed an uncaught `UnicodeDecodeError` when processing a PDF whose
|
||||
`/DocumentInfo` dictionary contains a `/Name` key encoded in Latin-1 (or
|
||||
another non-UTF-8 encoding), such as `/Saks#e5r`. `repair_docinfo_nuls` now
|
||||
treats such a block as malformed, logs a message, and continues instead of
|
||||
crashing the pipeline ({issue}`1540`). Current pikepdf releases tolerate these
|
||||
keys by surrogate-escaping them, but older versions raised while iterating the
|
||||
dictionary.
|
||||
|
||||
## v17.7.1
|
||||
|
||||
- Fixed a severe, Windows-specific performance regression in the "Scanning
|
||||
contents" phase, most visible with `--redo-ocr` ({issue}`1662`). Since
|
||||
v16.4.3, OCRmyPDF forced pdfminer's read buffer to 256 MiB to work around a
|
||||
pdfminer bug that mishandled tokens split across the buffer boundary
|
||||
({issue}`1361`). On Windows, CPython's `BufferedReader.read()` eagerly
|
||||
allocates a buffer of the requested size on every read, so the oversized
|
||||
buffer made each of pdfminer's thousands of reads cost tens of milliseconds
|
||||
(this allocation is lazy, and effectively free, on Linux). The underlying
|
||||
pdfminer bug was fixed upstream in pdfminer.six 20250327
|
||||
([#1030](https://github.com/pdfminer/pdfminer.six/pull/1030)), with a
|
||||
follow-up for tokens split across streams in 20260107
|
||||
([#1158](https://github.com/pdfminer/pdfminer.six/pull/1158)), so the
|
||||
workaround has been removed and the minimum pdfminer.six version raised to
|
||||
20260107.
|
||||
- The font discovery used to build the OCR text layer now finds variable fonts
|
||||
such as `NotoSansArabic[wdth,wght].ttf`, the form shipped by Homebrew casks
|
||||
and current Google Fonts releases. Previously only static `-Regular.ttf`/`.otf`
|
||||
files were matched, so users who had installed the correct Noto font still got
|
||||
the glyphless fallback and a "No font found" warning ({issue}`1652`).
|
||||
- Font discovery is now language-aware for CJK: each Chinese, Japanese, and
|
||||
Korean language maps to its own per-language Noto family (NotoSansSC, TC, HK,
|
||||
JP, KR), with the pan-CJK super font kept as a shared fallback, since the
|
||||
per-language fonts are region subsets that may lack glyphs from other scripts.
|
||||
- The warning shown when no installed font has glyphs for some text was reworded
|
||||
to explain the consequence — the text is still added as a searchable, copyable
|
||||
layer but appears blank when highlighted in a viewer — and to name the specific
|
||||
font family to install.
|
||||
|
||||
## v17.7.0
|
||||
|
||||
- The Docker images now run as a non-root user (`app`, uid/gid 1000) by default
|
||||
rather than as root, as a defense-in-depth measure. If you bind-mount a
|
||||
directory for input and output, you may now need to add a `--user` argument so
|
||||
the container can write to it; the correct value differs for rootless Docker,
|
||||
Podman, and rootful Docker, and is described in the Docker documentation.
|
||||
Piping the input and output through stdin/stdout still works with no
|
||||
permission setup.
|
||||
- The Docker images now default their working directory to `/data`, so files in
|
||||
a directory mounted there can be given as relative paths without an explicit
|
||||
`--workdir`.
|
||||
- The Ubuntu Docker image now installs Tesseract 5 from the Ubuntu archive
|
||||
instead of the third-party `alex-p/tesseract-ocr5` PPA, and the base images
|
||||
were updated to Ubuntu 26.04 and Alpine 3.24.
|
||||
- Fixed a missing space in the error message shown when OCRmyPDF cannot access
|
||||
its working directory inside a Docker container.
|
||||
- Updated packaged dependencies, including the optional web service stack
|
||||
(starlette, tornado, python-multipart) and cryptography.
|
||||
|
||||
## v17.6.0
|
||||
|
||||
- When the optimizer encounters an image it cannot process (for example, an
|
||||
exotic colorspace that cannot be transcoded), it now logs a concise warning
|
||||
that the image was left unchanged rather than printing an alarming
|
||||
traceback. The output file was already valid in these cases; only the
|
||||
reporting was misleading. The full traceback is still available at debug
|
||||
verbosity (`-v 1`) ({issue}`846`).
|
||||
- `--pdfa-image-compression=auto` (the default) now selects lossless image
|
||||
compression at `-O0` so Ghostscript no longer transcodes lossless images to
|
||||
JPEG during PDF/A generation. At `-O1` and above, `auto` continues to defer
|
||||
to Ghostscript's heuristic, which may recompress images lossily. `-O1` (the
|
||||
default level) is kept as a historical exception because coercing it to
|
||||
lossless can substantially bloat output; users who want guaranteed lossless
|
||||
image handling should pass `--pdfa-image-compression=lossless` or use `-O0`
|
||||
({issue}`1124`).
|
||||
- `--pdfa-image-compression=lossless` now passes existing JPEG images through
|
||||
unchanged rather than re-encoding them with a lossless codec. Re-encoding an
|
||||
already-lossy JPEG losslessly cannot recover quality and only inflates the
|
||||
file, so JPEGs are preserved while non-JPEG images are encoded losslessly.
|
||||
- OCRmyPDF now validates and repairs malformed page-boundary boxes
|
||||
(``/MediaBox``, ``/CropBox``, ``/TrimBox``, ``/ArtBox``, ``/BleedBox``) in its
|
||||
input, following the PDF 2.0 specification. Coordinates written in invalid
|
||||
exponential notation are reinterpreted ({issue}`1398`); rectangles whose
|
||||
corners are given in reversed order are normalized, which previously crashed
|
||||
with ``NegativeDimensionError`` ({issue}`1526`); and a crop/trim/art/bleed box
|
||||
that falls outside the MediaBox is clamped to their intersection, or discarded
|
||||
when that intersection is empty, which previously produced an output with a
|
||||
zero-height effective page that some viewers refused to open ({issue}`1400`).
|
||||
When a box is discarded, clamped, or reinterpreted, OCRmyPDF logs a warning
|
||||
recommending visual inspection of the output. Thanks @ajdlinux for the initial
|
||||
fix in PR #1691.
|
||||
- OCRmyPDF now discards an embedded Adobe full-text search index
|
||||
(``/Root/PieceInfo/SearchIndex``) from its output. This proprietary index,
|
||||
produced by Acrobat's "Embed Index" feature, is read only by Adobe Acrobat;
|
||||
other viewers ignore it and search the text on the fly. Because any change to
|
||||
a PDF invalidates the index, retaining it after OCRmyPDF rewrites the document
|
||||
would leave a stale index that returns incorrect search results in Acrobat.
|
||||
Modern viewers rebuild a search index on demand, so there is no loss of
|
||||
search capability.
|
||||
- OCRmyPDF now discards embedded per-page thumbnail images (the optional
|
||||
``/Thumb`` image XObject on a page) from its output. OCRmyPDF alters page
|
||||
appearance (deskew, clean, rasterize, re-render) and plugins may edit pages
|
||||
arbitrarily, so a retained thumbnail would be stale and no longer match its
|
||||
page. Embedded thumbnails are a navigation aid that modern viewers generate
|
||||
on demand, so there is no loss of functionality.
|
||||
- Fixed a regression in OCR quality for PDFs that paint a 1-bit image mask
|
||||
(stencil) with a gray or colored fill color. Previously such pages were
|
||||
rasterized as 1-bit black-and-white before OCR, so Ghostscript dithered
|
||||
mid-tone text into an unreadable stipple and Tesseract failed to recognize
|
||||
it. The rasterizer now inspects the fill color used to paint a mask and
|
||||
promotes the page to grayscale or full color as needed, so the distinction
|
||||
is preserved for the OCR engine. This applies to both the Ghostscript and
|
||||
pypdfium rasterizers. {issue}`1688`
|
||||
- The default 1-bit raster device for Ghostscript is now ``pngmonod``
|
||||
(error-diffusion) instead of ``pngmono`` (ordered dithering). It produces
|
||||
better input for OCR on faint or anti-aliased scans at negligible cost and
|
||||
no change to output file size, since the rasterized image is an
|
||||
intermediate that is discarded after OCR.
|
||||
- When rasterizing pages with Ghostscript, OCRmyPDF now enables text and
|
||||
graphics anti-aliasing (``-dTextAlphaBits=4 -dGraphicsAlphaBits=4``) for the
|
||||
grayscale and color raster devices. Ghostscript 10.x renders aliased glyphs
|
||||
that OCR frequently misreads as extra word breaks or substituted characters;
|
||||
anti-aliasing materially improves OCR accuracy on the Ghostscript
|
||||
rasterization path, especially for small fonts at moderate resolution. The
|
||||
1-bit monochrome devices are unaffected, since they perform their own
|
||||
anti-aliased downscaling and older Ghostscript versions reject alpha-bit
|
||||
options on them. Note that the default rasterizer (``--rasterizer auto``)
|
||||
prefers pypdfium2, which already anti-aliases; this change benefits users who
|
||||
select ``--rasterizer ghostscript`` or do not have pypdfium2 installed.
|
||||
OCRmyPDF now also logs which rasterizer rendered each page at debug verbosity
|
||||
(``-v 1``), and the ``--rasterizer`` help text explains the OCR-quality
|
||||
trade-off, to make such reports easier to diagnose. {issue}`1439`
|
||||
- When Tesseract reports a page with many diacritics, OCRmyPDF still logs its
|
||||
interpreted "lots of diacritics - possibly poor OCR" hint, but now also emits
|
||||
Tesseract's raw message at debug verbosity (``-v 1``) so the original wording
|
||||
is available for diagnosis. {issue}`1566`
|
||||
- Added ``--mode strip``, which removes the invisible OCR text layer from a PDF
|
||||
in place. Unlike ``--ocr-engine none --force-ocr``, it does not rasterize the
|
||||
page, so images and visible content are preserved unchanged and the output is
|
||||
smaller rather than larger. Only text drawn as invisible (PDF text render mode
|
||||
3) is removed; some OCR engines -- and OCRmyPDF v2.2 and earlier -- express
|
||||
text as visible glyphs covered by an opaque image, and that text cannot be
|
||||
removed this way. {issue}`1435`
|
||||
|
||||
## v17.5.0
|
||||
|
||||
- Added support for the ``end`` alias in ``--pages``, denoting the last page
|
||||
of the document. For example, ``--pages 3-end`` OCRs from page 3 through
|
||||
the final page. {issue}`1615`
|
||||
- Added ``--ghostscript-jpeg-quality`` and ``--ghostscript-jpeg-maxdpi``
|
||||
advanced options for tuning Ghostscript's PDF/A output. The optimizer's
|
||||
``--jpeg-quality`` remains the recommended file-size control.
|
||||
- Fixed pypdfium2 rasterizer clipping content when the CropBox was smaller
|
||||
than the MediaBox (e.g. JSTOR or cropped PDFs). {issue}`1685`
|
||||
- Fixed Form XObject cycle detection in the optimizer's image xref scan.
|
||||
Self-referential or DAG-shaped Form graphs (notably from PowerPoint
|
||||
exports) previously produced floods of recursion warnings and could hang
|
||||
for minutes. {issue}`1321`
|
||||
- Tesseract config errors are now surfaced as ``TesseractConfigError`` with
|
||||
actionable guidance, instead of crashing later with a confusing
|
||||
``FileNotFoundError`` on the missing hOCR output. {issue}`1687`
|
||||
- Refreshed the Chinese README translation. Thanks @cislunarspace.
|
||||
- Internal refactoring of the ``_exec`` and ``subprocess`` modules to
|
||||
separate probing from execution.
|
||||
- CI dependency updates.
|
||||
|
||||
## v17.4.2
|
||||
|
||||
- Fixed Python API unconditionally overriding ``PIL.Image.MAX_IMAGE_PIXELS``
|
||||
|
||||
+2
-3
@@ -16,7 +16,6 @@ from __future__ import annotations
|
||||
|
||||
import filecmp
|
||||
import logging
|
||||
import os
|
||||
import posixpath
|
||||
import shutil
|
||||
import sys
|
||||
@@ -39,7 +38,7 @@ script_dir = Path(__file__).parent
|
||||
# set archive_dir to a path for backup original documents. Leave empty if not required.
|
||||
archive_dir = "/pdfbak"
|
||||
|
||||
start_dir = Path(sys.argv[1]) if len(sys.argv) > 1 else Path(".")
|
||||
start_dir = Path(sys.argv[1]) if len(sys.argv) > 1 else Path()
|
||||
|
||||
if len(sys.argv) > 2:
|
||||
log_file = Path(sys.argv[2])
|
||||
@@ -68,7 +67,7 @@ for filename in start_dir.glob("**/*.pdf"):
|
||||
try:
|
||||
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||
except OSError:
|
||||
os.makedirs(posixpath.dirname(archive_filename))
|
||||
Path(posixpath.dirname(archive_filename)).mkdir(parents=True)
|
||||
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||
try:
|
||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Helper script for bisecting PDFs to find a page with an issue."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
|
||||
@@ -46,6 +46,8 @@ __ocrmypdf_arguments()
|
||||
--rasterizer (PDF page rasterizer)
|
||||
--rotate-pages-threshold (page rotation confidence)
|
||||
--pdfa-image-compression (set PDF/A image compression options)
|
||||
--ghostscript-jpeg-quality (Ghostscript JPEG quality during PDF/A [0..100])
|
||||
--ghostscript-jpeg-maxdpi (cap Ghostscript image DPI during PDF/A)
|
||||
--fast-web-view (if file size if above this amount in MB linearize PDF)
|
||||
--continue-on-soft-render-error (continue after recoverable render errors)
|
||||
--plugin (name of plugin to import)
|
||||
@@ -337,6 +339,7 @@ __ocrmypdf_check_previous()
|
||||
|
||||
--title|--author|--subject|--keywords|--unpaper-args|--pages|--plugin|\
|
||||
--jpeg-quality|--png-quality|--image-dpi|--oversample|--skip-big|--max-image-mpixels|\
|
||||
--ghostscript-jpeg-quality|--ghostscript-jpeg-maxdpi|\
|
||||
--tesseract-timeout|--tesseract-non-ocr-timeout|--tesseract-downsample-above|\
|
||||
--rotate-pages-threshold|--fast-web-view)
|
||||
# argument required but no completions available
|
||||
|
||||
@@ -102,6 +102,8 @@ function __fish_ocrmypdf_pdfa_compression
|
||||
echo -e "lossless\t"(_ "convert color and grayscale images to lossless (PNG)")
|
||||
end
|
||||
complete -c ocrmypdf -x -l pdfa-image-compression -a '(__fish_ocrmypdf_pdfa_compression)' -d "set PDF/A image compression options"
|
||||
complete -c ocrmypdf -x -l ghostscript-jpeg-quality -d "Ghostscript JPEG quality during PDF/A [0..100]"
|
||||
complete -c ocrmypdf -x -l ghostscript-jpeg-maxdpi -d "cap Ghostscript image DPI during PDF/A"
|
||||
|
||||
complete -c ocrmypdf -x -s j -l jobs -d "how many worker processes to use"
|
||||
complete -c ocrmypdf -x -l title -d "set metadata"
|
||||
|
||||
@@ -6,12 +6,19 @@ services:
|
||||
ocrmypdf:
|
||||
restart: always
|
||||
container_name: ocrmypdf
|
||||
image: jbarlow83/ocrmypdf
|
||||
image: jbarlow83/ocrmypdf-alpine
|
||||
volumes:
|
||||
- "/media/scan:/input"
|
||||
- "/mnt/scan:/output"
|
||||
environment:
|
||||
- OCR_OUTPUT_DIRECTORY_YEAR_MONTH=0
|
||||
# The image runs as the non-root "app" user (uid 1000) by default. The
|
||||
# correct value here depends on your runtime, so that the watcher can write
|
||||
# to the /output bind mount and the files end up owned by you:
|
||||
# rootful Docker -> your host uid:gid
|
||||
# rootless Docker -> "0:0" (container root maps to your host user)
|
||||
# Podman -> your host uid:gid, plus `userns_mode: "keep-id"`
|
||||
# See docs/docker.md ("Bind-mounted volumes") for the reasoning.
|
||||
user: "<SET TO YOUR USER ID>:<SET TO YOUR GROUP ID>"
|
||||
entrypoint: python3
|
||||
command: watcher.py
|
||||
command: /app/watcher.py
|
||||
|
||||
@@ -37,8 +37,8 @@ def do_column(label, suffix, d):
|
||||
env[k] = v
|
||||
args = shlex.split(
|
||||
cli.format(
|
||||
in_=os.path.join(d, "input.pdf"),
|
||||
out=os.path.join(d, f"output{suffix}.pdf"),
|
||||
in_=Path(d) / "input.pdf",
|
||||
out=Path(d) / f"output{suffix}.pdf",
|
||||
)
|
||||
)
|
||||
with st.expander("Environment variables", expanded=bool(env_text.strip())):
|
||||
@@ -106,10 +106,10 @@ def main():
|
||||
)
|
||||
)
|
||||
|
||||
doc1 = pymupdf.open(os.path.join(d, "output1.pdf"))
|
||||
doc2 = pymupdf.open(os.path.join(d, "output2.pdf"))
|
||||
doc1 = pymupdf.open(Path(d, "output1.pdf"))
|
||||
doc2 = pymupdf.open(Path(d, "output2.pdf"))
|
||||
for i, page1_2 in enumerate(zip(doc1, doc2, strict=False)):
|
||||
st.write(f"Page {i+1}")
|
||||
st.write(f"Page {i + 1}")
|
||||
page1, page2 = page1_2
|
||||
col1, col2 = st.columns(2)
|
||||
with col1, st.container(border=True):
|
||||
|
||||
+3
-4
@@ -5,7 +5,6 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from tempfile import TemporaryDirectory
|
||||
@@ -60,10 +59,10 @@ def main():
|
||||
Path(d, "2.pdf").write_bytes(pdf_bytes2)
|
||||
|
||||
with st.expander("Text"):
|
||||
doc1 = pymupdf.open(os.path.join(d, "1.pdf"))
|
||||
doc2 = pymupdf.open(os.path.join(d, "2.pdf"))
|
||||
doc1 = pymupdf.open(Path(d, "1.pdf"))
|
||||
doc2 = pymupdf.open(Path(d, "2.pdf"))
|
||||
for i, page1_2 in enumerate(zip(doc1, doc2, strict=False)):
|
||||
st.write(f"Page {i+1}")
|
||||
st.write(f"Page {i + 1}")
|
||||
page1, page2 = page1_2
|
||||
col1, col2 = st.columns(2)
|
||||
with col1, st.container(border=True):
|
||||
|
||||
@@ -23,7 +23,7 @@ def main(
|
||||
engine: Annotated[str, cyclopts.Parameter()] = 'pdftotext',
|
||||
):
|
||||
"""Compare text in PDFs."""
|
||||
with open(pdf1, 'rb') as f1, open(pdf2, 'rb') as f2:
|
||||
with pdf1.open('rb') as f1, pdf2.open('rb') as f2:
|
||||
text1 = run(
|
||||
['pdftotext', '-layout', '-', '-'],
|
||||
stdin=f1,
|
||||
|
||||
+10
-9
@@ -13,13 +13,14 @@ import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
# pylint: disable=logging-not-lazy
|
||||
|
||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
||||
script_dir = Path(os.path.realpath(__file__)).parent
|
||||
timestamp = time.strftime("%Y-%m-%d-%H%M_")
|
||||
log_file = script_dir + '/' + timestamp + 'ocrmypdf.log'
|
||||
log_file = script_dir / (timestamp + 'ocrmypdf.log')
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s %(message)s',
|
||||
@@ -33,10 +34,10 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||
logging.info(dir_name)
|
||||
os.chdir(dir_name)
|
||||
for filename in file_list:
|
||||
file_stem, file_ext = os.path.splitext(filename)
|
||||
file_stem, file_ext = Path(filename).stem, Path(filename).suffix
|
||||
if file_ext != '.pdf':
|
||||
continue
|
||||
full_path = os.path.join(dir_name, filename)
|
||||
full_path = Path(dir_name, filename)
|
||||
timestamp_ocr = time.strftime("%Y-%m-%d-%H%M_OCR_")
|
||||
filename_ocr = timestamp_ocr + file_stem + '.pdf'
|
||||
# create string for pdf processing
|
||||
@@ -52,10 +53,10 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||
'-',
|
||||
]
|
||||
logging.info(cmd)
|
||||
full_path_ocr = os.path.join(dir_name, filename_ocr)
|
||||
full_path_ocr = Path(dir_name, filename_ocr)
|
||||
with (
|
||||
open(filename, 'rb') as input_file,
|
||||
open(full_path_ocr, 'wb') as output_file,
|
||||
Path(filename).open('rb') as input_file,
|
||||
full_path_ocr.open('wb') as output_file,
|
||||
):
|
||||
proc = subprocess.run(
|
||||
cmd,
|
||||
@@ -67,8 +68,8 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||
errors='ignore',
|
||||
)
|
||||
logging.info(proc.stderr)
|
||||
os.chmod(full_path_ocr, 0o664)
|
||||
os.chmod(full_path, 0o664)
|
||||
full_path_ocr.chmod(0o664)
|
||||
full_path.chmod(0o664)
|
||||
full_path_ocr_archive = sys.argv[2]
|
||||
full_path_archive = sys.argv[2] + '/no_ocr'
|
||||
shutil.move(full_path_ocr, full_path_ocr_archive)
|
||||
|
||||
+2
-2
@@ -13,7 +13,7 @@ import logging
|
||||
import shutil
|
||||
import sys
|
||||
import time
|
||||
from enum import Enum
|
||||
from enum import StrEnum
|
||||
from pathlib import Path
|
||||
from typing import Annotated, Any
|
||||
|
||||
@@ -35,7 +35,7 @@ app = cyclopts.App(name="ocrmypdf-watcher")
|
||||
log = logging.getLogger('ocrmypdf-watcher')
|
||||
|
||||
|
||||
class LoggingLevelEnum(str, Enum):
|
||||
class LoggingLevelEnum(StrEnum):
|
||||
"""Enum for logging levels."""
|
||||
|
||||
DEBUG = "DEBUG"
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# prek pre-commit configuration — https://prek.j178.dev
|
||||
#
|
||||
# The local/system hooks below invoke the project's OWN pinned tools (ruff/mypy
|
||||
# from uv.lock) and mirror .github/workflows/build.yml's lint job exactly, so
|
||||
# they can never drift from CI's versions or rules. prek installs nothing of
|
||||
# its own for them — "system" language just execs whatever `uv run` resolves.
|
||||
#
|
||||
# The pre-commit/pre-commit-hooks repo hooks below are generic file checks with
|
||||
# no project-local tool equivalent, so they're kept as a normal (non-local) repo.
|
||||
#
|
||||
# Run all checks manually: `uv run prek run --all-files`
|
||||
# Install the git hooks: `uv run prek install`
|
||||
|
||||
default_install_hook_types = ["pre-commit", "pre-push"]
|
||||
default_stages = ["pre-commit"]
|
||||
|
||||
[[repos]]
|
||||
repo = "https://github.com/pre-commit/pre-commit-hooks"
|
||||
rev = "v4.4.0"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "check-case-conflict"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "check-merge-conflict"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "check-toml"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "check-yaml"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "debug-statements"
|
||||
|
||||
[[repos]]
|
||||
repo = "local"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "ruff-format"
|
||||
name = "ruff format (check)"
|
||||
language = "system"
|
||||
entry = "uv run ruff format --check ."
|
||||
types = ["python"]
|
||||
pass_filenames = false
|
||||
require_serial = true
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "ruff-check"
|
||||
name = "ruff check"
|
||||
language = "system"
|
||||
entry = "uv run ruff check ."
|
||||
types = ["python"]
|
||||
pass_filenames = false
|
||||
require_serial = true
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "mypy"
|
||||
name = "mypy"
|
||||
language = "system"
|
||||
entry = "uv run mypy src/ocrmypdf"
|
||||
types = ["python"]
|
||||
pass_filenames = false
|
||||
require_serial = true
|
||||
+17
-5
@@ -6,17 +6,16 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "ocrmypdf"
|
||||
version = "17.4.2"
|
||||
version = "17.9.0"
|
||||
description = "OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched"
|
||||
readme = "README.md"
|
||||
license = "MPL-2.0"
|
||||
requires-python = ">=3.11"
|
||||
dependencies = [
|
||||
"deprecation>=2.1.0",
|
||||
"fpdf2>=2.8.0",
|
||||
"img2pdf>=0.5",
|
||||
"packaging>=20",
|
||||
"pdfminer.six>=20220319",
|
||||
"pdfminer.six>=20260107", # fixes parsing of tokens split across the read buffer/streams (gh #1361)
|
||||
"pi-heif", # Heif image format - maintainers: if this is removed, it will NOT break
|
||||
"pikepdf>=10",
|
||||
"Pillow>=10.0.1",
|
||||
@@ -24,6 +23,7 @@ dependencies = [
|
||||
"pydantic>=2.12.5",
|
||||
"pypdfium2>=5.0.0",
|
||||
"rich>=13",
|
||||
"typing-extensions>=4.12; python_version < '3.13'",
|
||||
"uharfbuzz>=0.53.2",
|
||||
]
|
||||
authors = [{ name = "James R. Barlow", email = "james@purplerock.ca" }]
|
||||
@@ -98,18 +98,28 @@ filterwarnings = [
|
||||
]
|
||||
|
||||
[tool.mypy]
|
||||
check_untyped_defs = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = [
|
||||
'pluggy',
|
||||
'img2pdf',
|
||||
'pdfminer.*',
|
||||
'reportlab.*',
|
||||
'fitz',
|
||||
'libxmp.utils',
|
||||
'pypdfium2',
|
||||
'uharfbuzz',
|
||||
'pi_heif',
|
||||
]
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
# Test functions are not required to annotate their return type (almost
|
||||
# always None); it's a low-value hint that would just be noise here.
|
||||
module = 'tests.*'
|
||||
disallow_untyped_defs = false
|
||||
disallow_incomplete_defs = false
|
||||
|
||||
[tool.ruff]
|
||||
target-version = "py311"
|
||||
exclude = ["src/ocrmypdf/_version.py"] # Autogenerated
|
||||
@@ -125,6 +135,7 @@ exclude = ["src/ocrmypdf/_version.py"] # Autogenerated
|
||||
"SIM", # simplify
|
||||
"B", # flake8-bugbear
|
||||
"ICN", # flake8-import-conventions
|
||||
"PTH", # flake8-use-pathlib
|
||||
]
|
||||
ignore = [
|
||||
"B028", # warning with no explicit stacklevel
|
||||
@@ -158,6 +169,8 @@ quote-style = "preserve"
|
||||
# Developer-only tools - use `uv sync --group <name>`
|
||||
dev = [
|
||||
"mypy>=1.13.0",
|
||||
"ruff>=0.14.11",
|
||||
"prek>=0.4.8",
|
||||
"ipykernel>=6.29.5",
|
||||
"reportlab>=4.4.4",
|
||||
"cyclopts>=4.5.1",
|
||||
@@ -174,7 +187,6 @@ test = [
|
||||
"python-xmp-toolkit==2.0.1", # also requires apt-get install libexempi3
|
||||
"reportlab>=3.6.8",
|
||||
# Type stubs for testing
|
||||
"types-Pillow",
|
||||
"types-humanfriendly",
|
||||
# Extended test capabilities (merged from extended_test)
|
||||
"pymupdf>=1.24.14",
|
||||
|
||||
@@ -19,6 +19,7 @@ from ocrmypdf._version import __version__
|
||||
from ocrmypdf.api import (
|
||||
Verbosity,
|
||||
configure_logging,
|
||||
configure_stdout_protection,
|
||||
ocr,
|
||||
)
|
||||
from ocrmypdf.exceptions import (
|
||||
@@ -53,6 +54,7 @@ __all__ = [
|
||||
'BoundingBox',
|
||||
'configure_debug_logging',
|
||||
'configure_logging',
|
||||
'configure_stdout_protection',
|
||||
'DpiError',
|
||||
'EncryptedPdfError',
|
||||
'Executor',
|
||||
|
||||
@@ -16,7 +16,7 @@ from contextlib import suppress
|
||||
from ocrmypdf import __version__
|
||||
from ocrmypdf._pipelines.ocr import run_pipeline_cli
|
||||
from ocrmypdf._validation import check_options
|
||||
from ocrmypdf.api import Verbosity, configure_logging
|
||||
from ocrmypdf.api import Verbosity, configure_logging, configure_stdout_protection
|
||||
from ocrmypdf.cli import get_options_and_plugins
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
@@ -39,12 +39,17 @@ def sigbus(*args):
|
||||
|
||||
def run(args=None):
|
||||
"""Run the ocrmypdf command line interface."""
|
||||
# Protect the real stdout before loading plugins or starting any worker
|
||||
# processes/threads, so that only our final PDF output can reach it and
|
||||
# stray writes from plugins or libraries are diverted to stderr.
|
||||
configure_stdout_protection()
|
||||
|
||||
options, plugin_manager = get_options_and_plugins(args=args)
|
||||
|
||||
with suppress(AttributeError, PermissionError):
|
||||
os.nice(5)
|
||||
|
||||
verbosity = options.verbose
|
||||
verbosity = Verbosity(options.verbose)
|
||||
if not os.isatty(sys.stderr.fileno()):
|
||||
options.progress_bar = False
|
||||
if options.quiet:
|
||||
|
||||
@@ -8,7 +8,7 @@ from __future__ import annotations
|
||||
import threading
|
||||
from abc import ABC, abstractmethod
|
||||
from collections.abc import Callable, Iterable
|
||||
from typing import Any, TypeVar
|
||||
from typing import Any, TypeVar, cast
|
||||
|
||||
from ocrmypdf._progressbar import NullProgressBar, ProgressBar
|
||||
|
||||
@@ -72,7 +72,10 @@ class Executor(ABC):
|
||||
if not task_finished:
|
||||
task_finished = _task_finished_noop
|
||||
if not task:
|
||||
task = _task_noop
|
||||
# _task_noop always returns None, but T is unbound here (it's
|
||||
# only meaningful when a real task is supplied); task_finished's
|
||||
# own no-op default accepts Any, so this is safe.
|
||||
task = cast('Callable[..., T]', _task_noop)
|
||||
|
||||
with self.pool_lock:
|
||||
self._execute(
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Probe helper for external executables.
|
||||
|
||||
Each ``ocrmypdf._exec.<tool>`` module describes its external program with a
|
||||
module-level :class:`ToolProbe` and delegates ``version()`` / ``available()``
|
||||
to it. This separates the "is the tool installed and suitable?" question
|
||||
(probing) from the "run the tool" question (execution). Work functions stay
|
||||
as pure module-level functions so they are trivially picklable for use in
|
||||
subprocess workers.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolProbe:
|
||||
"""Describes how to detect an external executable and its version.
|
||||
|
||||
Attributes:
|
||||
program: The program name as it appears on PATH (or a full path).
|
||||
version_arg: The argument that elicits a version string.
|
||||
version_regex: A regex with a capturing group that extracts the
|
||||
version from the program's output.
|
||||
version_cls: A :class:`packaging.version.Version` subclass, used for
|
||||
tools with non-standard version strings (e.g. Tesseract).
|
||||
env: Optional environment overrides applied when probing the version.
|
||||
also_catch: Additional exception types that should be treated as
|
||||
"not available" by :meth:`available`. :class:`OSError` is useful
|
||||
for tools like verapdf whose launcher may fail with non-standard
|
||||
errors when the JVM is missing.
|
||||
"""
|
||||
|
||||
program: str
|
||||
version_arg: str = '--version'
|
||||
version_regex: str = r'(\d+(\.\d+)*)'
|
||||
version_cls: type[Version] = Version
|
||||
env: Mapping[str, str] | None = None
|
||||
also_catch: tuple[type[BaseException], ...] = ()
|
||||
|
||||
def version(self) -> Version:
|
||||
"""Return the installed version of the program.
|
||||
|
||||
Raises:
|
||||
MissingDependencyError: if the program cannot be found or its
|
||||
version string cannot be parsed.
|
||||
"""
|
||||
raw = get_version(
|
||||
self.program,
|
||||
version_arg=self.version_arg,
|
||||
regex=self.version_regex,
|
||||
env=self.env,
|
||||
)
|
||||
return self.version_cls(raw)
|
||||
|
||||
def available(self) -> bool:
|
||||
"""Return whether a usable version of the program is installed."""
|
||||
try:
|
||||
self.version()
|
||||
except MissingDependencyError:
|
||||
return False
|
||||
except self.also_catch:
|
||||
return False
|
||||
return True
|
||||
@@ -16,6 +16,7 @@ from subprocess import PIPE, CalledProcessError
|
||||
from packaging.version import Version
|
||||
from PIL import Image, UnidentifiedImageError
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import (
|
||||
ColorConversionNeededError,
|
||||
InputFileError,
|
||||
@@ -23,7 +24,7 @@ from ocrmypdf.exceptions import (
|
||||
)
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
||||
from ocrmypdf.subprocess import run, run_polling_stderr
|
||||
|
||||
COLOR_CONVERSION_STRATEGIES = frozenset(
|
||||
[
|
||||
@@ -69,11 +70,19 @@ class DuplicateFilter(logging.Filter):
|
||||
return True
|
||||
|
||||
|
||||
log.addFilter(DuplicateFilter(log))
|
||||
PROBE = ToolProbe(program=GS)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version(GS))
|
||||
def _ensure_log_filter_installed() -> None:
|
||||
"""Idempotently attach the duplicate-suppressing filter to the GS logger.
|
||||
|
||||
Called at the top of each work function so the filter is present in the
|
||||
main process *and* in any subprocess worker that calls Ghostscript.
|
||||
"""
|
||||
if not any(isinstance(f, DuplicateFilter) for f in log.filters):
|
||||
log.addFilter(DuplicateFilter(log))
|
||||
|
||||
|
||||
def _gs_error_reported(stream) -> bool:
|
||||
@@ -96,8 +105,8 @@ def _gs_devicen_reported(stream) -> bool:
|
||||
|
||||
|
||||
def rasterize_pdf(
|
||||
input_file: os.PathLike,
|
||||
output_file: os.PathLike,
|
||||
input_file: Path,
|
||||
output_file: Path,
|
||||
*,
|
||||
raster_device: GhostscriptRasterDevice,
|
||||
raster_dpi: Resolution,
|
||||
@@ -123,6 +132,7 @@ def rasterize_pdf(
|
||||
use_cropbox: If True, rasterize the CropBox instead of MediaBox.
|
||||
Default is False (use MediaBox).
|
||||
"""
|
||||
_ensure_log_filter_installed()
|
||||
raster_dpi = raster_dpi.round(6)
|
||||
if not page_dpi:
|
||||
page_dpi = raster_dpi
|
||||
@@ -140,6 +150,19 @@ def rasterize_pdf(
|
||||
else:
|
||||
effective_dpi = raster_dpi
|
||||
|
||||
# Anti-alias text and vector graphics when rendering to a contone device.
|
||||
# Ghostscript 10.x renders aliased glyphs that OCR frequently misreads as
|
||||
# extra word breaks; anti-aliasing empirically improves OCR accuracy on the
|
||||
# Ghostscript path, especially for small fonts at moderate DPI (#1439).
|
||||
# The 1-bit mono devices do not accept alpha bits (older Ghostscript
|
||||
# rejects them) and pngmonod performs its own anti-aliased downscaling.
|
||||
mono_devices = (GhostscriptRasterDevice.PNGMONO, GhostscriptRasterDevice.PNGMONOD)
|
||||
antialias_args = (
|
||||
[]
|
||||
if raster_device in mono_devices
|
||||
else ['-dTextAlphaBits=4', '-dGraphicsAlphaBits=4']
|
||||
)
|
||||
|
||||
args_gs = (
|
||||
[
|
||||
GS,
|
||||
@@ -152,6 +175,7 @@ def rasterize_pdf(
|
||||
f'-dLastPage={pageno}',
|
||||
f'-r{effective_dpi.x:f}x{effective_dpi.y:f}',
|
||||
]
|
||||
+ antialias_args
|
||||
+ (['-dUseCropBox'] if use_cropbox else [])
|
||||
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
||||
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||
@@ -185,6 +209,7 @@ def rasterize_pdf(
|
||||
)
|
||||
|
||||
try:
|
||||
im: Image.Image
|
||||
with Image.open(output_file) as im:
|
||||
if needs_low_dpi_resize:
|
||||
# Resize to the dimensions that would have resulted from the
|
||||
@@ -264,15 +289,18 @@ class GhostscriptFollower:
|
||||
|
||||
def generate_pdfa(
|
||||
pdf_pages,
|
||||
output_file: os.PathLike,
|
||||
output_file: Path,
|
||||
*,
|
||||
compression: str,
|
||||
color_conversion_strategy: str,
|
||||
jpeg_quality: int | None = None,
|
||||
jpeg_maxdpi: int | None = None,
|
||||
pdf_version: str = '1.5',
|
||||
pdfa_part: str = '2',
|
||||
progressbar_class=None,
|
||||
stop_on_error: bool = False,
|
||||
):
|
||||
_ensure_log_filter_installed()
|
||||
# Ghostscript's compression is all or nothing. We can either force all images
|
||||
# to JPEG, force all to Flate/PNG, or let it decide how to encode the images.
|
||||
# In most case it's best to let it decide.
|
||||
@@ -286,6 +314,11 @@ def generate_pdfa(
|
||||
]
|
||||
elif compression == 'lossless':
|
||||
compression_args = [
|
||||
# Re-encoding an existing JPEG with a lossless codec only inflates
|
||||
# its size: the lossy data is already baked in, so there is nothing
|
||||
# to gain. Pass JPEGs through untouched and apply lossless (Flate)
|
||||
# encoding only to images that are not already JPEG.
|
||||
"-dPassThroughJPEGImages=true",
|
||||
"-dAutoFilterColorImages=false",
|
||||
"-dColorImageFilter=/FlateEncode",
|
||||
"-dAutoFilterGrayImages=false",
|
||||
@@ -307,6 +340,35 @@ def generate_pdfa(
|
||||
# Windows has lots of fatal "permission denied" errors
|
||||
stop_on_error = False
|
||||
|
||||
# `-dJPEGQ=N` tells Ghostscript to use a JPEG quality of N, IF it decides
|
||||
# to transcode an image to JPEG. When there are existing JPEG images,
|
||||
# Ghostscript uses passthrough mode, so the quality level is not changed.
|
||||
# OCRmyPDF's optimizer separately uses the `--jpeg-quality` command line
|
||||
# option to potentially re-encode JPEG images, regardless of whether
|
||||
# Ghostscript decided to transcode them to JPEG or not.
|
||||
# `jpeg_quality=0` is meaningful to Ghostscript (maximum compression), so
|
||||
# only fall back to the default when the value is None.
|
||||
effective_jpeg_quality = jpeg_quality if jpeg_quality is not None else 95
|
||||
|
||||
# Downsampling images is a blunt-force way to reduce file size and almost
|
||||
# always degrades quality more than lowering JPEG quality at the original
|
||||
# resolution. We expose this for users with very specific needs (e.g.
|
||||
# producing very small files for screen-only viewing); the optimizer is
|
||||
# usually a better choice.
|
||||
downsample_args: list[str] = []
|
||||
if jpeg_maxdpi is not None:
|
||||
downsample_args = [
|
||||
"-dDownsampleColorImages=true",
|
||||
"-dColorImageDownsampleThreshold=1.0",
|
||||
"-dDownsampleGrayImages=true",
|
||||
"-dGrayImageDownsampleThreshold=1.0",
|
||||
"-dDownsampleMonoImages=true",
|
||||
"-dMonoImageDownsampleThreshold=1.0",
|
||||
f"-dColorImageResolution={jpeg_maxdpi}",
|
||||
f"-dGrayImageResolution={jpeg_maxdpi}",
|
||||
f"-dMonoImageResolution={jpeg_maxdpi}",
|
||||
]
|
||||
|
||||
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
||||
# is set; see:
|
||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699392
|
||||
@@ -323,8 +385,9 @@ def generate_pdfa(
|
||||
]
|
||||
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||
+ compression_args
|
||||
+ downsample_args
|
||||
+ [
|
||||
"-dJPEGQ=95",
|
||||
f"-dJPEGQ={effective_jpeg_quality}", # See note above on JPEG quality
|
||||
"-dSubsetFonts=false", # Prevents GS from messing up some encodings
|
||||
f"-dPDFA={pdfa_part}",
|
||||
"-dPDFACompatibilityPolicy=1",
|
||||
@@ -362,4 +425,8 @@ def generate_pdfa(
|
||||
for part in stderr.split('****'):
|
||||
log.error(part)
|
||||
if _gs_devicen_reported(stderr):
|
||||
raise ColorConversionNeededError()
|
||||
# Ghostscript could not normalize the DeviceN colorspace for PDF/A,
|
||||
# even if the user requested a conversion strategy. The output is
|
||||
# liable to render blank in some viewers, so raise regardless of the
|
||||
# strategy and tailor the guidance to what was attempted.
|
||||
raise ColorConversionNeededError(color_conversion_strategy)
|
||||
|
||||
@@ -9,21 +9,23 @@ from subprocess import PIPE, CalledProcessError
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
_PROBE = ToolProbe(program='jbig2', version_regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
try:
|
||||
version = get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
return _PROBE.version()
|
||||
except CalledProcessError as e:
|
||||
# TeX Live for Windows provides an incompatible jbig2.EXE which may
|
||||
# be on the PATH.
|
||||
raise MissingDependencyError('jbig2enc') from e
|
||||
return Version(version)
|
||||
|
||||
|
||||
def available():
|
||||
def available() -> bool:
|
||||
try:
|
||||
version()
|
||||
except MissingDependencyError:
|
||||
@@ -33,7 +35,7 @@ def available():
|
||||
|
||||
def convert_single(cwd, infile, outfile, threshold):
|
||||
args = ['jbig2', '--pdf', '-t', str(threshold), infile]
|
||||
with open(outfile, 'wb') as fstdout:
|
||||
with outfile.open('wb') as fstdout:
|
||||
proc = run(args, cwd=cwd, stdout=fstdout, stderr=PIPE)
|
||||
proc.check_returncode()
|
||||
return proc
|
||||
|
||||
@@ -8,22 +8,12 @@ from __future__ import annotations
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE
|
||||
|
||||
from packaging.version import Version
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('pngquant', regex=r'(\d+(\.\d+)*).*'))
|
||||
|
||||
|
||||
def available():
|
||||
try:
|
||||
version()
|
||||
except MissingDependencyError:
|
||||
return False
|
||||
return True
|
||||
PROBE = ToolProbe(program='pngquant', version_regex=r'(\d+(\.\d+)*).*')
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max: int):
|
||||
@@ -35,7 +25,7 @@ def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max:
|
||||
quality_min: Minimum quality to use
|
||||
quality_max: Maximum quality to use
|
||||
"""
|
||||
with open(input_file, 'rb') as input_stream:
|
||||
with input_file.open('rb') as input_stream:
|
||||
args = [
|
||||
'pngquant',
|
||||
'--force',
|
||||
|
||||
@@ -17,13 +17,14 @@ from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import (
|
||||
MissingDependencyError,
|
||||
SubprocessOutputError,
|
||||
TesseractConfigError,
|
||||
)
|
||||
from ocrmypdf.pluginspec import OrientationConfidence
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -115,8 +116,13 @@ class TesseractVersion(Version):
|
||||
)
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return TesseractVersion(get_version('tesseract', regex=r'tesseract\s(.+)'))
|
||||
PROBE = ToolProbe(
|
||||
program='tesseract',
|
||||
version_regex=r'tesseract\s(.+)',
|
||||
version_cls=TesseractVersion,
|
||||
)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def has_thresholding() -> bool:
|
||||
@@ -287,12 +293,14 @@ def tesseract_log_output(stream: bytes) -> None:
|
||||
|
||||
lines = text.splitlines()
|
||||
for line in lines:
|
||||
if line.startswith(
|
||||
("Tesseract Open Source", "Warning in pixReadMem")
|
||||
):
|
||||
if line.startswith(("Tesseract Open Source", "Warning in pixReadMem")):
|
||||
continue
|
||||
elif 'diacritics' in line:
|
||||
tlog.warning("lots of diacritics - possibly poor OCR")
|
||||
# Surface the raw Tesseract message at debug level so users can see
|
||||
# exactly what Tesseract reported (e.g. the affected count) without
|
||||
# losing the interpreted hint above (#1566).
|
||||
tlog.debug(line.strip())
|
||||
elif line.startswith('OSD: Weak margin'):
|
||||
tlog.warning("unsure about page orientation")
|
||||
elif 'Error in pixScanForForeground' in line:
|
||||
@@ -309,6 +317,23 @@ def tesseract_log_output(stream: bytes) -> None:
|
||||
tlog.warning(line.strip())
|
||||
elif 'read_params_file' in line.lower():
|
||||
tlog.error(line.strip())
|
||||
# Tesseract emits "read_params_file: Can't open <name>" when it
|
||||
# cannot locate a config file (e.g. 'hocr', 'txt') in its
|
||||
# tessdata configs/ directory, then exits 0 without producing
|
||||
# the requested output. Promote to a hard error so the user
|
||||
# sees the root cause instead of a downstream FileNotFoundError.
|
||||
if "Can't open" in line:
|
||||
missing = line.split("Can't open", 1)[1].strip()
|
||||
else:
|
||||
missing = line.strip()
|
||||
raise TesseractConfigError(
|
||||
f"Tesseract cannot open its config file '{missing}'. "
|
||||
"This usually means Tesseract is installed but its config "
|
||||
"files are missing from the tessdata configs/ directory. "
|
||||
"On Debian/Ubuntu, ensure the 'tesseract-ocr' package is "
|
||||
"fully installed. If you set TESSDATA_PREFIX, verify its "
|
||||
"configs/ subdirectory contains the required files."
|
||||
)
|
||||
else:
|
||||
tlog.info(line.strip())
|
||||
|
||||
@@ -389,6 +414,12 @@ def generate_hocr(
|
||||
raise SubprocessOutputError() from e
|
||||
else:
|
||||
tesseract_log_output(stdout)
|
||||
if not output_hocr.exists():
|
||||
raise SubprocessOutputError(
|
||||
"Tesseract exited successfully but did not produce the "
|
||||
f"expected hOCR output at {output_hocr}. Tesseract output:\n"
|
||||
+ (stdout.decode(errors='replace') if stdout else '(empty)')
|
||||
)
|
||||
# The sidecar text file will get the suffix .txt; rename it to
|
||||
# whatever caller wants it named
|
||||
with suppress(FileNotFoundError):
|
||||
@@ -457,6 +488,12 @@ def generate_pdf(
|
||||
stdout = p.stdout
|
||||
with suppress(FileNotFoundError):
|
||||
prefix.with_suffix('.txt').replace(output_text)
|
||||
if not output_pdf.exists():
|
||||
raise SubprocessOutputError(
|
||||
"Tesseract exited successfully but did not produce the "
|
||||
f"expected PDF output at {output_pdf}. Tesseract output:\n"
|
||||
+ (stdout.decode(errors='replace') if stdout else '(empty)')
|
||||
)
|
||||
except TimeoutExpired:
|
||||
page_timedout(timeout)
|
||||
use_skip_page(output_pdf, output_text)
|
||||
|
||||
@@ -14,11 +14,11 @@ from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
from packaging.version import Version
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import SubprocessOutputError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
# unpaper documentation:
|
||||
# https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md
|
||||
@@ -46,8 +46,9 @@ class UnpaperImageTooLargeError(Exception):
|
||||
super().__init__(self.message)
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('unpaper', regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)'))
|
||||
PROBE = ToolProbe(program='unpaper', version_regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)')
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
@contextmanager
|
||||
|
||||
@@ -11,10 +11,9 @@ from pathlib import Path
|
||||
from subprocess import PIPE
|
||||
from typing import NamedTuple
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -27,18 +26,13 @@ class ValidationResult(NamedTuple):
|
||||
message: str
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
"""Get verapdf version."""
|
||||
return Version(get_version('verapdf', regex=r'veraPDF (\d+(\.\d+)*)'))
|
||||
|
||||
|
||||
def available() -> bool:
|
||||
"""Check if verapdf is available."""
|
||||
try:
|
||||
version()
|
||||
except (MissingDependencyError, OSError):
|
||||
return False
|
||||
return True
|
||||
PROBE = ToolProbe(
|
||||
program='verapdf',
|
||||
version_regex=r'veraPDF (\d+(\.\d+)*)',
|
||||
also_catch=(OSError,),
|
||||
)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def output_type_to_flavour(output_type: str) -> str:
|
||||
|
||||
+119
-8
@@ -6,11 +6,12 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from collections.abc import Collection
|
||||
from contextlib import suppress
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import TYPE_CHECKING, cast
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from ocrmypdf.hocrtransform import OcrElement
|
||||
@@ -18,6 +19,7 @@ if TYPE_CHECKING:
|
||||
from pikepdf import (
|
||||
Dictionary,
|
||||
Name,
|
||||
Object,
|
||||
Operator,
|
||||
Page,
|
||||
Pdf,
|
||||
@@ -181,9 +183,13 @@ def strip_invisible_text(pdf: Pdf, page: Page):
|
||||
render_mode_stack = []
|
||||
text_objects = []
|
||||
|
||||
for operands, operator in parse_content_stream(page, ''):
|
||||
for instruction in parse_content_stream(page, ''):
|
||||
operands, operator = instruction.operands, instruction.operator
|
||||
if operator == Operator('Tr'):
|
||||
render_mode = operands[0]
|
||||
# operands[0] is already a plain int under pikepdf's default
|
||||
# (implicit) conversion mode, or a pikepdf.Object under explicit
|
||||
# conversion mode; int() handles both.
|
||||
render_mode = int(operands[0])
|
||||
|
||||
if operator == Operator('q'):
|
||||
render_mode_stack.append(render_mode)
|
||||
@@ -207,10 +213,103 @@ def strip_invisible_text(pdf: Pdf, page: Page):
|
||||
stream.extend(text_objects)
|
||||
text_objects.clear()
|
||||
|
||||
content_stream = unparse_content_stream(stream)
|
||||
# pikepdf's Collection[...] parameter doesn't structurally match our
|
||||
# _ObjectList-based tuples even though it works fine at runtime.
|
||||
content_stream = unparse_content_stream(
|
||||
cast('list[tuple[Collection[Object], Operator]]', stream)
|
||||
)
|
||||
page.Contents = Stream(pdf, content_stream)
|
||||
|
||||
|
||||
def discard_text_search_index(pdf: Pdf) -> bool:
|
||||
"""Discard an embedded Adobe full-text search index from the catalog.
|
||||
|
||||
Adobe Acrobat can embed a full-text search index in the document catalog at
|
||||
``/Root/PieceInfo/SearchIndex``. It is built from the page text, and only
|
||||
Acrobat reads it; other viewers ignore it and search the text on the fly.
|
||||
Any change to the PDF invalidates the index, so once OCRmyPDF rewrites the
|
||||
document (editing the text layer, rasterizing, optimizing) a retained index
|
||||
would be stale and return incorrect search results in Acrobat. We cannot
|
||||
update this vendor-private data, so we discard it; modern viewers rebuild a
|
||||
search index on demand. Returns True if the catalog was modified.
|
||||
"""
|
||||
try:
|
||||
pieceinfo = pdf.Root.get(Name.PieceInfo)
|
||||
if not isinstance(pieceinfo, Dictionary) or Name.SearchIndex not in pieceinfo:
|
||||
return False
|
||||
del pieceinfo[Name.SearchIndex]
|
||||
log.debug(
|
||||
"Discarded embedded text search index "
|
||||
"(/Root/PieceInfo/SearchIndex) because the PDF was rewritten; "
|
||||
"it would otherwise be stale."
|
||||
)
|
||||
# Drop an empty PieceInfo rather than leave a husk behind.
|
||||
if len(pieceinfo) == 0:
|
||||
del pdf.Root.PieceInfo
|
||||
return True
|
||||
except (KeyError, TypeError, AttributeError):
|
||||
return False
|
||||
|
||||
|
||||
def discard_page_thumbnails(pdf: Pdf) -> int:
|
||||
"""Discard embedded per-page thumbnail images.
|
||||
|
||||
A page object may carry an optional ``/Thumb`` image XObject — a miniature
|
||||
rendering of the page (ISO 32000-2, 12.3.4). It is only a navigation aid and
|
||||
modern viewers generate page thumbnails on demand. OCRmyPDF alters page
|
||||
appearance (deskew, clean, rasterize, re-render) and plugins may edit pages
|
||||
arbitrarily, so any retained thumbnail would be stale and misrepresent its
|
||||
page. We discard them; viewers rebuild thumbnails as needed. Returns the
|
||||
number of thumbnails removed.
|
||||
"""
|
||||
removed = 0
|
||||
for page in pdf.pages:
|
||||
pageobj = page.obj
|
||||
if Name.Thumb in pageobj:
|
||||
del pageobj[Name.Thumb]
|
||||
removed += 1
|
||||
if removed:
|
||||
log.debug(
|
||||
"Discarded %d embedded page thumbnail(s) (/Thumb) because the PDF "
|
||||
"was rewritten; they would otherwise be stale.",
|
||||
removed,
|
||||
)
|
||||
return removed
|
||||
|
||||
|
||||
def discard_structure_tree(pdf: Pdf) -> bool:
|
||||
"""Discard the logical structure (tagged-PDF) tree from the document.
|
||||
|
||||
The structure tree (``/Root/StructTreeRoot``, ``/Root/MarkInfo``) maps
|
||||
marked content in the page content streams to semantic elements via MCIDs.
|
||||
When OCRmyPDF rasterizes pages (force) or strips and rewrites the text layer
|
||||
(redo), those MCIDs are destroyed or renumbered, leaving the tree dangling
|
||||
and inconsistent with the new content. We cannot rebuild it to match, so we
|
||||
discard it; the page-level ``/StructParents`` keys go too. Returns True if
|
||||
the catalog was modified.
|
||||
"""
|
||||
modified = False
|
||||
try:
|
||||
if Name.StructTreeRoot in pdf.Root:
|
||||
del pdf.Root.StructTreeRoot
|
||||
modified = True
|
||||
if Name.MarkInfo in pdf.Root:
|
||||
del pdf.Root.MarkInfo
|
||||
modified = True
|
||||
for page in pdf.pages:
|
||||
if Name.StructParents in page.obj:
|
||||
del page.obj[Name.StructParents]
|
||||
modified = True
|
||||
except (KeyError, TypeError, AttributeError):
|
||||
return modified
|
||||
if modified:
|
||||
log.debug(
|
||||
"Discarded the logical structure tree (/Root/StructTreeRoot) "
|
||||
"because the PDF was re-OCR'd; it would otherwise be stale."
|
||||
)
|
||||
return modified
|
||||
|
||||
|
||||
class OcrGrafter:
|
||||
"""Manages grafting text-only PDFs onto regular PDFs."""
|
||||
|
||||
@@ -253,6 +352,14 @@ class OcrGrafter:
|
||||
ocr_tree: OCR tree for fpdf2 renderer.
|
||||
autorotate_correction: Orientation correction in degrees (0, 90, 180, 270).
|
||||
"""
|
||||
if self.context.options.mode == ProcessingMode.strip_text:
|
||||
# Strip mode: remove the invisible OCR text layer in place without
|
||||
# rasterizing or grafting anything. Honor --pages if specified.
|
||||
options = self.context.options
|
||||
if not options.pages or pageno in options.pages:
|
||||
strip_invisible_text(self.pdf_base, self.pdf_base.pages[pageno])
|
||||
return
|
||||
|
||||
if ocr_output and ocr_tree:
|
||||
raise ValueError(
|
||||
'Cannot specify both ocr_output and ocr_tree for fpdf2 renderer'
|
||||
@@ -319,9 +426,9 @@ class OcrGrafter:
|
||||
|
||||
def finalize(self):
|
||||
# Can have hocr OR parsed pages OR neither (no OCR), but not both
|
||||
assert not (
|
||||
self.fpdf2_hocr_pages and self.fpdf2_parsed_pages
|
||||
), "Can't have both hocr and ocrtree pages"
|
||||
assert not (self.fpdf2_hocr_pages and self.fpdf2_parsed_pages), (
|
||||
"Can't have both hocr and ocrtree pages"
|
||||
)
|
||||
|
||||
if self.fpdf2_hocr_pages:
|
||||
# Render all pages with fpdf2, then graft
|
||||
@@ -331,11 +438,15 @@ class OcrGrafter:
|
||||
if self.fpdf2_parsed_pages:
|
||||
self._render_and_graft_fpdf2_pages()
|
||||
|
||||
discard_text_search_index(self.pdf_base)
|
||||
discard_page_thumbnails(self.pdf_base)
|
||||
if self.context.options.mode in (ProcessingMode.force, ProcessingMode.redo):
|
||||
discard_structure_tree(self.pdf_base)
|
||||
self.pdf_base.save(self.output_file)
|
||||
self.pdf_base.close()
|
||||
return self.output_file
|
||||
|
||||
def _parse_hocr_pages(self):
|
||||
def _parse_hocr_pages(self) -> list[Fpdf2ParsedPage]:
|
||||
"""Render all pages to multi-page PDF with shared fonts, then graft."""
|
||||
from ocrmypdf.hocrtransform.hocr_parser import HocrParser
|
||||
|
||||
|
||||
@@ -7,7 +7,6 @@ from __future__ import annotations
|
||||
|
||||
import datetime as dt
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
@@ -88,8 +87,12 @@ def repair_docinfo_nuls(pdf):
|
||||
if isinstance(v, str) and b'\x00' in bytes(v):
|
||||
pdf.docinfo[k] = bytes(v).replace(b'\x00', b'')
|
||||
modified = True
|
||||
except TypeError:
|
||||
# TypeError can also be raised if dictionary items are unexpected types
|
||||
except (TypeError, UnicodeDecodeError):
|
||||
# TypeError: DocumentInfo is not a dictionary, or its items are
|
||||
# unexpected types.
|
||||
# UnicodeDecodeError: a DocumentInfo key or value contains bytes that
|
||||
# are not valid PDFDocEncoding/UTF-16, e.g. a Latin-1 /Name key such as
|
||||
# /Saks#e5r. Older pikepdf raised while iterating such a block (#1540).
|
||||
log.error("File contains a malformed DocumentInfo block - continuing anyway.")
|
||||
return modified
|
||||
|
||||
@@ -99,7 +102,7 @@ def should_linearize(working_file: Path, context: PdfContext) -> bool:
|
||||
|
||||
For smaller files, linearization is not worth the effort.
|
||||
"""
|
||||
filesize = os.stat(working_file).st_size
|
||||
filesize = working_file.stat().st_size
|
||||
return filesize > (context.options.fast_web_view * 1_000_000)
|
||||
|
||||
|
||||
|
||||
+66
-24
@@ -29,7 +29,7 @@ log = logging.getLogger(__name__)
|
||||
|
||||
# Module-level registry for plugin option models
|
||||
# This is populated by setup_plugin_infrastructure() after plugins are loaded
|
||||
_plugin_option_models: dict[str, type] = {}
|
||||
_plugin_option_models: dict[str, type[BaseModel]] = {}
|
||||
|
||||
PathOrIO = BinaryIO | IOBase | Path | str | bytes
|
||||
|
||||
@@ -43,12 +43,16 @@ class ProcessingMode(StrEnum):
|
||||
- ``force``: Rasterize all content and run OCR regardless of existing text
|
||||
- ``skip``: Skip OCR on pages that already have text
|
||||
- ``redo``: Re-OCR pages, stripping old invisible text layer
|
||||
- ``strip``: Remove the invisible OCR text layer in place; do not OCR
|
||||
"""
|
||||
|
||||
default = 'default'
|
||||
force = 'force'
|
||||
skip = 'skip'
|
||||
redo = 'redo'
|
||||
# User-facing value is '--mode strip'; the member is named strip_text to
|
||||
# avoid shadowing str.strip on this str-based enum.
|
||||
strip_text = 'strip'
|
||||
|
||||
|
||||
class TaggedPdfMode(StrEnum):
|
||||
@@ -65,8 +69,33 @@ class TaggedPdfMode(StrEnum):
|
||||
ignore = 'ignore'
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges: str) -> set[int]:
|
||||
"""Convert page range string to set of page numbers."""
|
||||
def _has_end_alias(ranges: str) -> bool:
|
||||
"""Return True if the page range string uses the ``end`` alias."""
|
||||
return 'end' in ranges.lower()
|
||||
|
||||
|
||||
def _resolve_page_token(token: str, total_pages: int | None) -> int:
|
||||
"""Convert a single page-number token to a 1-based integer.
|
||||
|
||||
The literal ``end`` (case-insensitive) is resolved to ``total_pages``. If
|
||||
``total_pages`` is None, an error is raised.
|
||||
"""
|
||||
if token.lower() == 'end':
|
||||
if total_pages is None:
|
||||
raise BadArgsError(
|
||||
"'end' was used in --pages but the total page count is not yet known"
|
||||
)
|
||||
return total_pages
|
||||
return int(token)
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges: str, total_pages: int | None = None) -> set[int]:
|
||||
"""Convert page range string to set of 0-based page numbers.
|
||||
|
||||
The token ``end`` (case-insensitive) is an alias for the last page of the
|
||||
document. It is resolved using ``total_pages``; if ``end`` appears in the
|
||||
string and ``total_pages`` is None, a :class:`BadArgsError` is raised.
|
||||
"""
|
||||
pages: list[int] = []
|
||||
page_groups = ranges.replace(' ', '').split(',')
|
||||
for group in page_groups:
|
||||
@@ -75,10 +104,15 @@ def _pages_from_ranges(ranges: str) -> set[int]:
|
||||
try:
|
||||
start, end = group.split('-')
|
||||
except ValueError:
|
||||
pages.append(int(group) - 1)
|
||||
try:
|
||||
pages.append(_resolve_page_token(group, total_pages) - 1)
|
||||
except ValueError:
|
||||
raise BadArgsError(f"invalid page number '{group}'") from None
|
||||
else:
|
||||
try:
|
||||
new_pages = list(range(int(start) - 1, int(end)))
|
||||
start_n = _resolve_page_token(start, total_pages)
|
||||
end_n = _resolve_page_token(end, total_pages)
|
||||
new_pages = list(range(start_n - 1, end_n))
|
||||
if not new_pages:
|
||||
raise BadArgsError(
|
||||
f"invalid page subrange '{start}-{end}'"
|
||||
@@ -173,20 +207,19 @@ class OcrOptions(BaseModel):
|
||||
|
||||
# Optimization
|
||||
optimize: int = 1
|
||||
jpg_quality: int | None = None
|
||||
jpeg_quality: int | None = None
|
||||
png_quality: int | None = None
|
||||
jbig2_threshold: float = 0.85
|
||||
|
||||
# Compatibility alias for plugins that expect jpeg_quality
|
||||
# Deprecated compatibility alias for code that still uses the old field name
|
||||
@property
|
||||
def jpeg_quality(self):
|
||||
"""Compatibility alias for jpg_quality."""
|
||||
return self.jpg_quality
|
||||
def jpg_quality(self):
|
||||
"""Deprecated compatibility alias for jpeg_quality."""
|
||||
return self.jpeg_quality
|
||||
|
||||
@jpeg_quality.setter
|
||||
def jpeg_quality(self, value):
|
||||
"""Compatibility alias for jpg_quality."""
|
||||
self.jpg_quality = value
|
||||
@jpg_quality.setter
|
||||
def jpg_quality(self, value):
|
||||
"""Deprecated compatibility alias for jpeg_quality."""
|
||||
self.jpeg_quality = value
|
||||
|
||||
# Output behavior
|
||||
no_overwrite: bool = False
|
||||
@@ -332,11 +365,19 @@ class OcrOptions(BaseModel):
|
||||
@field_validator('pages')
|
||||
@classmethod
|
||||
def validate_pages_format(cls, v):
|
||||
"""Convert page ranges string to set of page numbers."""
|
||||
"""Convert page ranges string to set of page numbers.
|
||||
|
||||
If the string uses the ``end`` alias, the original string is preserved
|
||||
so that resolution can happen later, once the document's page count is
|
||||
known.
|
||||
"""
|
||||
if v is None:
|
||||
return v
|
||||
if isinstance(v, set):
|
||||
return v # Already processed
|
||||
if _has_end_alias(v):
|
||||
# Defer resolution until total page count is known
|
||||
return v
|
||||
|
||||
# Convert string ranges to set of page numbers
|
||||
return _pages_from_ranges(v)
|
||||
@@ -423,7 +464,7 @@ class OcrOptions(BaseModel):
|
||||
):
|
||||
raise ValueError(
|
||||
"Since you specified `--output-type none`, the output file "
|
||||
f"{self.output_file} cannot be produced. Set the output file to "
|
||||
f"{str(self.output_file)} cannot be produced. Set the output file to "
|
||||
f"`-` to suppress this message."
|
||||
)
|
||||
return self
|
||||
@@ -534,7 +575,7 @@ class OcrOptions(BaseModel):
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def register_plugin_models(cls, models: dict[str, type]) -> None:
|
||||
def register_plugin_models(cls, models: dict[str, type[BaseModel]]) -> None:
|
||||
"""Register plugin option model classes for nested access.
|
||||
|
||||
Args:
|
||||
@@ -582,6 +623,13 @@ class OcrOptions(BaseModel):
|
||||
value = getattr(self, flat_name)
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
# Plugin-scoped fields that aren't in the central OcrOptions
|
||||
# registry: argparse stores them in extra_attrs under the
|
||||
# namespace_field name.
|
||||
elif flat_name in self.extra_attrs:
|
||||
value = self.extra_attrs[flat_name]
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
# Also check direct field name (for fields like jbig2_lossy)
|
||||
elif field_name in OcrOptions.model_fields:
|
||||
value = getattr(self, field_name)
|
||||
@@ -594,12 +642,6 @@ class OcrOptions(BaseModel):
|
||||
value = self.optimize
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
elif namespace == 'optimize' and field_name == 'jpeg_quality':
|
||||
# jpg_quality maps to jpeg_quality
|
||||
if 'jpg_quality' in OcrOptions.model_fields:
|
||||
value = self.jpg_quality
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
|
||||
# Create and cache the plugin options instance
|
||||
instance = model_class(**kwargs)
|
||||
|
||||
@@ -0,0 +1,253 @@
|
||||
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
||||
# SPDX-FileCopyrightText: 2025 ajdlinux
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Validate and repair malformed page-boundary boxes.
|
||||
|
||||
A page's boundary boxes (``/MediaBox``, ``/CropBox``, ``/TrimBox``, ``/ArtBox``,
|
||||
``/BleedBox``) are sometimes malformed in ways that PDF readers tolerate but
|
||||
that crash or corrupt downstream processing. This module normalizes them in
|
||||
place following the PDF 2.0 specification (ISO 32000-2:2020):
|
||||
|
||||
- **Non-decimal coordinates** (§7.3.3): a coordinate written in exponential
|
||||
notation is invalid PDF number syntax and is stored by qpdf/pikepdf as a
|
||||
string. We coerce it back to a number (issue #1398).
|
||||
- **Reversed corners** (§7.9.5): a rectangle is "a pair of diagonally opposite
|
||||
corners"; ``[llx lly urx ury]`` is only the typical order. We normalize to
|
||||
``[min_x, min_y, max_x, max_y]`` (issue #1526).
|
||||
- **Sub-box outside the MediaBox** (§14.11.2): "If the bounds of the crop,
|
||||
trim, bleed or art box extends outside of the bounds of the media box, a
|
||||
processor shall treat the box as its intersection with the media box." We
|
||||
clamp to that intersection, or discard the sub-box (so it inherits the
|
||||
MediaBox) when the intersection is empty (issue #1400).
|
||||
|
||||
A rectangle is treated as empty when its width or height is ``<= 0``; PDF 2.0
|
||||
permits zero-dimension rectangles and defines no minimum page size, so no other
|
||||
size floor is imposed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
from collections.abc import Iterable, Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pikepdf
|
||||
from pikepdf import Name
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_SUBBOXES = ('CropBox', 'TrimBox', 'ArtBox', 'BleedBox')
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BoxRepair:
|
||||
"""A single change made to a page box.
|
||||
|
||||
Attributes:
|
||||
box: The box name, e.g. ``"CropBox"``.
|
||||
kind: One of ``"reordered"`` (reversed corners normalized; lossless),
|
||||
``"recoded"`` (non-numeric/exponential coordinate coerced),
|
||||
``"clamped"`` (sub-box clamped to the MediaBox), ``"discarded"``
|
||||
(sub-box removed because its MediaBox intersection was empty), or
|
||||
``"degenerate_mediabox"`` (MediaBox has zero width or height).
|
||||
"""
|
||||
|
||||
box: str
|
||||
kind: str
|
||||
|
||||
|
||||
def _read_box(values: Sequence) -> tuple[list[float], bool, bool] | None:
|
||||
"""Coerce a box array to floats and normalize corner order.
|
||||
|
||||
Returns ``(normalized_values, recoded, reordered)`` where ``recoded`` is
|
||||
True if any element needed string/exponential coercion and ``reordered`` is
|
||||
True if the corners were given in non-standard order. Returns None if the
|
||||
array is not four finite numbers.
|
||||
"""
|
||||
if len(values) != 4:
|
||||
return None
|
||||
nums: list[float] = []
|
||||
recoded = False
|
||||
for v in values:
|
||||
try:
|
||||
n = float(v)
|
||||
except (TypeError, ValueError):
|
||||
try:
|
||||
n = float(str(v))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
recoded = True
|
||||
if not math.isfinite(n):
|
||||
return None
|
||||
nums.append(n)
|
||||
x0, y0, x1, y1 = nums
|
||||
normalized = [min(x0, x1), min(y0, y1), max(x0, x1), max(y0, y1)]
|
||||
reordered = normalized != nums
|
||||
return normalized, recoded, reordered
|
||||
|
||||
|
||||
def coerce_box(values: Iterable) -> list[float]:
|
||||
"""Return box values coerced to floats with corner order normalized.
|
||||
|
||||
Robust against exponential/string coordinates and reversed corners, so
|
||||
callers that only need to read a box (e.g. dimension calculations) do not
|
||||
crash on malformed input. Falls back to best-effort per-element coercion if
|
||||
the array is not four numbers.
|
||||
"""
|
||||
values = list(values)
|
||||
result = _read_box(values)
|
||||
if result is not None:
|
||||
return result[0]
|
||||
coerced = []
|
||||
for v in values:
|
||||
try:
|
||||
coerced.append(float(v))
|
||||
except (TypeError, ValueError):
|
||||
coerced.append(float(str(v)))
|
||||
return coerced
|
||||
|
||||
|
||||
def _is_empty(box: Sequence[float]) -> bool:
|
||||
"""A rectangle is empty when its width or height is non-positive."""
|
||||
return (box[2] - box[0]) <= 0 or (box[3] - box[1]) <= 0
|
||||
|
||||
|
||||
def repair_page_boxes(page: pikepdf.Page) -> list[BoxRepair]:
|
||||
"""Validate and repair the boundary boxes of a single page, in place.
|
||||
|
||||
Returns the list of changes made (empty if the page was already valid).
|
||||
Only boxes that actually change are written back, so valid pages are left
|
||||
untouched. Performs no logging or I/O.
|
||||
"""
|
||||
repairs: list[BoxRepair] = []
|
||||
|
||||
# MediaBox is the reference rectangle; read it inheritance-aware.
|
||||
mediabox: list[float] | None = None
|
||||
try:
|
||||
mb_result = _read_box(list(page.mediabox.as_list()))
|
||||
except (AttributeError, KeyError, RuntimeError):
|
||||
mb_result = None
|
||||
if mb_result is not None:
|
||||
mediabox, recoded, reordered = mb_result
|
||||
if reordered:
|
||||
repairs.append(BoxRepair('MediaBox', 'reordered'))
|
||||
if recoded:
|
||||
repairs.append(BoxRepair('MediaBox', 'recoded'))
|
||||
if recoded or reordered:
|
||||
page.obj.MediaBox = pikepdf.Array(mediabox)
|
||||
if _is_empty(mediabox):
|
||||
repairs.append(BoxRepair('MediaBox', 'degenerate_mediabox'))
|
||||
mediabox = None # don't clamp against a degenerate reference
|
||||
|
||||
for box in _SUBBOXES:
|
||||
name = Name('/' + box)
|
||||
if name not in page.obj:
|
||||
continue
|
||||
try:
|
||||
sub_result = _read_box(list(page.obj[name]))
|
||||
except (TypeError, RuntimeError):
|
||||
continue
|
||||
if sub_result is None:
|
||||
continue
|
||||
values, recoded, reordered = sub_result
|
||||
if reordered:
|
||||
repairs.append(BoxRepair(box, 'reordered'))
|
||||
if recoded:
|
||||
repairs.append(BoxRepair(box, 'recoded'))
|
||||
if recoded or reordered:
|
||||
page.obj[name] = pikepdf.Array(values)
|
||||
|
||||
if mediabox is None:
|
||||
continue
|
||||
intersection = [
|
||||
max(values[0], mediabox[0]),
|
||||
max(values[1], mediabox[1]),
|
||||
min(values[2], mediabox[2]),
|
||||
min(values[3], mediabox[3]),
|
||||
]
|
||||
if _is_empty(intersection):
|
||||
del page.obj[name]
|
||||
repairs.append(BoxRepair(box, 'discarded'))
|
||||
elif intersection != values:
|
||||
page.obj[name] = pikepdf.Array(intersection)
|
||||
repairs.append(BoxRepair(box, 'clamped'))
|
||||
|
||||
return repairs
|
||||
|
||||
|
||||
# Per-kind log severity and message template ({box} is substituted).
|
||||
_KIND_MESSAGES: dict[str, tuple[int, str]] = {
|
||||
'discarded': (
|
||||
logging.WARNING,
|
||||
'{box} lies outside the MediaBox and was discarded; '
|
||||
'the full page will be shown',
|
||||
),
|
||||
'clamped': (
|
||||
logging.WARNING,
|
||||
'{box} extended beyond the MediaBox and was clamped to it',
|
||||
),
|
||||
'recoded': (
|
||||
logging.WARNING,
|
||||
'{box} used invalid (e.g. exponential) coordinates, which were reinterpreted',
|
||||
),
|
||||
'degenerate_mediabox': (
|
||||
logging.WARNING,
|
||||
'MediaBox has zero width or height and could not be repaired; '
|
||||
'output may be invalid',
|
||||
),
|
||||
'reordered': (
|
||||
logging.DEBUG,
|
||||
'{box} corners were reversed and have been normalized',
|
||||
),
|
||||
}
|
||||
|
||||
# Kinds that change page appearance and warrant manual review of the output.
|
||||
_INSPECT_KINDS = frozenset({'discarded', 'clamped', 'recoded'})
|
||||
_INSPECT = ' Please visually inspect the output PDF.'
|
||||
|
||||
|
||||
def _format_pages(pagenos: Iterable[int]) -> str:
|
||||
"""Format 0-based page numbers as a compact 1-based range string."""
|
||||
nums = sorted(p + 1 for p in pagenos)
|
||||
ranges: list[tuple[int, int]] = []
|
||||
start = prev = nums[0]
|
||||
for n in nums[1:]:
|
||||
if n == prev + 1:
|
||||
prev = n
|
||||
continue
|
||||
ranges.append((start, prev))
|
||||
start = prev = n
|
||||
ranges.append((start, prev))
|
||||
return ', '.join(f'{a}' if a == b else f'{a}-{b}' for a, b in ranges)
|
||||
|
||||
|
||||
def summarize_box_repairs(
|
||||
repairs_by_page: Mapping[int, Sequence[BoxRepair]],
|
||||
) -> list[tuple[int, str]]:
|
||||
"""Aggregate per-page repairs into ``(log_level, message)`` pairs.
|
||||
|
||||
Repairs are grouped by ``(kind, box)`` so a defect shared across many pages
|
||||
yields a single message listing the affected pages, rather than one message
|
||||
per page.
|
||||
"""
|
||||
groups: dict[tuple[str, str], set[int]] = {}
|
||||
for pageno, repairs in repairs_by_page.items():
|
||||
for repair in repairs:
|
||||
groups.setdefault((repair.kind, repair.box), set()).add(pageno)
|
||||
|
||||
messages: list[tuple[int, str]] = []
|
||||
for (kind, box), pages in sorted(groups.items()):
|
||||
level, template = _KIND_MESSAGES[kind]
|
||||
text = f'Page(s) {_format_pages(pages)}: {template.format(box=box)}.'
|
||||
if kind in _INSPECT_KINDS:
|
||||
text += _INSPECT
|
||||
messages.append((level, text))
|
||||
return messages
|
||||
|
||||
|
||||
def log_box_repairs(repairs_by_page: Mapping[int, Sequence[BoxRepair]]) -> None:
|
||||
"""Emit aggregated log messages for the repairs made across all pages."""
|
||||
for level, message in summarize_box_repairs(repairs_by_page):
|
||||
log.log(level, message)
|
||||
+173
-71
@@ -28,23 +28,29 @@ from ocrmypdf._concurrent import Executor
|
||||
from ocrmypdf._exec import unpaper
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._metadata import repair_docinfo_nuls
|
||||
from ocrmypdf._options import OcrOptions, ProcessingMode, TaggedPdfMode
|
||||
from ocrmypdf._options import OcrOptions, PathOrIO, ProcessingMode, TaggedPdfMode
|
||||
from ocrmypdf._pageboxes import log_box_repairs, repair_page_boxes
|
||||
from ocrmypdf._stdoutprotect import get_protected_stdout_fd
|
||||
from ocrmypdf.exceptions import (
|
||||
ColorConversionNeededError,
|
||||
DigitalSignatureError,
|
||||
DpiError,
|
||||
EncryptedPdfError,
|
||||
InputFileError,
|
||||
NonEmbeddedFontsError,
|
||||
PriorOcrFoundError,
|
||||
SubprocessOutputError,
|
||||
TaggedPDFError,
|
||||
UnsupportedImageFormatError,
|
||||
)
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
|
||||
from ocrmypdf.pdfa import (
|
||||
file_claims_pdfa,
|
||||
find_nonembedded_cid_fonts,
|
||||
generate_pdfa_ps,
|
||||
speculative_pdfa_conversion,
|
||||
)
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, FloatRect, PageInfo, PdfInfo
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, FloatRect, Ink, PageInfo, PdfInfo
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice, OrientationConfidence
|
||||
|
||||
try:
|
||||
@@ -116,8 +122,7 @@ def triage_image_file(input_file: Path, output_file: Path, options: OcrOptions)
|
||||
|
||||
if im.mode in ('RGBA', 'LA'):
|
||||
raise UnsupportedImageFormatError(
|
||||
"The input image has an alpha channel. Remove the alpha "
|
||||
"channel first."
|
||||
"The input image has an alpha channel. Remove the alpha channel first."
|
||||
)
|
||||
|
||||
if 'iccprofile' not in im.info:
|
||||
@@ -135,7 +140,7 @@ def triage_image_file(input_file: Path, output_file: Path, options: OcrOptions)
|
||||
layout_fun = img2pdf.get_fixed_dpi_layout_fun(
|
||||
Resolution(options.image_dpi, options.image_dpi)
|
||||
)
|
||||
with open(output_file, 'wb') as outf:
|
||||
with output_file.open('wb') as outf:
|
||||
img2pdf.convert(
|
||||
os.fspath(input_file),
|
||||
layout_fun=layout_fun,
|
||||
@@ -154,7 +159,7 @@ def _pdf_guess_version(input_file: Path, search_window=1024) -> str:
|
||||
|
||||
Returns empty string if not found, indicating file is probably not PDF.
|
||||
"""
|
||||
with open(input_file, 'rb') as f:
|
||||
with input_file.open('rb') as f:
|
||||
signature = f.read(search_window)
|
||||
m = re.search(rb'%PDF-(\d\.\d)', signature)
|
||||
if m:
|
||||
@@ -175,6 +180,12 @@ def triage(
|
||||
)
|
||||
try:
|
||||
with pikepdf.open(input_file) as pdf:
|
||||
repairs_by_page = {
|
||||
n: repairs
|
||||
for n, page in enumerate(pdf.pages)
|
||||
if (repairs := repair_page_boxes(page))
|
||||
}
|
||||
log_box_repairs(repairs_by_page)
|
||||
pdf.save(output_file)
|
||||
except pikepdf.PdfError as e:
|
||||
raise InputFileError() from e
|
||||
@@ -250,12 +261,15 @@ def validate_pdfinfo_options(context: PdfContext) -> None:
|
||||
"image of the form and all filled form fields. The output PDF "
|
||||
"will be 'flattened' and will no longer be fillable."
|
||||
)
|
||||
if pdfinfo.is_tagged:
|
||||
if pdfinfo.is_tagged or pdfinfo.has_structure_tree:
|
||||
log.warning(
|
||||
"This PDF is marked as a Tagged PDF. This often indicates "
|
||||
"that the PDF was generated from an office document and does "
|
||||
"not need OCR. PDF pages processed by OCRmyPDF may not be "
|
||||
"tagged correctly."
|
||||
"This PDF contains structural markup (it is a Tagged PDF or "
|
||||
"carries a logical structure tree). This often indicates that the "
|
||||
"PDF was generated from an office document or is otherwise born "
|
||||
"digital, and does not need OCR. OCRmyPDF cannot rebuild this "
|
||||
"structure to match new text, so any page it re-OCRs with "
|
||||
"--force-ocr or --redo-ocr will have its structural markup "
|
||||
"discarded."
|
||||
)
|
||||
if (
|
||||
options.tagged_pdf_mode == TaggedPdfMode.default
|
||||
@@ -325,6 +339,11 @@ def is_ocr_required(page_context: PageContext) -> bool:
|
||||
pageinfo = page_context.pageinfo
|
||||
options = page_context.options
|
||||
|
||||
if options.mode == ProcessingMode.strip_text:
|
||||
# Strip mode removes the OCR text layer in place; it never rasterizes
|
||||
# or runs OCR. The stripping happens in OcrGrafter.graft_page.
|
||||
return False
|
||||
|
||||
ocr_required = True
|
||||
|
||||
if options.pages and pageinfo.pageno not in options.pages:
|
||||
@@ -508,6 +527,49 @@ def calculate_raster_dpi(page_context: PageContext):
|
||||
return canvas_dpi, page_dpi
|
||||
|
||||
|
||||
def _select_raster_device(pageinfo: PageInfo) -> GhostscriptRasterDevice:
|
||||
"""Choose the minimum raster device that preserves the page's color depth.
|
||||
|
||||
The device escalates from 1-bit mono through grayscale, indexed, and full
|
||||
color as required by the page's images, image masks, and vector content.
|
||||
Image masks are painted with the current fill color, so a mask painted in
|
||||
gray or color escalates the device even though the mask itself is 1-bit.
|
||||
"""
|
||||
colorspaces = [
|
||||
GhostscriptRasterDevice.PNGMONOD,
|
||||
GhostscriptRasterDevice.PNGGRAY,
|
||||
GhostscriptRasterDevice.PNG256,
|
||||
GhostscriptRasterDevice.PNG16M,
|
||||
]
|
||||
device_idx = 0
|
||||
|
||||
def at_least(colorspace):
|
||||
return max(device_idx, colorspaces.index(colorspace))
|
||||
|
||||
for image in pageinfo.images:
|
||||
if image.type_ == 'stencil':
|
||||
# The fill color used to paint the mask, not the 1-bit mask data,
|
||||
# determines the color depth OCR needs.
|
||||
if image.ink == Ink.color:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
elif image.ink == Ink.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
continue
|
||||
if image.bpc > 1:
|
||||
if image.color == Colorspace.index:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG256)
|
||||
elif image.color == Colorspace.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
else:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
if pageinfo.has_vector:
|
||||
log.debug(f"Page has vector content, using {GhostscriptRasterDevice.PNG16M}")
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
return colorspaces[device_idx]
|
||||
|
||||
|
||||
def rasterize(
|
||||
input_file: Path,
|
||||
page_context: PageContext,
|
||||
@@ -529,39 +591,13 @@ def rasterize(
|
||||
Returns:
|
||||
Path: The output PNG file path.
|
||||
"""
|
||||
colorspaces = [
|
||||
GhostscriptRasterDevice.PNGMONO,
|
||||
GhostscriptRasterDevice.PNGGRAY,
|
||||
GhostscriptRasterDevice.PNG256,
|
||||
GhostscriptRasterDevice.PNG16M,
|
||||
]
|
||||
device_idx = 0
|
||||
|
||||
if remove_vectors is None:
|
||||
remove_vectors = page_context.options.remove_vectors
|
||||
|
||||
output_file = page_context.get_path(f'rasterize{output_tag}.png')
|
||||
pageinfo = page_context.pageinfo
|
||||
|
||||
def at_least(colorspace):
|
||||
return max(device_idx, colorspaces.index(colorspace))
|
||||
|
||||
for image in pageinfo.images:
|
||||
if image.type_ != 'image':
|
||||
continue # ignore masks
|
||||
if image.bpc > 1:
|
||||
if image.color == Colorspace.index:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG256)
|
||||
elif image.color == Colorspace.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
else:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
if pageinfo.has_vector:
|
||||
log.debug(f"Page has vector content, using {GhostscriptRasterDevice.PNG16M}")
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
device = colorspaces[device_idx]
|
||||
device = _select_raster_device(pageinfo)
|
||||
|
||||
log.debug(
|
||||
f"Rasterize with {device}, rotation {correction}, mediabox {pageinfo.mediabox}"
|
||||
@@ -647,6 +683,7 @@ def create_ocr_image(image: Path, page_context: PageContext) -> Path:
|
||||
"""
|
||||
output_file = page_context.get_path('ocr.png')
|
||||
options = page_context.options
|
||||
im: Image.Image
|
||||
with Image.open(image) as im:
|
||||
log.debug('resolution %r', im.info['dpi'])
|
||||
|
||||
@@ -796,7 +833,7 @@ def create_pdf_page_from_image(
|
||||
|
||||
# Create a new single page PDF to hold
|
||||
bio = BytesIO()
|
||||
with open(image, 'rb') as imfile:
|
||||
with image.open('rb') as imfile:
|
||||
log.debug('convert')
|
||||
|
||||
layout_fun = img2pdf.get_layout_fun(pagesize)
|
||||
@@ -945,6 +982,12 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext) -
|
||||
# pikepdf can deal with this, but we make the world a better place by
|
||||
# stamping them out as soon as possible.
|
||||
with pikepdf.open(input_pdf) as pdf_file:
|
||||
# Ghostscript would substitute and re-embed any non-embedded CID font to
|
||||
# satisfy PDF/A, corrupting CJK text (e.g. an Acrobat OCR layer) in the
|
||||
# process. Refuse rather than silently damage the user's text layer.
|
||||
nonembedded = find_nonembedded_cid_fonts(pdf_file)
|
||||
if nonembedded:
|
||||
raise NonEmbeddedFontsError(nonembedded)
|
||||
if repair_docinfo_nuls(pdf_file):
|
||||
pdf_file.save(fix_docinfo_file)
|
||||
else:
|
||||
@@ -1038,14 +1081,46 @@ def try_speculative_pdfa(input_pdf: Path, context: PdfContext) -> Path | None:
|
||||
return None
|
||||
|
||||
|
||||
def _ghostscript_pdfa_fallback(input_pdf: Path, context: PdfContext) -> Path | None:
|
||||
"""Best-effort PDF/A conversion via Ghostscript for 'auto' output type.
|
||||
|
||||
Returns the converted PDF/A path, or None if Ghostscript is unavailable,
|
||||
fails, or cannot produce valid PDF/A. Never raises: 'auto' mode degrades to
|
||||
a regular PDF instead of erroring or emitting corrupted output.
|
||||
|
||||
Args:
|
||||
input_pdf: Path to the PDF to convert.
|
||||
context: The PDF context.
|
||||
"""
|
||||
from ocrmypdf._exec import ghostscript
|
||||
|
||||
if not ghostscript.available():
|
||||
return None
|
||||
try:
|
||||
ps_stub = generate_postscript_stub(context)
|
||||
gs_out = convert_to_pdfa(input_pdf, ps_stub, context)
|
||||
except (
|
||||
SubprocessOutputError,
|
||||
ColorConversionNeededError,
|
||||
NonEmbeddedFontsError,
|
||||
) as e:
|
||||
log.info('Auto mode: Ghostscript could not produce PDF/A (%s)', e)
|
||||
return None
|
||||
if not file_claims_pdfa(gs_out)['pass']:
|
||||
log.info('Auto mode: Ghostscript output is not valid PDF/A')
|
||||
return None
|
||||
return gs_out
|
||||
|
||||
|
||||
def try_auto_pdfa(input_pdf: Path, context: PdfContext) -> tuple[Path, str]:
|
||||
"""Best-effort PDF/A for 'auto' output type.
|
||||
|
||||
This function attempts to produce PDF/A without requiring Ghostscript:
|
||||
1. If verapdf is available, tries speculative conversion with validation
|
||||
2. Without verapdf, passes through as PDF/A if safe (input already PDF/A
|
||||
or force-ocr was used)
|
||||
3. Falls back to regular PDF if neither condition is met
|
||||
Order of attempts, first success wins:
|
||||
1. Non-embedded CID fonts -> regular PDF (Ghostscript would corrupt them).
|
||||
2. Speculative conversion validated by verapdf (no Ghostscript).
|
||||
3. Without verapdf, pass through if already PDF/A or rebuilt with force-ocr.
|
||||
4. Ghostscript conversion (best-effort; failures fall through).
|
||||
5. Regular PDF if none of the above produced PDF/A.
|
||||
|
||||
Args:
|
||||
input_pdf: Path to the PDF to convert
|
||||
@@ -1057,25 +1132,42 @@ def try_auto_pdfa(input_pdf: Path, context: PdfContext) -> tuple[Path, str]:
|
||||
"""
|
||||
from ocrmypdf._exec import verapdf
|
||||
|
||||
# If verapdf available, try speculative conversion with validation
|
||||
# Non-embedded CID fonts cannot be made PDF/A without Ghostscript font
|
||||
# substitution that corrupts CID/CJK text. Rather than risk an existing
|
||||
# text layer, downgrade to a regular PDF (the same outcome as any other
|
||||
# case where best-effort PDF/A is not achievable).
|
||||
with pikepdf.open(input_pdf) as pdf_file:
|
||||
nonembedded = find_nonembedded_cid_fonts(pdf_file)
|
||||
if nonembedded:
|
||||
log.info(
|
||||
"Auto mode: input has non-embedded CID fonts (%s) that cannot be "
|
||||
"converted to PDF/A without corrupting the text; outputting a "
|
||||
"regular PDF. Use --output-type pdf to select this explicitly.",
|
||||
', '.join(sorted(nonembedded)),
|
||||
)
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
# Cheap path: speculative conversion validated by verapdf (no Ghostscript).
|
||||
if verapdf.available():
|
||||
result = try_speculative_pdfa(input_pdf, context)
|
||||
if result is not None:
|
||||
return (result, 'pdfa')
|
||||
# verapdf validation failed - fall through to regular PDF
|
||||
log.info(
|
||||
'Auto mode: speculative PDF/A validation failed, outputting regular PDF'
|
||||
)
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
# Without verapdf, check if we can pass through as PDF/A
|
||||
if _is_safe_pdfa(input_pdf, context.options):
|
||||
# Pass through as-is (no modifications needed)
|
||||
log.info('Auto mode: speculative PDF/A validation failed')
|
||||
elif _is_safe_pdfa(input_pdf, context.options):
|
||||
# No verapdf, but the input is already PDF/A or was rebuilt with
|
||||
# --force-ocr, so we can pass it through without Ghostscript.
|
||||
log.info('Auto mode: passing through as PDF/A (input already compliant)')
|
||||
return (input_pdf, 'pdfa')
|
||||
|
||||
# Fall through to regular PDF
|
||||
log.info('Auto mode: no verapdf available and input is not PDF/A, outputting PDF')
|
||||
# Fall back to Ghostscript to produce real PDF/A (v16 behavior). Best-effort:
|
||||
# if Ghostscript is unavailable or cannot safely produce PDF/A, keep a
|
||||
# regular PDF rather than error.
|
||||
gs_out = _ghostscript_pdfa_fallback(input_pdf, context)
|
||||
if gs_out is not None:
|
||||
log.info('Auto mode: produced PDF/A via Ghostscript')
|
||||
return (gs_out, 'pdfa')
|
||||
|
||||
log.info('Auto mode: could not produce PDF/A, outputting regular PDF')
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
|
||||
@@ -1107,7 +1199,7 @@ def should_linearize(working_file: Path, context: PdfContext) -> bool:
|
||||
|
||||
For smaller files, linearization is not worth the effort.
|
||||
"""
|
||||
filesize = os.stat(working_file).st_size
|
||||
filesize = working_file.stat().st_size
|
||||
return filesize > (context.options.fast_web_view * 1_000_000)
|
||||
|
||||
|
||||
@@ -1193,7 +1285,8 @@ def enumerate_compress_ranges(
|
||||
A tuple containing a range of indices and the corresponding element.
|
||||
If the element is None, the range represents a skipped range of indices.
|
||||
"""
|
||||
skipped_from, index = None, None
|
||||
skipped_from: int | None = None
|
||||
index: int | None = None
|
||||
for index, txt_file in enumerate(iterable):
|
||||
index += 1
|
||||
if txt_file:
|
||||
@@ -1205,6 +1298,9 @@ def enumerate_compress_ranges(
|
||||
if skipped_from is None:
|
||||
skipped_from = index
|
||||
if skipped_from is not None:
|
||||
# skipped_from can only be set inside the loop above, so the loop
|
||||
# must have run at least once and index is guaranteed to be an int.
|
||||
assert index is not None
|
||||
yield (skipped_from, index), None
|
||||
|
||||
|
||||
@@ -1216,7 +1312,7 @@ def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext) -> Pat
|
||||
and returns the path to the merged file.
|
||||
"""
|
||||
output_file = context.get_path('sidecar.txt')
|
||||
with open(output_file, 'w', encoding="utf-8") as stream:
|
||||
with output_file.open('w', encoding="utf-8") as stream:
|
||||
for (from_, to_), txt_file in enumerate_compress_ranges(txt_files):
|
||||
if from_ != 1:
|
||||
stream.write('\f') # Form feed between pages for all pages after first
|
||||
@@ -1231,24 +1327,28 @@ def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext) -> Pat
|
||||
return output_file
|
||||
|
||||
|
||||
def copy_final(
|
||||
input_file: Path, output_file: str | Path | BinaryIO, original_file: Path | None
|
||||
) -> None:
|
||||
def copy_final(input_file: Path, output_file: PathOrIO) -> None:
|
||||
"""Copy the final temporary file to the output destination.
|
||||
|
||||
Args:
|
||||
input_file (Path): The intermediate input file to copy.
|
||||
output_file (str | Path | BinaryIO): The output file to copy to.
|
||||
original_file: The original file to copy attributes from.
|
||||
|
||||
Returns:
|
||||
None
|
||||
input_file: The intermediate input file to copy.
|
||||
output_file: The output file to copy to.
|
||||
"""
|
||||
log.debug('%s -> %s', input_file, output_file)
|
||||
with input_file.open('rb') as input_stream:
|
||||
if output_file == '-':
|
||||
copyfileobj(input_stream, sys.stdout.buffer) # type: ignore[misc]
|
||||
sys.stdout.flush()
|
||||
fd = get_protected_stdout_fd()
|
||||
if fd is not None:
|
||||
# Stdout protection is active: write to the preserved real
|
||||
# stdout. dup the saved fd so the with-block's close() does not
|
||||
# close our long-lived descriptor.
|
||||
with os.fdopen(os.dup(fd), 'wb') as stdout_stream:
|
||||
copyfileobj(input_stream, stdout_stream)
|
||||
stdout_stream.flush()
|
||||
else:
|
||||
# No protection installed (e.g. plain API use): legacy behavior.
|
||||
copyfileobj(input_stream, sys.stdout.buffer) # type: ignore[misc]
|
||||
sys.stdout.flush()
|
||||
elif hasattr(output_file, 'writable'):
|
||||
output_stream = cast(BinaryIO, output_file)
|
||||
copyfileobj(input_stream, output_stream) # type: ignore[misc]
|
||||
@@ -1258,5 +1358,7 @@ def copy_final(
|
||||
# At this point we overwrite the output_file specified by the user
|
||||
# use copyfileobj because then we use open() to create the file and
|
||||
# get the appropriate umask, ownership, etc.
|
||||
with open(output_file, 'w+b') as output_stream:
|
||||
# The `hasattr` check above already ruled out stream-like objects.
|
||||
assert isinstance(output_file, str | bytes | os.PathLike)
|
||||
with Path(os.fsdecode(output_file)).open('w+b') as output_stream:
|
||||
copyfileobj(input_stream, output_stream)
|
||||
|
||||
@@ -343,12 +343,17 @@ def setup_pipeline(
|
||||
|
||||
|
||||
def do_get_pdfinfo(pdf_path: Path, executor: Executor, options) -> PdfInfo:
|
||||
# Handle pages field - it might be a string that needs conversion
|
||||
# Handle pages field - it might be a string that needs conversion.
|
||||
# A string indicates the ``end`` alias was used and resolution was
|
||||
# deferred; we resolve it now using the document's actual page count.
|
||||
check_pages = options.pages
|
||||
if isinstance(check_pages, str):
|
||||
from ocrmypdf._options import _pages_from_ranges
|
||||
|
||||
check_pages = _pages_from_ranges(check_pages)
|
||||
with Pdf.open(pdf_path) as pdf:
|
||||
total_pages = len(pdf.pages)
|
||||
check_pages = _pages_from_ranges(check_pages, total_pages=total_pages)
|
||||
options.pages = check_pages
|
||||
|
||||
return get_pdfinfo(
|
||||
pdf_path,
|
||||
@@ -484,7 +489,7 @@ def postprocess(
|
||||
else:
|
||||
pdf_out = pdf_file
|
||||
if context.options.output_type == 'auto':
|
||||
# Best effort PDF/A - never uses Ghostscript
|
||||
# Best effort PDF/A - may use Ghostscript as a last resort
|
||||
pdf_out, actual_type = try_auto_pdfa(pdf_out, context)
|
||||
# Store actual output type for reporting
|
||||
context.options.extra_attrs['_actual_output_type'] = actual_type
|
||||
|
||||
@@ -98,8 +98,8 @@ def exec_hocr_to_ocr_pdf(context: PdfContext, executor: Executor) -> Sequence[st
|
||||
log.info("Postprocessing...")
|
||||
pdf, messages = postprocess(pdf, context, executor)
|
||||
|
||||
# Copy PDF file to destination (we don't know the input PDF file name)
|
||||
copy_final(pdf, options.output_file, None)
|
||||
# Copy PDF file to destination
|
||||
copy_final(pdf, options.output_file)
|
||||
return messages
|
||||
|
||||
|
||||
@@ -109,6 +109,9 @@ def run_hocr_to_ocr_pdf_pipeline(
|
||||
plugin_manager: OcrmypdfPluginManager,
|
||||
) -> ExitCode:
|
||||
"""Run pipeline to convert hOCR to final output PDF."""
|
||||
# The _hocr_to_ocr_pdf() API requires work_folder: Path and stores it on
|
||||
# options before this pipeline runs, so it is always set at this point.
|
||||
assert options.work_folder is not None
|
||||
with manage_work_folder(
|
||||
work_folder=options.work_folder, retain=True, print_location=False
|
||||
) as work_folder:
|
||||
|
||||
@@ -145,7 +145,7 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
||||
if options.sidecar:
|
||||
text = merge_sidecars(sidecars, context)
|
||||
# Copy text file to destination
|
||||
copy_final(text, options.sidecar, options.input_file)
|
||||
copy_final(text, options.sidecar)
|
||||
|
||||
# Merge layers to one single pdf
|
||||
pdf = ocrgraft.finalize()
|
||||
@@ -157,7 +157,7 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
||||
pdf, messages = postprocess(pdf, context, executor)
|
||||
|
||||
# Copy PDF file to destination
|
||||
copy_final(pdf, options.output_file, options.input_file)
|
||||
copy_final(pdf, options.output_file)
|
||||
return messages
|
||||
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import logging.handlers
|
||||
import os
|
||||
import shutil
|
||||
from functools import partial
|
||||
|
||||
@@ -91,6 +92,9 @@ def run_hocr_pipeline(
|
||||
"""Run pipeline to output hOCR."""
|
||||
if options.output_folder is None:
|
||||
raise ValueError("output_folder must be specified for hOCR pipeline")
|
||||
# This pipeline is only reachable via the _pdf_to_hocr() API, which
|
||||
# declares input_pdf: Path - streams and raw bytes paths are not supported.
|
||||
assert isinstance(options.input_file, str | os.PathLike)
|
||||
with manage_work_folder(
|
||||
work_folder=options.output_folder, retain=True, print_location=False
|
||||
) as work_folder:
|
||||
@@ -100,9 +104,7 @@ def run_hocr_pipeline(
|
||||
|
||||
# Gather pdfinfo and create context
|
||||
pdfinfo = do_get_pdfinfo(origin_pdf, executor, options)
|
||||
context = PdfContext(
|
||||
options, work_folder, options.input_file, pdfinfo, plugin_manager
|
||||
)
|
||||
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
||||
# Validate options are okay for this pdf
|
||||
validate_pdfinfo_options(context)
|
||||
exec_pdf_to_hocr(context, executor)
|
||||
|
||||
@@ -21,6 +21,7 @@ from pydantic import BaseModel
|
||||
import ocrmypdf.builtin_plugins
|
||||
from ocrmypdf import Executor, PdfContext, pluginspec
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._plugin_registry import PluginOptionRegistry
|
||||
from ocrmypdf._progressbar import ProgressBar
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pluginspec import OcrEngine
|
||||
@@ -53,10 +54,11 @@ class OcrmypdfPluginManager:
|
||||
self._plugins = plugins
|
||||
self._builtins = builtins
|
||||
self._pm = pluggy.PluginManager(*args, **kwargs)
|
||||
self._option_registry: PluginOptionRegistry | None = None
|
||||
self._setup_plugins()
|
||||
|
||||
@property
|
||||
def pluggy(self) -> pluggy.PluginManager:
|
||||
def pluggy_manager(self) -> pluggy.PluginManager:
|
||||
"""Access the underlying pluggy.PluginManager for advanced use cases.
|
||||
|
||||
This is useful for plugins that need to call methods like set_blocked()
|
||||
@@ -74,7 +76,8 @@ class OcrmypdfPluginManager:
|
||||
return state
|
||||
|
||||
def __setstate__(self, state):
|
||||
self.__init__(
|
||||
OcrmypdfPluginManager.__init__(
|
||||
self,
|
||||
*state['init_args'],
|
||||
plugins=state['plugins'],
|
||||
builtins=state['builtins'],
|
||||
@@ -86,10 +89,10 @@ class OcrmypdfPluginManager:
|
||||
|
||||
# 1. Register builtins
|
||||
if self._builtins:
|
||||
for module in sorted(
|
||||
for module_info in sorted(
|
||||
pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__)
|
||||
):
|
||||
name = f'ocrmypdf.builtin_plugins.{module.name}'
|
||||
name = f'ocrmypdf.builtin_plugins.{module_info.name}'
|
||||
module = importlib.import_module(name)
|
||||
self._pm.register(module)
|
||||
|
||||
@@ -97,17 +100,20 @@ class OcrmypdfPluginManager:
|
||||
self._pm.load_setuptools_entrypoints('ocrmypdf')
|
||||
|
||||
# 3. Register plugins specified on command line
|
||||
for name in self._plugins:
|
||||
if isinstance(name, Path) or name.endswith('.py'):
|
||||
for plugin in self._plugins:
|
||||
if isinstance(plugin, Path) or plugin.endswith('.py'):
|
||||
# Import by filename
|
||||
module_name = Path(name).stem
|
||||
spec = importlib.util.spec_from_file_location(module_name, name)
|
||||
plugin_path = Path(plugin)
|
||||
module_name = plugin_path.stem
|
||||
spec = importlib.util.spec_from_file_location(module_name, plugin_path)
|
||||
if spec is None or spec.loader is None:
|
||||
raise ImportError(f'Could not load plugin from {plugin_path}')
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
sys.modules[module_name] = module
|
||||
spec.loader.exec_module(module)
|
||||
else:
|
||||
# Import by dotted module name
|
||||
module = importlib.import_module(name)
|
||||
module = importlib.import_module(plugin)
|
||||
self._pm.register(module)
|
||||
|
||||
# =========================================================================
|
||||
|
||||
@@ -21,7 +21,7 @@ class PluginOptionRegistry:
|
||||
compatibility (e.g., options.tesseract_timeout).
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
def __init__(self) -> None:
|
||||
self._option_models: dict[str, type[BaseModel]] = {}
|
||||
|
||||
def register_option_model(
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Protect the real standard output from corruption by stray writes.
|
||||
|
||||
When OCRmyPDF writes its final PDF to standard output (``ocrmypdf in.pdf -``),
|
||||
the bytes on stdout must be exactly the PDF and nothing else. Any accidental
|
||||
write to file descriptor 1 anywhere in the process -- from a third-party
|
||||
library, a plugin, or a stray ``print()`` -- would silently corrupt the output.
|
||||
|
||||
This module enforces that guarantee at the operating system level. It saves a
|
||||
private duplicate of the real stdout and points file descriptor 1 at standard
|
||||
error, so that anything that writes to stdout lands harmlessly on stderr. Only
|
||||
OCRmyPDF's final "produce the PDF" step writes to the preserved real stdout, via
|
||||
:func:`get_protected_stdout_fd`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
|
||||
_lock = threading.Lock()
|
||||
_saved_fd: int | None = None
|
||||
_active = False
|
||||
|
||||
|
||||
def protect_stdout() -> bool:
|
||||
"""Redirect file descriptor 1 to stderr and preserve the real stdout.
|
||||
|
||||
After this call, any write to file descriptor 1 -- including ``print()`` and
|
||||
writes from third-party C libraries -- is redirected to standard error and
|
||||
cannot corrupt the real standard output. The real stdout is preserved on a
|
||||
private file descriptor available from :func:`get_protected_stdout_fd`.
|
||||
|
||||
This mutates process-global state and affects the whole process. It must be
|
||||
called once, early, before any plugins are loaded or any worker
|
||||
process/thread is started, so that all of them inherit the redirected
|
||||
descriptor.
|
||||
|
||||
Returns:
|
||||
True if protection was installed (or was already active). False if
|
||||
stdout is not backed by a real OS file descriptor -- for example under
|
||||
a test harness that captures stdout -- in which case nothing is changed.
|
||||
"""
|
||||
global _saved_fd, _active
|
||||
with _lock:
|
||||
if _active:
|
||||
return True
|
||||
try:
|
||||
fd1 = sys.stdout.fileno()
|
||||
except (AttributeError, OSError, ValueError):
|
||||
# stdout is not backed by a real file descriptor (e.g. captured by
|
||||
# a test harness or replaced with an in-memory stream).
|
||||
return False
|
||||
try:
|
||||
sys.stdout.flush()
|
||||
saved = os.dup(fd1)
|
||||
os.dup2(2, fd1) # point stdout at stderr
|
||||
except OSError:
|
||||
return False
|
||||
_saved_fd = saved
|
||||
_active = True
|
||||
return True
|
||||
|
||||
|
||||
def get_protected_stdout_fd() -> int | None:
|
||||
"""Return the preserved real stdout file descriptor, or None if inactive."""
|
||||
return _saved_fd if _active else None
|
||||
|
||||
|
||||
def protected_stdout_isatty() -> bool | None:
|
||||
"""Whether the preserved real stdout is a terminal.
|
||||
|
||||
Returns None if protection is not active, in which case the caller should
|
||||
fall back to ``sys.stdout.isatty()``. When protection is active,
|
||||
``sys.stdout`` reports the terminal status of stderr (its descriptor was
|
||||
redirected), so this consults the saved real-stdout descriptor instead.
|
||||
"""
|
||||
if not _active or _saved_fd is None:
|
||||
return None
|
||||
return os.isatty(_saved_fd)
|
||||
+63
-13
@@ -10,15 +10,18 @@ import logging
|
||||
import os
|
||||
import sys
|
||||
from collections.abc import Sequence
|
||||
from collections.abc import Set as AbstractSet
|
||||
from pathlib import Path
|
||||
from shutil import copyfileobj
|
||||
from typing import BinaryIO, cast
|
||||
|
||||
import pikepdf
|
||||
|
||||
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
from ocrmypdf._exec import unpaper
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._options import OcrOptions, ProcessingMode
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager
|
||||
from ocrmypdf._stdoutprotect import protected_stdout_isatty
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
InputFileError,
|
||||
@@ -47,7 +50,7 @@ def check_platform() -> None:
|
||||
|
||||
|
||||
def check_options_languages(
|
||||
options: OcrOptions, ocr_engine_languages: list[str]
|
||||
options: OcrOptions, ocr_engine_languages: AbstractSet[str]
|
||||
) -> None:
|
||||
# Check for blocked languages first, before checking if they're installed
|
||||
DENIED_LANGUAGES = {'equ', 'osd'}
|
||||
@@ -91,7 +94,15 @@ def check_options_sidecar(options: OcrOptions) -> None:
|
||||
raise BadArgsError(
|
||||
"--sidecar filename needed when output file is /dev/null or NUL."
|
||||
)
|
||||
options.sidecar = options.output_file + '.txt'
|
||||
elif not isinstance(options.output_file, str | Path):
|
||||
# The '\0' sentinel is only ever set by the CLI, which always
|
||||
# supplies output_file as a plain path - not a stream. If this
|
||||
# somehow fires, the caller mixed a CLI-only sentinel with the
|
||||
# stream-based API.
|
||||
raise BadArgsError(
|
||||
"--sidecar filename needed when output file is not a path."
|
||||
)
|
||||
options.sidecar = os.fspath(options.output_file) + '.txt'
|
||||
if options.sidecar == options.input_file or options.sidecar == options.output_file:
|
||||
raise BadArgsError(
|
||||
"--sidecar file must be different from the input and output files"
|
||||
@@ -118,8 +129,36 @@ def check_options_preprocessing(options: OcrOptions) -> None:
|
||||
)
|
||||
|
||||
|
||||
def check_options_strip(options: OcrOptions) -> None:
|
||||
"""Reject options that cannot apply in strip mode.
|
||||
|
||||
``--mode strip`` removes the OCR text layer in place without rasterizing or
|
||||
running OCR, so image-processing and OCR-output options have no effect.
|
||||
"""
|
||||
if options.mode != ProcessingMode.strip_text:
|
||||
return
|
||||
incompatible = {
|
||||
'--deskew': options.deskew,
|
||||
'--clean': options.clean,
|
||||
'--clean-final': options.clean_final,
|
||||
'--remove-background': options.remove_background,
|
||||
'--rotate-pages': options.rotate_pages,
|
||||
'--oversample': options.oversample,
|
||||
'--remove-vectors': options.remove_vectors,
|
||||
'--sidecar': options.sidecar,
|
||||
}
|
||||
used = sorted(name for name, value in incompatible.items() if value)
|
||||
if used:
|
||||
raise BadArgsError(
|
||||
"--mode strip removes the OCR text layer without rasterizing or "
|
||||
"running OCR, so these options have no effect and are not allowed: "
|
||||
f"{', '.join(used)}"
|
||||
)
|
||||
|
||||
|
||||
def _check_plugin_invariant_options(options: OcrOptions) -> None:
|
||||
check_platform()
|
||||
check_options_strip(options)
|
||||
check_options_sidecar(options)
|
||||
check_options_preprocessing(options)
|
||||
|
||||
@@ -161,28 +200,32 @@ def create_input_file(options: OcrOptions, work_folder: Path) -> tuple[Path, str
|
||||
# stdin
|
||||
log.info('reading file from standard input')
|
||||
target = work_folder / 'stdin'
|
||||
with open(target, 'wb') as stream_buffer:
|
||||
with target.open('wb') as stream_buffer:
|
||||
copyfileobj(sys.stdin.buffer, stream_buffer)
|
||||
return target, "stdin"
|
||||
elif hasattr(options.input_file, 'readable'):
|
||||
if not options.input_file.readable():
|
||||
input_stream = cast(BinaryIO, options.input_file)
|
||||
if not input_stream.readable():
|
||||
raise InputFileError("Input file stream is not readable")
|
||||
log.info('reading file from input stream')
|
||||
target = work_folder / 'stream'
|
||||
with open(target, 'wb') as stream_buffer:
|
||||
copyfileobj(options.input_file, stream_buffer)
|
||||
with target.open('wb') as stream_buffer:
|
||||
copyfileobj(input_stream, stream_buffer)
|
||||
return target, "stream"
|
||||
else:
|
||||
# The branches above already ruled out the stdin sentinel and
|
||||
# stream-like objects, so this must be a filesystem path.
|
||||
assert isinstance(options.input_file, str | bytes | os.PathLike)
|
||||
try:
|
||||
target = work_folder / 'origin'
|
||||
safe_symlink(options.input_file, target)
|
||||
return target, os.fspath(options.input_file)
|
||||
return target, os.fsdecode(options.input_file)
|
||||
except FileNotFoundError as e:
|
||||
msg = f"File not found - {options.input_file}"
|
||||
msg = f"File not found - {os.fsdecode(options.input_file)}"
|
||||
if running_in_docker(): # pragma: no cover
|
||||
msg += (
|
||||
"\nDocker cannot access your working directory unless you "
|
||||
"explicitly share it with the Docker container and set up"
|
||||
"explicitly share it with the Docker container and set up "
|
||||
"permissions correctly.\n"
|
||||
"You may find it easier to use stdin/stdout:"
|
||||
"\n"
|
||||
@@ -203,7 +246,13 @@ def create_input_file(options: OcrOptions, work_folder: Path) -> tuple[Path, str
|
||||
|
||||
def check_requested_output_file(options: OcrOptions) -> None:
|
||||
if options.output_file == '-':
|
||||
if sys.stdout.isatty():
|
||||
# When stdout protection is active, fd 1 has been redirected to stderr,
|
||||
# so sys.stdout.isatty() would report stderr's status. Consult the
|
||||
# preserved real stdout instead, falling back when protection is off.
|
||||
is_tty = protected_stdout_isatty()
|
||||
if is_tty is None:
|
||||
is_tty = sys.stdout.isatty()
|
||||
if is_tty:
|
||||
raise BadArgsError(
|
||||
"Output was set to stdout '-' but it looks like stdout "
|
||||
"is connected to a terminal. Please redirect stdout to a "
|
||||
@@ -214,7 +263,8 @@ def check_requested_output_file(options: OcrOptions) -> None:
|
||||
raise OutputFileAccessError("Output stream is not writable")
|
||||
elif not is_file_writable(options.output_file):
|
||||
raise OutputFileAccessError(
|
||||
f"Output file location ({options.output_file}) is not a writable file."
|
||||
f"Output file location ({os.fsdecode(options.output_file)}) is not a "
|
||||
"writable file."
|
||||
)
|
||||
|
||||
if (
|
||||
@@ -224,7 +274,7 @@ def check_requested_output_file(options: OcrOptions) -> None:
|
||||
and Path(str(options.output_file)).exists()
|
||||
):
|
||||
raise OutputFileAccessError(
|
||||
f"Output file already exists: {options.output_file}\n"
|
||||
f"Output file already exists: {os.fsdecode(options.output_file)}\n"
|
||||
"To overwrite it, omit the --no-overwrite / -n option."
|
||||
)
|
||||
|
||||
|
||||
@@ -10,9 +10,8 @@ import os
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import pluggy
|
||||
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -20,9 +19,8 @@ log = logging.getLogger(__name__)
|
||||
class ValidationCoordinator:
|
||||
"""Coordinates validation across plugin models and core options."""
|
||||
|
||||
def __init__(self, plugin_manager: pluggy.PluginManager):
|
||||
def __init__(self, plugin_manager: OcrmypdfPluginManager):
|
||||
self.plugin_manager = plugin_manager
|
||||
self.registry = getattr(plugin_manager, '_option_registry', None)
|
||||
|
||||
def validate_all_options(self, options: OcrOptions) -> None:
|
||||
"""Run comprehensive validation on all options.
|
||||
@@ -110,13 +108,18 @@ class ValidationCoordinator:
|
||||
)
|
||||
|
||||
# Validate output type compatibility
|
||||
if options.output_type == 'none' and str(options.output_file) not in (
|
||||
output_file_display = (
|
||||
os.fsdecode(options.output_file)
|
||||
if isinstance(options.output_file, bytes)
|
||||
else str(options.output_file)
|
||||
)
|
||||
if options.output_type == 'none' and output_file_display not in (
|
||||
os.devnull,
|
||||
'-',
|
||||
):
|
||||
raise ValueError(
|
||||
"Since you specified `--output-type none`, the output file "
|
||||
f"{options.output_file} cannot be produced. Set the output file to "
|
||||
f"{output_file_display} cannot be produced. Set the output file to "
|
||||
"`-` to suppress this message."
|
||||
)
|
||||
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
__version__ = "17.4.2"
|
||||
__version__ = "17.9.0"
|
||||
|
||||
+62
-5
@@ -50,12 +50,15 @@ from pathlib import Path
|
||||
from typing import BinaryIO, overload
|
||||
from warnings import warn
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
from ocrmypdf._logging import PageNumberFilter
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._pipelines.hocr_to_ocr_pdf import run_hocr_to_ocr_pdf_pipeline
|
||||
from ocrmypdf._pipelines.ocr import run_pipeline, run_pipeline_cli
|
||||
from ocrmypdf._pipelines.pdf_to_hocr import run_hocr_pipeline
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager, get_plugin_manager
|
||||
from ocrmypdf._stdoutprotect import protect_stdout
|
||||
from ocrmypdf._validation import check_options
|
||||
from ocrmypdf.cli import ArgumentParser, get_parser
|
||||
from ocrmypdf.exceptions import ExitCode
|
||||
@@ -105,7 +108,7 @@ def setup_plugin_infrastructure(
|
||||
plugin_manager = get_plugin_manager(plugins)
|
||||
|
||||
# Initialize plugins (pass the underlying pluggy manager)
|
||||
plugin_manager.initialize(plugin_manager=plugin_manager.pluggy)
|
||||
plugin_manager.initialize(plugin_manager=plugin_manager.pluggy_manager)
|
||||
|
||||
# Initialize plugin option registry
|
||||
from ocrmypdf._plugin_registry import PluginOptionRegistry
|
||||
@@ -114,7 +117,7 @@ def setup_plugin_infrastructure(
|
||||
|
||||
# Let plugins register their option models
|
||||
option_models = plugin_manager.register_options()
|
||||
all_plugin_models: dict[str, type] = {}
|
||||
all_plugin_models: dict[str, type[BaseModel]] = {}
|
||||
for plugin_options in option_models:
|
||||
if plugin_options: # Skip None returns
|
||||
for namespace, model_class in plugin_options.items():
|
||||
@@ -233,6 +236,37 @@ def configure_logging(
|
||||
return log
|
||||
|
||||
|
||||
def configure_stdout_protection() -> bool:
|
||||
"""Protect the process's real standard output from corruption.
|
||||
|
||||
When OCRmyPDF writes its final PDF to standard output (``output_file='-'``),
|
||||
the bytes on stdout must be exactly the PDF and nothing else. By default
|
||||
OCRmyPDF relies on no in-process code -- third party libraries, plugins, or
|
||||
stray ``print()`` calls -- ever writing to stdout. This function makes that
|
||||
guarantee real: it redirects file descriptor 1 to standard error and
|
||||
preserves a private copy of the real stdout, so that any accidental write to
|
||||
stdout lands harmlessly on stderr while OCRmyPDF still emits its final PDF to
|
||||
the preserved descriptor.
|
||||
|
||||
This is the same protection the ``ocrmypdf`` command line program installs.
|
||||
It is optional for API users and works like :func:`configure_logging`: call
|
||||
it before :func:`ocr` if you want command-line-like behavior. It must be
|
||||
called once, early -- before any plugins are loaded or any worker
|
||||
process/thread is started -- so that they inherit the redirected descriptor.
|
||||
|
||||
Because it mutates process-global file descriptors and affects the entire
|
||||
process, applications that manage their own standard output (for example,
|
||||
a long-lived service that calls :func:`ocr` in-process) should **not** call
|
||||
this function.
|
||||
|
||||
Returns:
|
||||
True if protection was installed (or was already active). False if
|
||||
stdout is not backed by a real operating system file descriptor, in
|
||||
which case nothing is changed.
|
||||
"""
|
||||
return protect_stdout()
|
||||
|
||||
|
||||
def _check_no_conflicting_ocr_params(
|
||||
locals_dict: dict,
|
||||
kwargs: dict,
|
||||
@@ -311,6 +345,23 @@ def _remap_language_to_languages(options_kwargs: dict) -> None:
|
||||
del options_kwargs['language']
|
||||
|
||||
|
||||
def _remap_jpg_quality_to_jpeg_quality(options_kwargs: dict) -> None:
|
||||
"""Map the deprecated 'jpg_quality' parameter to 'jpeg_quality'.
|
||||
|
||||
'jpg_quality' was the original API parameter name. 'jpeg_quality' is the
|
||||
canonical OcrOptions field, matching the primary --jpeg-quality CLI flag.
|
||||
Prefer an explicitly-given 'jpeg_quality' if both are set.
|
||||
"""
|
||||
if 'jpg_quality' not in options_kwargs:
|
||||
return
|
||||
old_value = options_kwargs.pop('jpg_quality')
|
||||
if old_value is None:
|
||||
return
|
||||
warn("ocrmypdf.ocr(jpg_quality=...) is deprecated, use jpeg_quality= instead.")
|
||||
if options_kwargs.get('jpeg_quality') is None:
|
||||
options_kwargs['jpeg_quality'] = old_value
|
||||
|
||||
|
||||
def create_options(
|
||||
*, input_file: PathOrIO, output_file: PathOrIO, parser: ArgumentParser, **kwargs
|
||||
) -> OcrOptions:
|
||||
@@ -335,6 +386,9 @@ def create_options(
|
||||
# Map API parameter 'language' to OcrOptions field 'languages'
|
||||
_remap_language_to_languages(options_kwargs)
|
||||
|
||||
# Map deprecated 'jpg_quality' parameter to 'jpeg_quality'
|
||||
_remap_jpg_quality_to_jpeg_quality(options_kwargs)
|
||||
|
||||
# Set input and output files
|
||||
options_kwargs['input_file'] = input_file
|
||||
options_kwargs['output_file'] = output_file
|
||||
@@ -414,7 +468,8 @@ def ocr(
|
||||
redo_ocr: bool | None = None,
|
||||
skip_big: float | None = None,
|
||||
optimize: int | None = None,
|
||||
jpg_quality: int | None = None,
|
||||
jpeg_quality: int | None = None,
|
||||
jpg_quality: int | None = None, # Deprecated, use jpeg_quality instead
|
||||
png_quality: int | None = None,
|
||||
jbig2_lossy: bool | None = None,
|
||||
jbig2_page_group_size: int | None = None,
|
||||
@@ -477,7 +532,8 @@ def ocr( # noqa: D417
|
||||
redo_ocr: bool | None = None, # Legacy, use mode='redo' instead
|
||||
skip_big: float | None = None,
|
||||
optimize: int | None = None,
|
||||
jpg_quality: int | None = None,
|
||||
jpeg_quality: int | None = None,
|
||||
jpg_quality: int | None = None, # Deprecated, use jpeg_quality instead
|
||||
png_quality: int | None = None,
|
||||
jbig2_lossy: bool | None = None, # Deprecated, ignored
|
||||
jbig2_page_group_size: int | None = None, # Deprecated, ignored
|
||||
@@ -849,7 +905,7 @@ def _hocr_to_ocr_pdf( # noqa: D417
|
||||
jobs: int | None = None,
|
||||
use_threads: bool | None = None,
|
||||
optimize: int | None = None,
|
||||
jpg_quality: int | None = None,
|
||||
jpeg_quality: int | None = None,
|
||||
png_quality: int | None = None,
|
||||
jbig2_lossy: bool | None = None, # Deprecated, ignored
|
||||
jbig2_page_group_size: int | None = None, # Deprecated, ignored
|
||||
@@ -965,6 +1021,7 @@ __all__ = [
|
||||
'Verbosity',
|
||||
'check_options',
|
||||
'configure_logging',
|
||||
'configure_stdout_protection',
|
||||
'create_options',
|
||||
'get_parser',
|
||||
'get_plugin_manager',
|
||||
|
||||
@@ -27,9 +27,12 @@ from ocrmypdf.exceptions import InputFileError
|
||||
from ocrmypdf.helpers import remove_all_log_handlers
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from logging import LogRecord
|
||||
from typing import TypeAlias
|
||||
|
||||
Queue: TypeAlias = multiprocessing.queues.Queue | queue.Queue
|
||||
Queue: TypeAlias = (
|
||||
multiprocessing.queues.Queue[LogRecord | None] | queue.Queue[LogRecord | None]
|
||||
)
|
||||
UserInit: TypeAlias = Callable[[], None]
|
||||
WorkerInit: TypeAlias = Callable[[Queue, UserInit, int], None]
|
||||
|
||||
@@ -99,7 +102,9 @@ def thread_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
||||
return
|
||||
|
||||
|
||||
def setup_executor(use_threads: bool) -> tuple[Queue, Executor, WorkerInit]:
|
||||
def setup_executor(
|
||||
use_threads: bool,
|
||||
) -> tuple[Queue, FuturesExecutorClass, WorkerInit]:
|
||||
if not use_threads:
|
||||
# Some execution environments like AWS Lambda and Termux do not support
|
||||
# semaphores. Check if semaphore support is available, and if not, fall back
|
||||
@@ -112,6 +117,8 @@ def setup_executor(use_threads: bool) -> tuple[Queue, Executor, WorkerInit]:
|
||||
except ImportError:
|
||||
use_threads = True
|
||||
|
||||
loq_queue: Queue
|
||||
executor_class: FuturesExecutorClass
|
||||
if use_threads:
|
||||
loq_queue = queue.Queue(-1)
|
||||
executor_class = ThreadPoolExecutor
|
||||
|
||||
@@ -44,6 +44,30 @@ class PdfaImageCompression(StrEnum):
|
||||
LOSSLESS = 'lossless'
|
||||
|
||||
|
||||
def _resolve_auto_compression(
|
||||
compression: PdfaImageCompression, optimize_level: int
|
||||
) -> PdfaImageCompression:
|
||||
"""Resolve 'auto' image compression based on the optimization level.
|
||||
|
||||
At ``-O0`` (no optimization) ``auto`` maps to ``lossless`` so Ghostscript
|
||||
will not transcode lossless images to JPEG during PDF/A generation. At all
|
||||
other levels ``auto`` defers to Ghostscript's heuristic, which may
|
||||
recompress images lossily.
|
||||
|
||||
``-O1`` is a historical exception: although it is otherwise a
|
||||
lossless-only optimization level, coercing ``auto`` to ``lossless`` there
|
||||
can bloat output substantially (Ghostscript's heuristic often picks JPEG
|
||||
for photographic content), so the default is left alone for backwards
|
||||
compatibility. Users who want guaranteed lossless image handling at any
|
||||
level can pass ``--pdfa-image-compression=lossless`` explicitly.
|
||||
|
||||
Explicit ``jpeg`` and ``lossless`` choices are always respected.
|
||||
"""
|
||||
if compression == PdfaImageCompression.AUTO and optimize_level == 0:
|
||||
return PdfaImageCompression.LOSSLESS
|
||||
return compression
|
||||
|
||||
|
||||
class GhostscriptOptions(BaseModel):
|
||||
"""Options specific to Ghostscript operations."""
|
||||
|
||||
@@ -54,6 +78,27 @@ class GhostscriptOptions(BaseModel):
|
||||
pdfa_image_compression: Annotated[
|
||||
PdfaImageCompression, Field(description="PDF/A image compression method")
|
||||
] = PdfaImageCompression.AUTO
|
||||
jpeg_quality: Annotated[
|
||||
int | None,
|
||||
Field(
|
||||
ge=0,
|
||||
le=100,
|
||||
description=(
|
||||
"JPEG quality (0-100) for Ghostscript image recompression during "
|
||||
"PDF/A generation; None uses Ghostscript's default."
|
||||
),
|
||||
),
|
||||
] = None
|
||||
jpeg_maxdpi: Annotated[
|
||||
int | None,
|
||||
Field(
|
||||
ge=1,
|
||||
description=(
|
||||
"Maximum DPI for Ghostscript image downsampling during PDF/A "
|
||||
"generation."
|
||||
),
|
||||
),
|
||||
] = None
|
||||
|
||||
@classmethod
|
||||
def add_arguments_to_parser(cls, parser, namespace: str = 'ghostscript'):
|
||||
@@ -78,14 +123,48 @@ class GhostscriptOptions(BaseModel):
|
||||
choices=[pc.value for pc in PdfaImageCompression],
|
||||
default=PdfaImageCompression.AUTO.value,
|
||||
help="Specify how to compress images in the output PDF/A. 'auto' lets "
|
||||
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
|
||||
"OCRmyPDF decide: at -O0 it uses lossless image compression so "
|
||||
"Ghostscript does not transcode lossless images to JPEG; at -O1 and "
|
||||
"above it defers to Ghostscript's heuristic, which may recompress "
|
||||
"images lossily. 'jpeg' changes all grayscale and color images to "
|
||||
"JPEG compression. 'lossless' uses PNG-style lossless compression "
|
||||
"for all images. Monochrome images are always compressed using a "
|
||||
"for non-JPEG images and passes existing JPEGs through unchanged "
|
||||
"(re-encoding them losslessly would only inflate them). Monochrome "
|
||||
"images are always compressed using a "
|
||||
"lossless codec. Compression settings "
|
||||
"are applied to all pages, including those for which OCR was "
|
||||
"skipped. Not supported for --output-type=pdf ; that setting "
|
||||
"preserves the original compression of all images.",
|
||||
)
|
||||
gs.add_argument(
|
||||
'--ghostscript-jpeg-quality',
|
||||
type=int,
|
||||
metavar='Q',
|
||||
default=None,
|
||||
dest=f'{namespace}_jpeg_quality',
|
||||
help=(
|
||||
"Advanced: Set Ghostscript's -dJPEGQ for images that Ghostscript "
|
||||
"transcodes to JPEG during PDF/A generation. 0 is maximum "
|
||||
"compression; 100 is best quality. If omitted, Ghostscript's "
|
||||
"default is used. This only affects images Ghostscript chooses "
|
||||
"to recompress; for general JPEG quality tuning prefer "
|
||||
"--jpeg-quality, which is applied by the OCRmyPDF optimizer."
|
||||
),
|
||||
)
|
||||
gs.add_argument(
|
||||
'--ghostscript-jpeg-maxdpi',
|
||||
type=int,
|
||||
metavar='DPI',
|
||||
default=None,
|
||||
dest=f'{namespace}_jpeg_maxdpi',
|
||||
help=(
|
||||
"Advanced: Force Ghostscript to downsample color, grayscale, "
|
||||
"and monochrome images in PDF/A output to the given maximum DPI. "
|
||||
"Reducing JPEG quality usually gives better results than "
|
||||
"downsampling at the same file size, and can degrade quality "
|
||||
"of high-resolution monochrome masks."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@hookimpl
|
||||
@@ -177,6 +256,8 @@ def rasterize_pdf_page(
|
||||
# Let pypdfium handle it (it will error in check_options if unavailable)
|
||||
return None
|
||||
|
||||
log.debug("Rasterizing page %d with the Ghostscript rasterizer", pageno)
|
||||
|
||||
ghostscript.rasterize_pdf(
|
||||
input_file,
|
||||
output_file,
|
||||
@@ -347,11 +428,18 @@ def generate_pdfa(
|
||||
if output_type == 'pdfa':
|
||||
output_type = 'pdfa-2'
|
||||
|
||||
compression = _resolve_auto_compression(
|
||||
context.options.ghostscript.pdfa_image_compression,
|
||||
context.options.optimize,
|
||||
)
|
||||
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[pdfmark, *pdf_pages],
|
||||
output_file=output_file,
|
||||
compression=context.options.ghostscript.pdfa_image_compression,
|
||||
compression=compression,
|
||||
color_conversion_strategy=context.options.ghostscript.color_conversion_strategy,
|
||||
jpeg_quality=context.options.ghostscript.jpeg_quality,
|
||||
jpeg_maxdpi=context.options.ghostscript.jpeg_maxdpi,
|
||||
pdf_version=pdf_version,
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=progressbar_class,
|
||||
|
||||
@@ -48,27 +48,18 @@ def _open_pdf_document(input_file: Path):
|
||||
return pdfium.PdfDocument(input_file)
|
||||
|
||||
|
||||
def _calculate_mediabox_crop(page) -> tuple[float, float, float, float]:
|
||||
"""Calculate crop values to expand rendering from CropBox to MediaBox.
|
||||
def _expand_cropbox_to_mediabox(page) -> None:
|
||||
"""Set the page's CropBox to its MediaBox so PDFium renders the full page.
|
||||
|
||||
By default pypdfium2 renders to the CropBox. To render the full MediaBox,
|
||||
we need negative crop values to expand the rendering area.
|
||||
|
||||
Returns:
|
||||
Tuple of (left, bottom, right, top) crop values. Negative values
|
||||
expand the rendering area beyond the CropBox to the MediaBox.
|
||||
PDFium renders to the CropBox by default. Negative ``crop`` values to
|
||||
``render()`` are not supported and only pad the output canvas without
|
||||
expanding the rendered area — content outside the CropBox is clipped.
|
||||
The supported approach is to widen the CropBox in memory before rendering.
|
||||
The document is never saved back to disk, so this mutation is local.
|
||||
See https://github.com/ocrmypdf/OCRmyPDF/issues/1685.
|
||||
"""
|
||||
mediabox = page.get_mediabox() # (left, bottom, right, top)
|
||||
cropbox = page.get_cropbox() # (left, bottom, right, top), defaults to mediabox
|
||||
|
||||
# Calculate how much to expand from cropbox to mediabox
|
||||
# Negative values = expand, positive = shrink
|
||||
return (
|
||||
mediabox[0] - cropbox[0], # Expand left
|
||||
mediabox[1] - cropbox[1], # Expand bottom
|
||||
cropbox[2] - mediabox[2], # Expand right
|
||||
cropbox[3] - mediabox[3], # Expand top
|
||||
)
|
||||
page.set_cropbox(*mediabox)
|
||||
|
||||
|
||||
def _render_page_to_bitmap(
|
||||
@@ -105,16 +96,20 @@ def _render_page_to_bitmap(
|
||||
# Render the page to a bitmap
|
||||
# The scale parameter controls the resolution
|
||||
# Render in grayscale for mono and gray devices (better input for 1-bit conversion)
|
||||
grayscale = raster_device.lower() in ('pngmono', 'pnggray', 'jpeggray')
|
||||
grayscale = raster_device.lower() in (
|
||||
'pngmono',
|
||||
'pngmonod',
|
||||
'pnggray',
|
||||
'jpeggray',
|
||||
)
|
||||
|
||||
# Calculate crop to render the appropriate box
|
||||
# Default (use_cropbox=False) renders MediaBox for consistency with Ghostscript
|
||||
crop = (0, 0, 0, 0) if use_cropbox else _calculate_mediabox_crop(page)
|
||||
if not use_cropbox:
|
||||
_expand_cropbox_to_mediabox(page)
|
||||
|
||||
bitmap = page.render(
|
||||
scale=scale,
|
||||
rotation=0, # We already set rotation on the page
|
||||
crop=crop,
|
||||
may_draw_forms=True,
|
||||
draw_annots=True,
|
||||
grayscale=grayscale,
|
||||
@@ -167,8 +162,8 @@ def _process_image_for_output(
|
||||
# This ensures pypdfium output matches Ghostscript's native device output
|
||||
raster_device_lower = raster_device.lower()
|
||||
|
||||
if raster_device_lower == 'pngmono':
|
||||
# Convert to 1-bit black and white (matches Ghostscript pngmono device)
|
||||
if raster_device_lower in ('pngmono', 'pngmonod'):
|
||||
# Convert to 1-bit black and white (matches Ghostscript pngmono/pngmonod)
|
||||
if pil_image.mode != '1':
|
||||
if pil_image.mode not in ('L', '1'):
|
||||
pil_image = pil_image.convert('L')
|
||||
@@ -194,7 +189,16 @@ def _process_image_for_output(
|
||||
# pngalpha: keep RGBA as-is
|
||||
|
||||
# Determine output format based on raster_device
|
||||
png_devices = ('png', 'pngmono', 'pnggray', 'png256', 'png16m', 'pngalpha')
|
||||
png_devices = (
|
||||
'png',
|
||||
'pngmono',
|
||||
'pngmonod',
|
||||
'pnggray',
|
||||
'png256',
|
||||
'png16m',
|
||||
'pngalpha',
|
||||
)
|
||||
format_name: Literal['PNG', 'TIFF', 'JPEG']
|
||||
if raster_device_lower in png_devices:
|
||||
format_name = 'PNG'
|
||||
elif raster_device_lower in ('jpeg', 'jpeggray', 'jpg'):
|
||||
@@ -252,6 +256,8 @@ def rasterize_pdf_page(
|
||||
if pdfium is None:
|
||||
return None # Fall back to Ghostscript
|
||||
|
||||
log.debug("Rasterizing page %d with the pypdfium2 rasterizer", pageno)
|
||||
|
||||
# Acquire lock to ensure thread-safe access to pypdfium2
|
||||
with (
|
||||
_pdfium_lock,
|
||||
|
||||
@@ -130,7 +130,7 @@ class TesseractOptions(BaseModel):
|
||||
metavar='PSM',
|
||||
choices=range(0, 14),
|
||||
dest=f'{namespace}_pagesegmode',
|
||||
help="Set Tesseract page segmentation mode (see tesseract --help).",
|
||||
help="Set Tesseract page segmentation mode (see tesseract --help-extra).",
|
||||
)
|
||||
|
||||
tess.add_argument(
|
||||
@@ -168,7 +168,7 @@ class TesseractOptions(BaseModel):
|
||||
tess.add_argument(
|
||||
f'--{namespace}-timeout',
|
||||
default=180.0,
|
||||
type=numeric(float, 0),
|
||||
type=numeric(float, 0.0),
|
||||
metavar='SECONDS',
|
||||
dest=f'{namespace}_timeout',
|
||||
help=(
|
||||
@@ -183,7 +183,7 @@ class TesseractOptions(BaseModel):
|
||||
tess.add_argument(
|
||||
f'--{namespace}-non-ocr-timeout',
|
||||
default=180.0,
|
||||
type=numeric(float, 0),
|
||||
type=numeric(float, 0.0),
|
||||
metavar='SECONDS',
|
||||
dest=f'{namespace}_non_ocr_timeout',
|
||||
help=(
|
||||
|
||||
+18
-9
@@ -327,7 +327,11 @@ Online documentation is located at:
|
||||
"'default' errors if text is found. "
|
||||
"'force' rasterizes all content and runs OCR (same as --force-ocr). "
|
||||
"'skip' skips pages with existing text (same as --skip-text). "
|
||||
"'redo' re-OCRs pages, replacing old invisible text (same as --redo-ocr).",
|
||||
"'redo' re-OCRs pages, replacing old invisible text (same as --redo-ocr). "
|
||||
"'strip' removes the invisible OCR text layer without rasterizing or "
|
||||
"running OCR, producing a smaller file; only text drawn as invisible "
|
||||
"(render mode 3) is removed, so text from some OCR engines cannot be "
|
||||
"removed this way.",
|
||||
)
|
||||
# Legacy flags for backward compatibility - these set the mode internally
|
||||
ocrsettings.add_argument(
|
||||
@@ -358,7 +362,7 @@ Online documentation is located at:
|
||||
)
|
||||
ocrsettings.add_argument(
|
||||
'--skip-big',
|
||||
type=numeric(float, 0, 5000),
|
||||
type=numeric(float, 0.0, 5000.0),
|
||||
metavar='MPixels',
|
||||
help="Skip OCR on pages larger than the specified amount of megapixels, "
|
||||
"but include skipped pages in final output",
|
||||
@@ -387,13 +391,14 @@ Online documentation is located at:
|
||||
type=str,
|
||||
help=(
|
||||
"Limit OCR to the specified pages (ranges or comma separated), "
|
||||
"skipping others"
|
||||
"skipping others. The token 'end' is an alias for the last page, "
|
||||
"so e.g. '3-end' OCRs from page 3 to the last page."
|
||||
),
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--max-image-mpixels',
|
||||
action='store',
|
||||
type=numeric(float, 0),
|
||||
type=numeric(float, 0.0),
|
||||
metavar='MPixels',
|
||||
help="Set maximum number of megapixels to unpack before treating an image as a "
|
||||
"decompression bomb",
|
||||
@@ -422,21 +427,25 @@ Online documentation is located at:
|
||||
'--rasterizer',
|
||||
choices=['auto', 'ghostscript', 'pypdfium'],
|
||||
default='auto',
|
||||
help="Choose PDF page rasterizer. 'auto' prefers pypdfium when available, "
|
||||
"falling back to Ghostscript. 'pypdfium' is faster but requires the "
|
||||
"pypdfium2 package. 'ghostscript' uses the traditional Ghostscript rasterizer.",
|
||||
help="Choose PDF page rasterizer. 'auto' (the default) prefers pypdfium2 "
|
||||
"when the pypdfium2 package is installed, falling back to Ghostscript "
|
||||
"otherwise. pypdfium2 anti-aliases page content and generally produces "
|
||||
"better input for OCR than Ghostscript 10.x, which can render aliased "
|
||||
"glyphs that OCR misreads as extra word breaks. 'pypdfium' forces the "
|
||||
"pypdfium2 rasterizer (requires the pypdfium2 package); 'ghostscript' "
|
||||
"forces the traditional Ghostscript rasterizer.",
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--rotate-pages-threshold',
|
||||
default=DEFAULT_ROTATE_PAGES_THRESHOLD,
|
||||
type=numeric(float, 0, 1000),
|
||||
type=numeric(float, 0.0, 1000.0),
|
||||
metavar='CONFIDENCE',
|
||||
help="Only rotate pages when confidence is above this value (arbitrary "
|
||||
"units reported by tesseract)",
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--fast-web-view',
|
||||
type=numeric(float, 0),
|
||||
type=numeric(float, 0.0),
|
||||
default=1.0,
|
||||
metavar="MEGABYTES",
|
||||
help="If the size of file is more than this threshold (in MB), then "
|
||||
|
||||
+70
-10
@@ -139,14 +139,74 @@ class TaggedPDFError(InputFileError):
|
||||
)
|
||||
|
||||
|
||||
class ColorConversionNeededError(BadArgsError):
|
||||
"""PDF needs color conversion."""
|
||||
class NonEmbeddedFontsError(InputFileError):
|
||||
"""Input has non-embedded CID fonts that PDF/A conversion would corrupt.
|
||||
|
||||
message = dedent(
|
||||
"""\
|
||||
The input PDF has an unusual color space. Use
|
||||
--color-conversion-strategy to convert to a common color space
|
||||
such as RGB, or use --output-type pdf to skip PDF/A conversion
|
||||
and retain the original color space.
|
||||
"""
|
||||
)
|
||||
PDF/A requires all fonts to be embedded. Ghostscript substitutes and embeds
|
||||
a replacement for non-embedded CID (CJK) fonts, which corrupts the
|
||||
character-to-Unicode mapping and silently destroys an existing text layer
|
||||
(commonly an Adobe Acrobat CJK OCR layer). OCRmyPDF refuses to produce such
|
||||
output rather than damage the user's data
|
||||
(see https://github.com/ocrmypdf/OCRmyPDF/issues/1561).
|
||||
"""
|
||||
|
||||
def __init__(self, fonts: set[str]):
|
||||
"""Build guidance naming the offending fonts."""
|
||||
super().__init__()
|
||||
font_list = ', '.join(sorted(fonts))
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
The input PDF contains non-embedded CID (character ID) fonts: {font_list}.
|
||||
|
||||
PDF/A requires all fonts to be embedded. Converting to PDF/A would
|
||||
make Ghostscript substitute and embed replacement fonts, which
|
||||
corrupts CID (e.g. CJK/Chinese-Japanese-Korean) text and silently
|
||||
destroys an existing text layer such as one produced by Adobe Acrobat.
|
||||
|
||||
Use --output-type pdf to keep the existing text layer intact without
|
||||
PDF/A conversion, or --force-ocr to discard the existing layer and
|
||||
rebuild it with embedded fonts.
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
class ColorConversionNeededError(BadArgsError):
|
||||
"""PDF needs color conversion to a standard color space.
|
||||
|
||||
Ghostscript reported a DeviceN colorspace with an inappropriate alternate.
|
||||
The resulting PDF/A is liable to render incorrectly (often blank) in some
|
||||
viewers such as Adobe Reader, so the colorspace must be normalized to a
|
||||
common one. RGB, CMYK, and Gray are known to work; LeaveColorUnchanged
|
||||
performs no conversion and UseDeviceIndependentColor does not resolve the
|
||||
problem (see https://github.com/ocrmypdf/OCRmyPDF/issues/1187).
|
||||
"""
|
||||
|
||||
# Strategies that can normalize an unusual DeviceN colorspace into one that
|
||||
# PDF/A viewers render correctly.
|
||||
_effective_strategies = "RGB, CMYK, or Gray"
|
||||
|
||||
def __init__(self, color_conversion_strategy: str = "LeaveColorUnchanged"):
|
||||
"""Build guidance tailored to the conversion strategy that was used."""
|
||||
super().__init__()
|
||||
if color_conversion_strategy == "LeaveColorUnchanged":
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
The input PDF has an unusual DeviceN color space that cannot be
|
||||
represented in PDF/A; the output may appear blank in some viewers
|
||||
such as Adobe Reader. Convert it to a common color space with
|
||||
--color-conversion-strategy ({self._effective_strategies}), or use
|
||||
--output-type pdf to skip PDF/A conversion and retain the original
|
||||
color space.
|
||||
"""
|
||||
)
|
||||
else:
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
Color conversion with --color-conversion-strategy
|
||||
{color_conversion_strategy} did not resolve the input PDF's unusual
|
||||
DeviceN color space; the output may appear blank in some viewers
|
||||
such as Adobe Reader. Try a different --color-conversion-strategy
|
||||
({self._effective_strategies}), or use --output-type pdf to skip
|
||||
PDF/A conversion and retain the original color space.
|
||||
"""
|
||||
)
|
||||
|
||||
@@ -10,6 +10,7 @@ This module provides font infrastructure for the fpdf2 PDF renderer. It includes
|
||||
- MultiFontManager: Automatic font selection for multilingual documents
|
||||
- SystemFontProvider: System font discovery
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ocrmypdf.font.font_manager import FontManager
|
||||
@@ -17,6 +18,7 @@ from ocrmypdf.font.font_provider import (
|
||||
BuiltinFontProvider,
|
||||
ChainedFontProvider,
|
||||
FontProvider,
|
||||
GlyphSearchingFontProvider,
|
||||
)
|
||||
from ocrmypdf.font.multi_font_manager import MultiFontManager
|
||||
from ocrmypdf.font.system_font_provider import SystemFontProvider
|
||||
@@ -24,6 +26,7 @@ from ocrmypdf.font.system_font_provider import SystemFontProvider
|
||||
__all__ = [
|
||||
"FontManager",
|
||||
"FontProvider",
|
||||
"GlyphSearchingFontProvider",
|
||||
"BuiltinFontProvider",
|
||||
"ChainedFontProvider",
|
||||
"MultiFontManager",
|
||||
|
||||
@@ -7,7 +7,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Protocol
|
||||
from typing import Protocol, runtime_checkable
|
||||
|
||||
from ocrmypdf.font.font_manager import FontManager
|
||||
|
||||
@@ -52,6 +52,34 @@ class FontProvider(Protocol):
|
||||
...
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class GlyphSearchingFontProvider(Protocol):
|
||||
"""Optional capability: find a font by glyph coverage rather than by name.
|
||||
|
||||
A provider only knows a limited set of logical font names, but it may have
|
||||
access to many more fonts than it can name (e.g. the ~100 script-specific
|
||||
Noto faces macOS installs). Implementing this lets MultiFontManager use
|
||||
them as a last resort instead of falling back to glyphless rendering.
|
||||
|
||||
Providers that do not implement this are used as-is; the capability is
|
||||
detected at runtime with ``isinstance``.
|
||||
"""
|
||||
|
||||
def find_font_with_glyphs(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Find a font that has glyphs for every character in text.
|
||||
|
||||
The returned name must subsequently resolve through ``get_font()``, so
|
||||
that callers can cache the selection by name.
|
||||
|
||||
Args:
|
||||
text: Text the font must fully cover
|
||||
|
||||
Returns:
|
||||
(logical font name, FontManager), or None if no font covers text
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
class BuiltinFontProvider:
|
||||
"""Font provider using builtin fonts from ocrmypdf/data directory."""
|
||||
|
||||
@@ -119,6 +147,18 @@ class BuiltinFontProvider:
|
||||
"""Get the glyphless fallback font."""
|
||||
return self._fonts['Occulta']
|
||||
|
||||
def find_font_with_glyphs(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Find a bundled font that covers text, ignoring glyphless Occulta."""
|
||||
if not text:
|
||||
return None
|
||||
codepoints = {ord(c) for c in text}
|
||||
for name, font in self._fonts.items():
|
||||
if name == 'Occulta':
|
||||
continue
|
||||
if all(font.has_glyph(cp) for cp in codepoints):
|
||||
return name, font
|
||||
return None
|
||||
|
||||
|
||||
class ChainedFontProvider:
|
||||
"""Font provider that tries multiple providers in order.
|
||||
@@ -170,6 +210,25 @@ class ChainedFontProvider:
|
||||
result.append(name)
|
||||
return result
|
||||
|
||||
def find_font_with_glyphs(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Ask each capable provider in turn for a font that covers text.
|
||||
|
||||
Providers that don't implement the search are skipped.
|
||||
|
||||
Args:
|
||||
text: Text the font must fully cover
|
||||
|
||||
Returns:
|
||||
(logical font name, FontManager) from the first provider with a
|
||||
match, or None if no provider found one
|
||||
"""
|
||||
for provider in self.providers:
|
||||
if not isinstance(provider, GlyphSearchingFontProvider):
|
||||
continue
|
||||
if found := provider.find_font_with_glyphs(text):
|
||||
return found
|
||||
return None
|
||||
|
||||
def get_fallback_font(self) -> FontManager:
|
||||
"""Get the glyphless fallback font.
|
||||
|
||||
|
||||
@@ -10,6 +10,7 @@ language hints and glyph coverage analysis.
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
|
||||
from ocrmypdf.font.font_manager import FontManager
|
||||
@@ -17,6 +18,7 @@ from ocrmypdf.font.font_provider import (
|
||||
BuiltinFontProvider,
|
||||
ChainedFontProvider,
|
||||
FontProvider,
|
||||
GlyphSearchingFontProvider,
|
||||
)
|
||||
from ocrmypdf.font.system_font_provider import SystemFontProvider
|
||||
|
||||
@@ -33,9 +35,17 @@ class MultiFontManager:
|
||||
Font selection strategy:
|
||||
1. Try language-preferred font (if language hint available)
|
||||
2. Try fallback fonts in order by glyph coverage
|
||||
3. Fall back to Occulta.ttf (glyphless fallback)
|
||||
3. Ask the provider for any installed font that covers the text
|
||||
4. Fall back to Occulta.ttf (glyphless fallback)
|
||||
"""
|
||||
|
||||
# How many uncoverable characters to name in the missing-font warning
|
||||
MAX_REPORTED_CHARS = 3
|
||||
|
||||
# How many characters of a word to look up individually when composing that
|
||||
# warning; each lookup may scan every font installed on the system
|
||||
MAX_EXAMINED_CHARS = 8
|
||||
|
||||
# Language to font mapping
|
||||
# Keys are ISO 639-2/3 codes or Tesseract language codes
|
||||
LANGUAGE_FONT_MAP = {
|
||||
@@ -54,13 +64,15 @@ class MultiFontManager:
|
||||
'kok': 'NotoSansDevanagari-Regular', # Konkani
|
||||
'bho': 'NotoSansDevanagari-Regular', # Bhojpuri
|
||||
'mai': 'NotoSansDevanagari-Regular', # Maithili
|
||||
# CJK
|
||||
'chi': 'NotoSansCJK-Regular', # Chinese (generic)
|
||||
'zho': 'NotoSansCJK-Regular', # Chinese (ISO 639-3)
|
||||
'chi_sim': 'NotoSansCJK-Regular', # Chinese Simplified (Tesseract)
|
||||
'chi_tra': 'NotoSansCJK-Regular', # Chinese Traditional (Tesseract)
|
||||
'jpn': 'NotoSansCJK-Regular', # Japanese
|
||||
'kor': 'NotoSansCJK-Regular', # Korean
|
||||
# CJK — prefer the family matching the document language, because the
|
||||
# modern per-language Noto fonts are region subsets (e.g. NotoSansSC
|
||||
# lacks Japanese kana). The pan-CJK super font is a shared fallback.
|
||||
'chi': 'NotoSansSC-Regular', # Chinese (generic → Simplified)
|
||||
'zho': 'NotoSansSC-Regular', # Chinese (ISO 639-3)
|
||||
'chi_sim': 'NotoSansSC-Regular', # Chinese Simplified (Tesseract)
|
||||
'chi_tra': 'NotoSansTC-Regular', # Chinese Traditional (Tesseract)
|
||||
'jpn': 'NotoSansJP-Regular', # Japanese
|
||||
'kor': 'NotoSansKR-Regular', # Korean
|
||||
# Thai
|
||||
'tha': 'NotoSansThai-Regular', # Thai
|
||||
# Hebrew
|
||||
@@ -113,7 +125,14 @@ class MultiFontManager:
|
||||
'NotoSans-Regular', # Latin, Greek, Cyrillic
|
||||
'NotoSansArabic-Regular',
|
||||
'NotoSansDevanagari-Regular',
|
||||
# Pan-CJK super font first (full coverage), then the per-language
|
||||
# subsets so a glyph missing from one CJK family is found in another.
|
||||
'NotoSansCJK-Regular',
|
||||
'NotoSansSC-Regular',
|
||||
'NotoSansTC-Regular',
|
||||
'NotoSansHK-Regular',
|
||||
'NotoSansJP-Regular',
|
||||
'NotoSansKR-Regular',
|
||||
'NotoSansThai-Regular',
|
||||
'NotoSansHebrew-Regular',
|
||||
'NotoSansBengali-Regular',
|
||||
@@ -164,6 +183,9 @@ class MultiFontManager:
|
||||
self._selection_cache: dict[tuple[str, str | None], str] = {}
|
||||
# Track whether we've warned about missing fonts (warn once per script)
|
||||
self._warned_scripts: set[str] = set()
|
||||
# Fonts found by glyph coverage rather than by name, tried before
|
||||
# repeating the (expensive) provider search
|
||||
self._discovered_fonts: list[str] = []
|
||||
|
||||
@property
|
||||
def fonts(self) -> dict[str, FontManager]:
|
||||
@@ -199,7 +221,8 @@ class MultiFontManager:
|
||||
Uses a hybrid approach:
|
||||
1. Language-based selection (if language hint available)
|
||||
2. Ordered fallback through available fonts by glyph coverage
|
||||
3. Final fallback to Occulta.ttf (glyphless)
|
||||
3. Provider search over every installed font, by glyph coverage
|
||||
4. Final fallback to Occulta.ttf (glyphless)
|
||||
|
||||
Args:
|
||||
word_text: The text content of the word
|
||||
@@ -224,19 +247,50 @@ class MultiFontManager:
|
||||
if result := self._try_font(preferred, word_text, cache_key):
|
||||
return result
|
||||
|
||||
# Phase 2: Try fallback fonts in order
|
||||
for font_name in self.FALLBACK_FONTS:
|
||||
# Phase 2: Try fallback fonts in order, then anything a previous
|
||||
# coverage search turned up
|
||||
for font_name in [*self.FALLBACK_FONTS, *self._discovered_fonts]:
|
||||
if font_name in tried_fonts:
|
||||
continue
|
||||
tried_fonts.add(font_name)
|
||||
if result := self._try_font(font_name, word_text, cache_key):
|
||||
return result
|
||||
|
||||
# Phase 3: Glyphless fallback (always succeeds)
|
||||
# Phase 3: Ask the provider to search every installed font. The named
|
||||
# families cover common scripts only, but systems ship many more (macOS
|
||||
# installs ~100 Noto faces), and those should be used before giving up
|
||||
# on rendering the text at all. See issue #1722.
|
||||
if found := self._search_font_by_coverage(word_text):
|
||||
font_name, font = found
|
||||
self._selection_cache[cache_key] = font_name
|
||||
return font
|
||||
|
||||
# Phase 4: Glyphless fallback (always succeeds)
|
||||
# Warn if we're falling back for non-ASCII text (likely missing font)
|
||||
self._warn_missing_font(word_text, line_language)
|
||||
self._selection_cache[cache_key] = 'Occulta'
|
||||
return self.font_provider.get_fallback_font()
|
||||
|
||||
def _search_font_by_coverage(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Search the provider for any font covering text, if it supports it.
|
||||
|
||||
Args:
|
||||
text: Text the font must fully cover
|
||||
|
||||
Returns:
|
||||
(font name, FontManager), or None if unsupported or nothing matched
|
||||
"""
|
||||
provider = self.font_provider
|
||||
if not isinstance(provider, GlyphSearchingFontProvider):
|
||||
return None
|
||||
found = provider.find_font_with_glyphs(text)
|
||||
if found is None:
|
||||
return None
|
||||
font_name, _font = found
|
||||
if font_name not in self._discovered_fonts:
|
||||
self._discovered_fonts.append(font_name)
|
||||
return found
|
||||
|
||||
def _warn_missing_font(self, word_text: str, line_language: str | None) -> None:
|
||||
"""Warn user about missing font for non-Latin text.
|
||||
|
||||
@@ -255,22 +309,97 @@ class MultiFontManager:
|
||||
|
||||
self._warned_scripts.add(warn_key)
|
||||
|
||||
if line_language and line_language in self.LANGUAGE_FONT_MAP:
|
||||
font_name = self.LANGUAGE_FONT_MAP[line_language]
|
||||
uncoverable = self._uncoverable_characters(word_text)
|
||||
if not uncoverable:
|
||||
# Every character has a font, but no single font has them all.
|
||||
# Telling the user to install fonts would be wrong advice here.
|
||||
log.warning(
|
||||
"No font found with glyphs for '%s' text. "
|
||||
"Install %s for better rendering. "
|
||||
"See https://fonts.google.com/noto",
|
||||
"Text mixing scripts that no single installed font covers (%r) "
|
||||
"was added as an invisible text layer: it stays searchable and "
|
||||
"copyable, but appears blank when highlighted in a PDF viewer. "
|
||||
"Installing more fonts will not help; OCRmyPDF uses one font "
|
||||
"per word.",
|
||||
word_text,
|
||||
)
|
||||
return
|
||||
|
||||
missing = self._describe_characters(uncoverable)
|
||||
if line_language and line_language in self.LANGUAGE_FONT_MAP:
|
||||
font_family = self.LANGUAGE_FONT_MAP[line_language].removesuffix('-Regular')
|
||||
log.warning(
|
||||
"No installed font has glyphs for the detected '%s' text (%s), "
|
||||
"so it was added as an invisible text layer: it stays searchable "
|
||||
"and copyable, but appears blank when highlighted in a PDF "
|
||||
"viewer. Install the %s font family (via your OS package "
|
||||
"manager or https://fonts.google.com/noto) for full rendering.",
|
||||
line_language,
|
||||
font_name,
|
||||
missing,
|
||||
font_family,
|
||||
)
|
||||
else:
|
||||
log.warning(
|
||||
"No font found with glyphs for some text. "
|
||||
"Install Noto fonts for better rendering. "
|
||||
"See https://fonts.google.com/noto"
|
||||
"No installed font has glyphs for some of the detected text "
|
||||
"(%s), so it was added as an invisible text layer: it stays "
|
||||
"searchable and copyable, but appears blank when highlighted "
|
||||
"in a PDF viewer. Install a Noto font covering that script "
|
||||
"(https://fonts.google.com/noto) for full rendering.",
|
||||
missing,
|
||||
)
|
||||
|
||||
def _uncoverable_characters(self, word_text: str) -> list[str]:
|
||||
"""Find the characters of word_text that no installed font can render.
|
||||
|
||||
Args:
|
||||
word_text: The word that fell back to glyphless rendering
|
||||
|
||||
Returns:
|
||||
The distinct uncoverable characters, in order of first appearance,
|
||||
considering at most MAX_EXAMINED_CHARS of them
|
||||
"""
|
||||
candidates = [
|
||||
char
|
||||
for char in dict.fromkeys(word_text) # de-duplicate, keep order
|
||||
if not char.isspace() and not self._is_char_renderable(char)
|
||||
]
|
||||
# The named fonts missed these, but the provider may still have a font
|
||||
# for them, so confirm before telling the user to install anything. The
|
||||
# search walks every installed font, hence the cap on how many
|
||||
# characters we are willing to look up for one warning.
|
||||
return [
|
||||
char
|
||||
for char in candidates[: self.MAX_EXAMINED_CHARS]
|
||||
if self._search_font_by_coverage(char) is None
|
||||
]
|
||||
|
||||
def _describe_characters(self, chars: list[str]) -> str:
|
||||
"""Describe characters by codepoint and Unicode name.
|
||||
|
||||
Naming the codepoints tells the user which font to install even for
|
||||
scripts OCRmyPDF has no language mapping for, which the generic
|
||||
"install the matching Noto fonts" advice did not. See issue #1722.
|
||||
|
||||
Args:
|
||||
chars: Characters to describe
|
||||
|
||||
Returns:
|
||||
Human-readable description, truncated to MAX_REPORTED_CHARS
|
||||
"""
|
||||
described = ", ".join(
|
||||
f"{char!r} U+{ord(char):04X} {unicodedata.name(char, 'unnamed character')}"
|
||||
for char in chars[: self.MAX_REPORTED_CHARS]
|
||||
)
|
||||
if len(chars) > self.MAX_REPORTED_CHARS:
|
||||
described += f", and {len(chars) - self.MAX_REPORTED_CHARS} more"
|
||||
return described
|
||||
|
||||
def _is_char_renderable(self, char: str) -> bool:
|
||||
"""Check whether any font already known to us has a glyph for char."""
|
||||
for font_name in [*self.FALLBACK_FONTS, *self._discovered_fonts]:
|
||||
font = self.font_provider.get_font(font_name)
|
||||
if font is not None and self._has_all_glyphs(font, char):
|
||||
return True
|
||||
return False
|
||||
|
||||
def _has_all_glyphs(self, font: FontManager, text: str) -> bool:
|
||||
"""Check if a font has glyphs for all characters in text.
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ Linux, macOS, and Windows platforms.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import glob
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
@@ -75,6 +76,35 @@ class SystemFontProvider:
|
||||
# Variable fonts
|
||||
'NotoSansCJKsc-VF.otf',
|
||||
],
|
||||
# Per-language CJK families. Modern Google Fonts / Homebrew ship these
|
||||
# as region subset variable fonts ('NotoSansJP[wght].ttf'), matched by
|
||||
# the flexible base search; the legacy per-region super OTFs (full
|
||||
# coverage) are listed here so they also satisfy the logical name.
|
||||
'NotoSansSC-Regular': [
|
||||
'NotoSansSC-Regular.otf',
|
||||
'NotoSansSC-Regular.ttf',
|
||||
'NotoSansCJKsc-Regular.otf',
|
||||
],
|
||||
'NotoSansTC-Regular': [
|
||||
'NotoSansTC-Regular.otf',
|
||||
'NotoSansTC-Regular.ttf',
|
||||
'NotoSansCJKtc-Regular.otf',
|
||||
],
|
||||
'NotoSansHK-Regular': [
|
||||
'NotoSansHK-Regular.otf',
|
||||
'NotoSansHK-Regular.ttf',
|
||||
'NotoSansCJKhk-Regular.otf',
|
||||
],
|
||||
'NotoSansJP-Regular': [
|
||||
'NotoSansJP-Regular.otf',
|
||||
'NotoSansJP-Regular.ttf',
|
||||
'NotoSansCJKjp-Regular.otf',
|
||||
],
|
||||
'NotoSansKR-Regular': [
|
||||
'NotoSansKR-Regular.otf',
|
||||
'NotoSansKR-Regular.ttf',
|
||||
'NotoSansCJKkr-Regular.otf',
|
||||
],
|
||||
'NotoSansThai-Regular': [
|
||||
'NotoSansThai-Regular.ttf',
|
||||
'NotoSansThai-Regular.otf',
|
||||
@@ -149,6 +179,28 @@ class SystemFontProvider:
|
||||
],
|
||||
}
|
||||
|
||||
# Font file extensions we know how to load.
|
||||
_FONT_EXTENSIONS = ('.ttf', '.otf', '.ttc')
|
||||
|
||||
# Acceptable filename variants for a font family, ranked best-first.
|
||||
# Lower rank wins when multiple variants of the same family are present.
|
||||
_VARIANT_RANK = {'regular': 0, 'variable': 1, 'vf': 2, 'plain': 3}
|
||||
|
||||
# Extra family bases that can satisfy a logical font, tried after its own
|
||||
# base (so the listed order is the preference). CJK is the case that needs
|
||||
# this: the legacy Adobe-style 'NotoSansCJKsc-Regular.otf' is handled by
|
||||
# NOTO_FONT_PATTERNS, but Homebrew casks and current Google Fonts ship the
|
||||
# per-language families as variable fonts (e.g. 'NotoSansSC[wght].ttf').
|
||||
_ALTERNATE_BASES: dict[str, list[str]] = {
|
||||
'NotoSansCJK-Regular': [
|
||||
'NotoSansSC', # Simplified Chinese
|
||||
'NotoSansTC', # Traditional Chinese
|
||||
'NotoSansHK', # Hong Kong
|
||||
'NotoSansJP', # Japanese
|
||||
'NotoSansKR', # Korean
|
||||
],
|
||||
}
|
||||
|
||||
def __init__(self) -> None:
|
||||
"""Initialize system font provider with empty caches."""
|
||||
# Cache: font_name -> FontManager (successfully loaded fonts)
|
||||
@@ -157,6 +209,13 @@ class SystemFontProvider:
|
||||
self._not_found: set[str] = set()
|
||||
# Cached font directories (computed lazily)
|
||||
self._font_dirs: list[Path] | None = None
|
||||
# Cached (logical name, path) of every Noto face on the system, in the
|
||||
# order the coverage search should try them (computed lazily)
|
||||
self._noto_candidates: list[tuple[str, Path]] | None = None
|
||||
# Memoized results of find_font_with_glyphs(), keyed by codepoint set
|
||||
self._coverage_cache: dict[frozenset[int], str | None] = {}
|
||||
# Font files that failed to load, so we only complain about them once
|
||||
self._unloadable: set[Path] = set()
|
||||
|
||||
def _get_platform(self) -> str:
|
||||
"""Get the current platform identifier.
|
||||
@@ -222,14 +281,221 @@ class SystemFontProvider:
|
||||
try:
|
||||
matches = list(font_dir.rglob(pattern))
|
||||
if matches:
|
||||
log.debug(
|
||||
"Found system font %s at %s", font_name, matches[0]
|
||||
)
|
||||
log.debug("Found system font %s at %s", font_name, matches[0])
|
||||
return matches[0]
|
||||
except PermissionError:
|
||||
# Skip directories we can't read
|
||||
continue
|
||||
|
||||
# No exact static '-Regular' file. Many distributors (Homebrew casks,
|
||||
# current Google Fonts releases) ship Noto fonts as variable fonts with
|
||||
# bracketed axis filenames such as 'NotoSansArabic[wdth,wght].ttf'.
|
||||
# Fall back to a flexible search that also accepts those. See #1652.
|
||||
return self._find_variant_font_file(font_name)
|
||||
|
||||
@staticmethod
|
||||
def _classify_variant(stem: str, base: str) -> str | None:
|
||||
"""Classify a font filename stem as a usable variant of ``base``.
|
||||
|
||||
Args:
|
||||
stem: Filename without extension (e.g. 'NotoSansArabic[wdth,wght]')
|
||||
base: Family base name (e.g. 'NotoSansArabic')
|
||||
|
||||
Returns:
|
||||
The variant kind ('regular', 'variable', 'vf', 'plain') or None if
|
||||
the stem is not an acceptable representative of the family. The
|
||||
boundary after ``base`` is required so that 'NotoSans' does not
|
||||
match 'NotoSansArabic', and 'NotoSansArabicUI'/'NotoSansArabic-Bold'
|
||||
do not match a request for 'NotoSansArabic'.
|
||||
"""
|
||||
if stem == f'{base}-Regular':
|
||||
return 'regular'
|
||||
if stem.startswith(f'{base}['): # variable font, e.g. Base[wdth,wght]
|
||||
return 'variable'
|
||||
if stem == f'{base}-VF': # alternate variable-font naming
|
||||
return 'vf'
|
||||
if stem == base: # bare family name
|
||||
return 'plain'
|
||||
return None
|
||||
|
||||
def _find_variant_font_file(self, font_name: str) -> Path | None:
|
||||
"""Search for a variable font or other acceptable filename variant.
|
||||
|
||||
Tries the font's own family base first, then any alternate bases (used
|
||||
for the modern per-language CJK families). Within that, a static Regular
|
||||
is preferred over a variable font. See issue #1652.
|
||||
|
||||
Args:
|
||||
font_name: Logical font name (e.g. 'NotoSansArabic-Regular')
|
||||
|
||||
Returns:
|
||||
Path to the best-ranked matching font file, or None.
|
||||
"""
|
||||
bases = [font_name.removesuffix('-Regular')]
|
||||
bases.extend(self._ALTERNATE_BASES.get(font_name, []))
|
||||
|
||||
# Selection key (base_index, variant_rank): earlier base wins, then the
|
||||
# better variant. Path is carried along but not part of the comparison.
|
||||
best: tuple[tuple[int, int], Path] | None = None
|
||||
for base_index, base in enumerate(bases):
|
||||
for font_dir in self._get_font_dirs():
|
||||
if not font_dir.exists():
|
||||
continue
|
||||
try:
|
||||
for path in font_dir.rglob(glob.escape(base) + '*'):
|
||||
if path.suffix.lower() not in self._FONT_EXTENSIONS:
|
||||
continue
|
||||
kind = self._classify_variant(path.stem, base)
|
||||
if kind is None:
|
||||
continue
|
||||
key = (base_index, self._VARIANT_RANK[kind])
|
||||
if best is None or key < best[0]:
|
||||
best = (key, path)
|
||||
except PermissionError:
|
||||
# Skip directories we can't read
|
||||
continue
|
||||
if best is not None:
|
||||
log.debug("Found system font %s at %s (variant match)", font_name, best[1])
|
||||
return best[1]
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _family_base(stem: str) -> str | None:
|
||||
"""Get the Noto family base a filename stem is the Regular face of.
|
||||
|
||||
Args:
|
||||
stem: Filename without extension, e.g. 'NotoSansCherokee-Regular'
|
||||
|
||||
Returns:
|
||||
The family base ('NotoSansCherokee') or None if the stem is not a
|
||||
Noto font, or is a weight/slope variant such as '-Bold' or
|
||||
'-Italic' that should not stand in for the family.
|
||||
"""
|
||||
head = stem.split('[', 1)[0] # drop variable-font axes, e.g. '[wght]'
|
||||
if head.endswith('-Regular'):
|
||||
head = head[: -len('-Regular')]
|
||||
elif head.endswith('-VF'):
|
||||
head = head[: -len('-VF')]
|
||||
elif '-' in head:
|
||||
return None
|
||||
return head if head.startswith('Noto') else None
|
||||
|
||||
@classmethod
|
||||
def _candidate_sort_key(cls, base: str) -> tuple[int, int, str]:
|
||||
"""Rank a family base for the coverage search.
|
||||
|
||||
Sans comes before serif before everything else, and plain families come
|
||||
ahead of their narrower UI and Mono cousins.
|
||||
"""
|
||||
if base.startswith('NotoSans'):
|
||||
family_rank = 0
|
||||
elif base.startswith('NotoSerif'):
|
||||
family_rank = 1
|
||||
else:
|
||||
family_rank = 2
|
||||
narrow_use = base.endswith('UI') or base.startswith('NotoSansMono')
|
||||
return (family_rank, int(narrow_use), base)
|
||||
|
||||
def _get_noto_candidates(self) -> list[tuple[str, Path]]:
|
||||
"""Enumerate every Noto family installed on the system.
|
||||
|
||||
Scans each font directory once and keeps the best-ranked file per
|
||||
family, so a family present in several directories or in several
|
||||
variants contributes a single candidate.
|
||||
|
||||
Returns:
|
||||
List of (logical font name, path) in the order to try them.
|
||||
"""
|
||||
if self._noto_candidates is not None:
|
||||
return self._noto_candidates
|
||||
|
||||
best: dict[str, tuple[int, Path]] = {}
|
||||
for font_dir in self._get_font_dirs():
|
||||
if not font_dir.exists():
|
||||
continue
|
||||
try:
|
||||
paths = sorted(font_dir.rglob('Noto*'))
|
||||
except OSError:
|
||||
# Skip directories we can't read
|
||||
continue
|
||||
for path in paths:
|
||||
if path.suffix.lower() not in self._FONT_EXTENSIONS:
|
||||
continue
|
||||
base = self._family_base(path.stem)
|
||||
if base is None:
|
||||
continue
|
||||
kind = self._classify_variant(path.stem, base)
|
||||
if kind is None:
|
||||
continue
|
||||
rank = self._VARIANT_RANK[kind]
|
||||
if base not in best or rank < best[base][0]:
|
||||
best[base] = (rank, path)
|
||||
|
||||
self._noto_candidates = [
|
||||
(f'{base}-Regular', path)
|
||||
for base, (_rank, path) in sorted(
|
||||
best.items(), key=lambda item: self._candidate_sort_key(item[0])
|
||||
)
|
||||
]
|
||||
return self._noto_candidates
|
||||
|
||||
def find_font_with_glyphs(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Find any installed Noto font that covers every character in text.
|
||||
|
||||
``NOTO_FONT_PATTERNS`` enumerates the couple dozen scripts OCRmyPDF
|
||||
knows by name, but systems ship far more: macOS alone installs around a
|
||||
hundred script-specific Noto faces in
|
||||
``/System/Library/Fonts/Supplemental``. This is the last resort that
|
||||
makes those usable, so a document is only rendered glyphless when no
|
||||
installed font can actually cover it. See issue #1722.
|
||||
|
||||
This walks every Noto face on the system and is therefore expensive;
|
||||
results are memoized, and callers should only reach it after the named
|
||||
fonts have failed.
|
||||
|
||||
Args:
|
||||
text: Text that the returned font must fully cover
|
||||
|
||||
Returns:
|
||||
(logical font name, FontManager) of the first covering font, or
|
||||
None if nothing installed covers the text.
|
||||
"""
|
||||
if not text:
|
||||
return None
|
||||
needed = frozenset(ord(c) for c in text)
|
||||
|
||||
if needed in self._coverage_cache:
|
||||
cached_name = self._coverage_cache[needed]
|
||||
if cached_name is None:
|
||||
return None
|
||||
if cached := self._font_cache.get(cached_name):
|
||||
return cached_name, cached
|
||||
|
||||
for font_name, path in self._get_noto_candidates():
|
||||
font = self._font_cache.get(font_name)
|
||||
if font is None:
|
||||
if path in self._unloadable:
|
||||
continue
|
||||
try:
|
||||
font = FontManager(path)
|
||||
except Exception as e:
|
||||
log.debug("Skipping unreadable font %s: %s", path, e)
|
||||
self._unloadable.add(path)
|
||||
continue
|
||||
if all(font.has_glyph(cp) for cp in needed):
|
||||
# Keep only fonts we actually use; the rest are released so a
|
||||
# full scan doesn't retain every font file on the system.
|
||||
self._font_cache[font_name] = font
|
||||
self._not_found.discard(font_name)
|
||||
self._coverage_cache[needed] = font_name
|
||||
log.debug(
|
||||
"Found system font %s at %s (glyph coverage match)",
|
||||
font_name,
|
||||
path,
|
||||
)
|
||||
return font_name, font
|
||||
|
||||
self._coverage_cache[needed] = None
|
||||
return None
|
||||
|
||||
def get_font(self, font_name: str) -> FontManager | None:
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
This module provides the PDF renderer using fpdf2 for creating
|
||||
searchable OCR text layers.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ocrmypdf.fpdf_renderer.renderer import (
|
||||
|
||||
@@ -14,9 +14,11 @@ import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from math import atan, cos, degrees, radians, sin, sqrt
|
||||
from pathlib import Path
|
||||
from typing import cast
|
||||
|
||||
from fpdf import FPDF
|
||||
from fpdf.enums import PDFResourceType, TextMode
|
||||
from fpdf.fonts import TTFFont
|
||||
from pikepdf import Matrix, Rectangle
|
||||
|
||||
from ocrmypdf.font import FontManager, MultiFontManager
|
||||
@@ -241,10 +243,16 @@ class Fpdf2PdfRenderer:
|
||||
pdf: FPDF instance to render into
|
||||
"""
|
||||
# Add page with correct dimensions
|
||||
# fpdf2's add_page() stub says format: str, but its docstring and
|
||||
# get_page_format() helper confirm a (width, height) tuple is
|
||||
# supported too - the annotation on add_page() itself is just wrong.
|
||||
pdf.add_page(
|
||||
format=(
|
||||
self.coord_transform.page_width_pt,
|
||||
self.coord_transform.page_height_pt,
|
||||
format=cast(
|
||||
'str',
|
||||
(
|
||||
self.coord_transform.page_width_pt,
|
||||
self.coord_transform.page_height_pt,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
@@ -448,8 +456,13 @@ class Fpdf2PdfRenderer:
|
||||
# entirely (slope=0, no textangle) and produced garbage text in a
|
||||
# bounding box whose shape doesn't match the text content at all.
|
||||
if not self._check_aspect_ratio_plausible(
|
||||
pdf, words, font_size, slope_angle_deg,
|
||||
line_size_width, line_size_height, line_language,
|
||||
pdf,
|
||||
words,
|
||||
font_size,
|
||||
slope_angle_deg,
|
||||
line_size_width,
|
||||
line_size_height,
|
||||
line_language,
|
||||
):
|
||||
return
|
||||
|
||||
@@ -507,13 +520,15 @@ class Fpdf2PdfRenderer:
|
||||
else:
|
||||
word_tz = 100.0
|
||||
|
||||
word_render_data.append(WordRenderData(
|
||||
text=word.text,
|
||||
x_baseline=box_llx,
|
||||
font_family=font_family,
|
||||
word_tz=word_tz,
|
||||
is_rtl=word_is_rtl,
|
||||
))
|
||||
word_render_data.append(
|
||||
WordRenderData(
|
||||
text=word.text,
|
||||
x_baseline=box_llx,
|
||||
font_family=font_family,
|
||||
word_tz=word_tz,
|
||||
is_rtl=word_is_rtl,
|
||||
)
|
||||
)
|
||||
|
||||
if not word_render_data:
|
||||
return
|
||||
@@ -561,9 +576,7 @@ class Fpdf2PdfRenderer:
|
||||
if line_size_width >= line_size_height:
|
||||
return True
|
||||
|
||||
line_text = ' '.join(
|
||||
w.text for w in words if w is not None and w.text
|
||||
)
|
||||
line_text = ' '.join(w.text for w in words if w is not None and w.text)
|
||||
if not line_text:
|
||||
return True
|
||||
|
||||
@@ -603,9 +616,7 @@ class Fpdf2PdfRenderer:
|
||||
line_text[:80],
|
||||
)
|
||||
if not self._logged_aspect_ratio_suppression:
|
||||
log.info(
|
||||
"Suppressing OCR output text with improbable aspect ratio"
|
||||
)
|
||||
log.info("Suppressing OCR output text with improbable aspect ratio")
|
||||
self._logged_aspect_ratio_suppression = True
|
||||
return False
|
||||
|
||||
@@ -679,9 +690,7 @@ class Fpdf2PdfRenderer:
|
||||
ops.append(f'{first_x_baseline:.2f} 0 Td')
|
||||
else:
|
||||
# Direct PDF coordinates
|
||||
page_x, page_y_fpdf = transform_point(
|
||||
baseline_matrix, first_x_baseline, 0
|
||||
)
|
||||
page_x, page_y_fpdf = transform_point(baseline_matrix, first_x_baseline, 0)
|
||||
page_y_pdf = page_height - page_y_fpdf
|
||||
ops.append(f'{page_x:.2f} {page_y_pdf:.2f} Td')
|
||||
|
||||
@@ -694,13 +703,15 @@ class Fpdf2PdfRenderer:
|
||||
# Set font if changed
|
||||
if word.font_family != prev_font_family:
|
||||
pdf.set_font(word.font_family, size=font_size)
|
||||
# We only ever register fonts via add_font() with a TTF file
|
||||
# (see _register_font), so set_font() always resolves to a
|
||||
# TTFFont, never a built-in CoreFont or leaves it unset.
|
||||
assert pdf.current_font is not None
|
||||
# Register font resource on this page
|
||||
pdf._resource_catalog.add(
|
||||
PDFResourceType.FONT, pdf.current_font.i, pdf.page
|
||||
)
|
||||
ops.append(
|
||||
f'/F{pdf.current_font.i} {pdf.font_size_pt:.2f} Tf'
|
||||
)
|
||||
ops.append(f'/F{pdf.current_font.i} {pdf.font_size_pt:.2f} Tf')
|
||||
prev_font_family = word.font_family
|
||||
|
||||
# Relative positioning (for words after the first)
|
||||
@@ -728,12 +739,8 @@ class Fpdf2PdfRenderer:
|
||||
advance = next_word.x_baseline - word.x_baseline
|
||||
|
||||
# Add trailing space for text extraction unless both are CJK
|
||||
if (
|
||||
advance > 0
|
||||
and not (
|
||||
self._is_cjk_only(word.text)
|
||||
and self._is_cjk_only(next_word.text)
|
||||
)
|
||||
if advance > 0 and not (
|
||||
self._is_cjk_only(word.text) and self._is_cjk_only(next_word.text)
|
||||
):
|
||||
text_to_render = word.text + ' '
|
||||
else:
|
||||
@@ -744,9 +751,7 @@ class Fpdf2PdfRenderer:
|
||||
# Use word_tz (fits word into its hOCR bbox) — Td handles
|
||||
# inter-word gaps, so Tz should not stretch to fill them.
|
||||
ops.append(f'{word.word_tz:.2f} Tz')
|
||||
ops.append(
|
||||
self._encode_shaped_text(pdf, text_to_render, word.is_rtl)
|
||||
)
|
||||
ops.append(self._encode_shaped_text(pdf, text_to_render, word.is_rtl))
|
||||
|
||||
prev_x_baseline = word.x_baseline
|
||||
|
||||
@@ -762,9 +767,7 @@ class Fpdf2PdfRenderer:
|
||||
# don't think Tz is still set from our raw operators
|
||||
pdf.font_stretching = 100
|
||||
|
||||
def _encode_shaped_text(
|
||||
self, pdf: FPDF, text: str, is_rtl: bool = False
|
||||
) -> str:
|
||||
def _encode_shaped_text(self, pdf: FPDF, text: str, is_rtl: bool = False) -> str:
|
||||
"""Encode text using HarfBuzz text shaping for complex script support.
|
||||
|
||||
Unlike font.encode_text() which maps unicode characters one-by-one to
|
||||
@@ -782,6 +785,10 @@ class Fpdf2PdfRenderer:
|
||||
joining forms and ligature shaping is harmless.
|
||||
"""
|
||||
font = pdf.current_font
|
||||
# We only ever register fonts via add_font() with a TTF file (see
|
||||
# _register_font), so current_font is always a TTFFont - never the
|
||||
# built-in CoreFont (which lacks shape_text()/escape_text()) or None.
|
||||
assert isinstance(font, TTFFont)
|
||||
if is_rtl:
|
||||
# Reverse the text so that after bidi reversal by the text
|
||||
# extractor, the characters end up in correct logical order.
|
||||
|
||||
+52
-17
@@ -18,6 +18,7 @@ from math import isclose, isfinite
|
||||
from pathlib import Path
|
||||
from statistics import harmonic_mean
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Generic,
|
||||
TypeVar,
|
||||
@@ -26,6 +27,9 @@ from typing import (
|
||||
import img2pdf
|
||||
import pikepdf
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from _typeshed import StrOrBytesPath
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
IMG2PDF_KWARGS = dict(engine=img2pdf.Engine.pikepdf, rotation=img2pdf.Rotation.ifvalid)
|
||||
@@ -135,7 +139,7 @@ class Resolution(Generic[T]):
|
||||
return self._isclose(self.x, other.x) and self._isclose(self.y, other.y)
|
||||
|
||||
|
||||
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike) -> None:
|
||||
def safe_symlink(input_file: StrOrBytesPath, soft_link_name: StrOrBytesPath) -> None:
|
||||
"""Create a symbolic link at ``soft_link_name``, which references ``input_file``.
|
||||
|
||||
Think of this as copying ``input_file`` to ``soft_link_name`` with less overhead.
|
||||
@@ -144,11 +148,11 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike) -> None:
|
||||
used since symlinks may require administrator privileges. An existing link at the
|
||||
destination is removed.
|
||||
"""
|
||||
input_file = os.fspath(input_file)
|
||||
soft_link_name = os.fspath(soft_link_name)
|
||||
input_path = Path(os.fsdecode(input_file))
|
||||
soft_link_path = Path(os.fsdecode(soft_link_name))
|
||||
|
||||
# Guard against soft linking to oneself
|
||||
if input_file == soft_link_name:
|
||||
if input_path == soft_link_path:
|
||||
log.warning(
|
||||
"No symbolic link created. You are using the original data directory "
|
||||
"as the working directory."
|
||||
@@ -156,24 +160,24 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike) -> None:
|
||||
return
|
||||
|
||||
# Soft link already exists: delete for relink?
|
||||
if os.path.lexists(soft_link_name):
|
||||
if os.path.lexists(soft_link_path):
|
||||
# do not delete or overwrite real (non-soft link) file
|
||||
if not os.path.islink(soft_link_name):
|
||||
raise FileExistsError(f"{soft_link_name} exists and is not a link")
|
||||
os.unlink(soft_link_name)
|
||||
if not soft_link_path.is_symlink():
|
||||
raise FileExistsError(f"{soft_link_path} exists and is not a link")
|
||||
soft_link_path.unlink()
|
||||
|
||||
if not os.path.exists(input_file):
|
||||
raise FileNotFoundError(f"trying to create a broken symlink to {input_file}")
|
||||
if not input_path.exists():
|
||||
raise FileNotFoundError(f"trying to create a broken symlink to {input_path}")
|
||||
|
||||
if os.name == 'nt':
|
||||
# Don't actually use symlinks on Windows due to permission issues
|
||||
shutil.copyfile(input_file, soft_link_name)
|
||||
shutil.copyfile(input_path, soft_link_path)
|
||||
return
|
||||
|
||||
log.debug("os.symlink(%s, %s)", input_file, soft_link_name)
|
||||
log.debug("os.symlink(%s, %s)", input_path, soft_link_path)
|
||||
|
||||
# Create symbolic link using absolute path
|
||||
os.symlink(os.path.abspath(input_file), soft_link_name)
|
||||
soft_link_path.symlink_to(input_path.resolve())
|
||||
|
||||
|
||||
def samefile(file1: os.PathLike, file2: os.PathLike) -> bool:
|
||||
@@ -184,7 +188,7 @@ def samefile(file1: os.PathLike, file2: os.PathLike) -> bool:
|
||||
if os.name == 'nt':
|
||||
return file1 == file2
|
||||
else:
|
||||
return os.path.samefile(file1, file2)
|
||||
return Path(file1).samefile(file2)
|
||||
|
||||
|
||||
def is_iterable_notstr(thing: Any) -> bool:
|
||||
@@ -199,7 +203,7 @@ def monotonic(seq: Sequence) -> bool:
|
||||
|
||||
def page_number(input_file: os.PathLike) -> int:
|
||||
"""Get one-based page number implied by filename (000002.pdf -> 2)."""
|
||||
return int(os.path.basename(os.fspath(input_file))[0:6])
|
||||
return int(Path(input_file).name[0:6])
|
||||
|
||||
|
||||
def available_cpu_count() -> int:
|
||||
@@ -214,7 +218,7 @@ def available_cpu_count() -> int:
|
||||
return 1
|
||||
|
||||
|
||||
def is_file_writable(test_file: os.PathLike) -> bool:
|
||||
def is_file_writable(test_file: StrOrBytesPath) -> bool:
|
||||
"""Intentionally racy test if target is writable.
|
||||
|
||||
We intend to write to the output file if and only if we succeed and
|
||||
@@ -222,7 +226,7 @@ def is_file_writable(test_file: os.PathLike) -> bool:
|
||||
the location is writable.
|
||||
"""
|
||||
try:
|
||||
p = Path(test_file)
|
||||
p = Path(os.fsdecode(test_file))
|
||||
if p.is_symlink():
|
||||
p = p.resolve(strict=False)
|
||||
|
||||
@@ -329,6 +333,37 @@ def pikepdf_enable_mmap() -> None:
|
||||
log.debug("pikepdf mmap not available")
|
||||
|
||||
|
||||
def pikepdf_get_int(obj: pikepdf.Object, key: pikepdf.Name, default: int = 0) -> int:
|
||||
"""Look up a key on a pikepdf dictionary/stream, returning a plain int.
|
||||
|
||||
``.get(key, default)``'s return type is the ambiguous ``Object | int``,
|
||||
which does not support arithmetic or comparison against a plain int. In
|
||||
pikepdf's default (implicit) conversion mode, a PDF Integer is already
|
||||
unboxed to a native ``int`` by the time we see it here; under explicit
|
||||
conversion mode it would instead be a ``pikepdf.Object``. ``int()``
|
||||
handles both, since ``Object`` implements ``__int__``.
|
||||
"""
|
||||
value = obj.get(key)
|
||||
return int(value) if value is not None else default
|
||||
|
||||
|
||||
def pikepdf_get_bool(
|
||||
obj: pikepdf.Object, key: pikepdf.Name, default: bool = False
|
||||
) -> bool:
|
||||
"""Look up a key on a pikepdf dictionary/stream, returning a plain bool.
|
||||
|
||||
Unlike ``int()``/``float()``, ``bool()`` is not supported on
|
||||
``pikepdf.Object`` (it raises), so both conversion modes must be
|
||||
handled explicitly. See :func:`pikepdf_get_int` for background.
|
||||
"""
|
||||
value = obj.get(key)
|
||||
if value is None:
|
||||
return default
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
return value.as_bool(default)
|
||||
|
||||
|
||||
def running_in_docker() -> bool:
|
||||
"""Returns True if we seem to be running in a Docker container."""
|
||||
return Path('/.dockerenv').exists()
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
Derived from
|
||||
https://www.loc.gov/standards/iso639-2/ascii_8bits.html
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import NamedTuple
|
||||
|
||||
+42
-18
@@ -12,7 +12,7 @@ import threading
|
||||
from collections.abc import Callable, Iterator, MutableSet, Sequence
|
||||
from os import fspath
|
||||
from pathlib import Path
|
||||
from typing import Any, NamedTuple, NewType
|
||||
from typing import Any, NamedTuple, NewType, cast
|
||||
from zlib import compress
|
||||
|
||||
import img2pdf
|
||||
@@ -37,7 +37,7 @@ from ocrmypdf._exec import ghostscript, jbig2enc, pngquant
|
||||
from ocrmypdf._jobcontext import PdfContext
|
||||
from ocrmypdf._progressbar import ProgressBar
|
||||
from ocrmypdf.exceptions import OutputFileAccessError
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, safe_symlink
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, pikepdf_get_int, safe_symlink
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -260,10 +260,19 @@ def _find_image_xrefs_container(
|
||||
exclude_xrefs: MutableSet[Xref],
|
||||
pageno_for_xref: dict[Xref, int],
|
||||
depth: int = 0,
|
||||
visited_forms: MutableSet[Xref] | None = None,
|
||||
):
|
||||
"""Find all image XRefs or Form XObject and add to the include/exclude sets."""
|
||||
# Form XObjects are not added to include/exclude_xrefs, so the dedup
|
||||
# check below doesn't catch Form-XObject cycles or DAGs. Track them in
|
||||
# a shared set so each Form is only descended into once per document
|
||||
# (issue #1321).
|
||||
if visited_forms is None:
|
||||
visited_forms = set()
|
||||
if depth > 10:
|
||||
log.warning("Recursion depth exceeded in _find_image_xrefs_page")
|
||||
# With visited_forms memoization, this is a soft DAG-height guard
|
||||
# rather than a cycle defense, so a debug log is sufficient.
|
||||
log.debug("Recursion depth exceeded in _find_image_xrefs_page")
|
||||
return
|
||||
try:
|
||||
xobjs = container.Resources.XObject
|
||||
@@ -276,7 +285,9 @@ def _find_image_xrefs_container(
|
||||
if xref in include_xrefs or xref in exclude_xrefs:
|
||||
continue # Already processed
|
||||
if Name.Subtype in image and image.Subtype == Name.Form:
|
||||
# Recurse into Form XObjects
|
||||
if xref in visited_forms:
|
||||
continue
|
||||
visited_forms.add(xref)
|
||||
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
||||
_find_image_xrefs_container(
|
||||
pdf,
|
||||
@@ -286,6 +297,7 @@ def _find_image_xrefs_container(
|
||||
exclude_xrefs,
|
||||
pageno_for_xref,
|
||||
depth + 1,
|
||||
visited_forms,
|
||||
)
|
||||
continue
|
||||
if Name.SMask in image:
|
||||
@@ -342,9 +354,16 @@ def extract_images(
|
||||
pdf=pdf, root=root, image=image, xref=xref, options=options
|
||||
)
|
||||
except Exception: # pylint: disable=broad-except
|
||||
log.exception(
|
||||
f"xref {xref}: While extracting this image, an error occurred"
|
||||
# Optimization is best-effort: an image we cannot process is simply
|
||||
# left unchanged in the output, which remains valid. Report this as
|
||||
# a concise warning rather than an alarming traceback (issue #846);
|
||||
# the full detail is still available at debug verbosity.
|
||||
log.warning(
|
||||
f"xref {xref}: this image could not be processed by the "
|
||||
"optimizer and was left unchanged. The output file is still "
|
||||
"valid."
|
||||
)
|
||||
log.debug(f"xref {xref}: image optimization error detail", exc_info=True)
|
||||
errors += 1
|
||||
else:
|
||||
if result:
|
||||
@@ -430,12 +449,12 @@ def convert_to_jbig2(
|
||||
|
||||
|
||||
def _optimize_jpeg(
|
||||
xref: Xref, in_jpg: Path, opt_jpg: Path, jpg_quality: int
|
||||
xref: Xref, in_jpg: Path, opt_jpg: Path, jpeg_quality: int
|
||||
) -> tuple[Xref, Path | None]:
|
||||
with Image.open(in_jpg) as im:
|
||||
save_kwargs: dict[str, Any] = {'optimize': True}
|
||||
if isinstance(jpg_quality, int) and 0 < jpg_quality <= 100:
|
||||
save_kwargs['quality'] = jpg_quality
|
||||
if isinstance(jpeg_quality, int) and 0 < jpeg_quality <= 100:
|
||||
save_kwargs['quality'] = jpeg_quality
|
||||
im.save(opt_jpg, **save_kwargs)
|
||||
|
||||
if opt_jpg.stat().st_size > in_jpg.stat().st_size:
|
||||
@@ -454,7 +473,7 @@ def transcode_jpegs(
|
||||
for xref in jpegs:
|
||||
in_jpg = jpg_name(root, xref)
|
||||
opt_jpg = in_jpg.with_suffix('.opt.jpg')
|
||||
yield xref, in_jpg, opt_jpg, options.jpg_quality
|
||||
yield xref, in_jpg, opt_jpg, options.jpeg_quality
|
||||
|
||||
def finish_jpeg(result: tuple[Xref, Path | None], pbar: ProgressBar):
|
||||
xref, opt_jpg = result
|
||||
@@ -508,8 +527,8 @@ def _find_deflatable_jpeg(
|
||||
(
|
||||
# Don't flate very large images because it will slow down PDF viewers
|
||||
1 <= options.optimize <= 2
|
||||
and image.get(Name.Width, 0) < FLATE_JPEG_THRESHOLD
|
||||
and image.get(Name.Height, 0) < FLATE_JPEG_THRESHOLD
|
||||
and pikepdf_get_int(image, Name.Width) < FLATE_JPEG_THRESHOLD
|
||||
and pikepdf_get_int(image, Name.Height) < FLATE_JPEG_THRESHOLD
|
||||
)
|
||||
or options.optimize == 3
|
||||
)
|
||||
@@ -589,10 +608,13 @@ def _transcode_png(pdf: Pdf, filename: Path, xref: Xref) -> bool:
|
||||
local_image = pdf.copy_foreign(foreign_image)
|
||||
|
||||
im_obj = pdf.get_object(xref, 0)
|
||||
# pikepdf's Object attribute access can't statically know Filter/
|
||||
# DecodeParms hold these specific subtypes, but a copied image's
|
||||
# stream dictionary always does per the PDF spec.
|
||||
im_obj.write(
|
||||
local_image.read_raw_bytes(),
|
||||
filter=local_image.Filter,
|
||||
decode_parms=local_image.DecodeParms,
|
||||
filter=cast('Name | Array | list[Name] | None', local_image.Filter),
|
||||
decode_parms=cast('Dictionary | Array | None', local_image.DecodeParms),
|
||||
)
|
||||
|
||||
# Don't copy keys from the new image...
|
||||
@@ -681,8 +703,8 @@ def optimize(
|
||||
safe_symlink(input_file, output_file)
|
||||
return output_file
|
||||
|
||||
if not options.jpg_quality:
|
||||
options.jpg_quality = DEFAULT_JPEG_QUALITY if options.optimize < 3 else 40
|
||||
if not options.jpeg_quality:
|
||||
options.jpeg_quality = DEFAULT_JPEG_QUALITY if options.optimize < 3 else 40
|
||||
if not options.png_quality:
|
||||
options.png_quality = DEFAULT_PNG_QUALITY if options.optimize < 3 else 30
|
||||
|
||||
@@ -744,7 +766,7 @@ def main(infile, outfile, level, jobs=1):
|
||||
output_file=outfile, # Required field
|
||||
jobs=jobs,
|
||||
optimize=int(level),
|
||||
jpg_quality=0, # Use default
|
||||
jpeg_quality=0, # Use default
|
||||
png_quality=0,
|
||||
jbig2_threshold=0.85,
|
||||
quiet=True,
|
||||
@@ -752,7 +774,9 @@ def main(infile, outfile, level, jobs=1):
|
||||
)
|
||||
|
||||
with TemporaryDirectory() as tmpdir:
|
||||
context = PdfContext(options, Path(tmpdir), infile, None, None)
|
||||
# optimize() only reads context.options on this standalone path, so
|
||||
# pdfinfo and plugin_manager are not needed.
|
||||
context = PdfContext(options, Path(tmpdir), infile, None, None) # type: ignore[arg-type]
|
||||
tmpout = Path(tmpdir) / 'out.pdf'
|
||||
optimize(
|
||||
infile,
|
||||
|
||||
+74
-7
@@ -12,7 +12,7 @@ from importlib.resources import files as package_files
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
from pikepdf import Array, Dictionary, Name, Pdf, Stream
|
||||
from pikepdf import Array, Dictionary, Name, Object, Pdf, Stream
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -137,6 +137,71 @@ def file_claims_pdfa(filename: Path):
|
||||
return pdfa_dict
|
||||
|
||||
|
||||
def _cid_font_is_embedded(type0_font: Object) -> bool:
|
||||
"""Return True if a Type0 font's CID descendant carries embedded glyphs."""
|
||||
for descendant in type0_font.get(Name.DescendantFonts, []):
|
||||
descriptor = descendant.get(Name.FontDescriptor, None)
|
||||
# A malformed PDF may store a non-dictionary here; `key in descriptor`
|
||||
# raises on those, so require a real dictionary before probing it.
|
||||
if isinstance(descriptor, Dictionary) and any(
|
||||
key in descriptor for key in (Name.FontFile, Name.FontFile2, Name.FontFile3)
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def find_nonembedded_cid_fonts(pdf: Pdf) -> set[str]:
|
||||
"""Find CID-keyed (Type0) fonts that lack embedded glyph data.
|
||||
|
||||
PDF/A requires every font to be embedded. When Ghostscript converts a PDF
|
||||
to PDF/A it must substitute and embed a replacement for any non-embedded
|
||||
font. For CID-keyed fonts -- which is how CJK text is encoded, including the
|
||||
OCR text layers produced by Adobe Acrobat -- this substitution routinely
|
||||
corrupts the character-to-Unicode mapping, silently destroying the
|
||||
searchable text. Detecting these fonts lets the caller refuse PDF/A
|
||||
conversion rather than emit corrupted output.
|
||||
|
||||
Simple (non-CID) non-embedded fonts are not reported: Ghostscript
|
||||
substitutes standard encodings for them without corrupting the text, and
|
||||
they are far too common to treat as conversion blockers.
|
||||
|
||||
Args:
|
||||
pdf: An open ``pikepdf.Pdf`` to scan.
|
||||
|
||||
Returns:
|
||||
The set of ``BaseFont`` names of non-embedded CID fonts found.
|
||||
"""
|
||||
found: set[str] = set()
|
||||
|
||||
def scan_resources(resources, depth: int = 0) -> None:
|
||||
if resources is None or depth > 10:
|
||||
return
|
||||
# A well-formed PDF stores dictionaries under /Font and /XObject, but a
|
||||
# malformed one (common in OCR workloads) may store an array, a name, or
|
||||
# another non-dictionary object. Only such dictionaries have .values(),
|
||||
# so guard with isinstance rather than let the scan crash (issue #1713).
|
||||
fonts = resources.get(Name.Font, None)
|
||||
if isinstance(fonts, Dictionary):
|
||||
for font in fonts.as_dict().values():
|
||||
try:
|
||||
if font.get(Name.Subtype) != Name.Type0:
|
||||
continue
|
||||
if not _cid_font_is_embedded(font):
|
||||
basefont = str(font.get(Name.BaseFont, '/(unnamed)'))
|
||||
found.add(basefont.lstrip('/'))
|
||||
except (AttributeError, TypeError, KeyError):
|
||||
continue
|
||||
xobjects = resources.get(Name.XObject, None)
|
||||
if isinstance(xobjects, Dictionary):
|
||||
for xobj in xobjects.as_dict().values():
|
||||
if xobj.get(Name.Subtype) == Name.Form and Name.Resources in xobj:
|
||||
scan_resources(xobj[Name.Resources], depth + 1)
|
||||
|
||||
for page in pdf.pages:
|
||||
scan_resources(page.get(Name.Resources, None))
|
||||
return found
|
||||
|
||||
|
||||
def _load_srgb_icc_profile() -> bytes:
|
||||
"""Load the sRGB ICC profile from package data."""
|
||||
return (package_files('ocrmypdf.data') / SRGB_ICC_PROFILE_NAME).read_bytes()
|
||||
@@ -191,12 +256,14 @@ def add_srgb_output_intent(pdf: Pdf) -> None:
|
||||
icc_stream[Name.N] = 3 # RGB has 3 components
|
||||
|
||||
# Create OutputIntent dictionary
|
||||
output_intent = Dictionary({
|
||||
'/Type': Name.OutputIntent,
|
||||
'/S': Name('/GTS_PDFA1'),
|
||||
'/OutputConditionIdentifier': 'sRGB',
|
||||
'/DestOutputProfile': icc_stream,
|
||||
})
|
||||
output_intent = Dictionary(
|
||||
{
|
||||
'/Type': Name.OutputIntent,
|
||||
'/S': Name('/GTS_PDFA1'),
|
||||
'/OutputConditionIdentifier': 'sRGB',
|
||||
'/DestOutputProfile': icc_stream,
|
||||
}
|
||||
)
|
||||
|
||||
# Add to catalog's OutputIntents array
|
||||
if Name.OutputIntents not in pdf.Root:
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ocrmypdf.pdfinfo._types import Colorspace, Encoding, FloatRect
|
||||
from ocrmypdf.pdfinfo._types import Colorspace, Encoding, FloatRect, Ink
|
||||
from ocrmypdf.pdfinfo.info import PageInfo, PdfInfo
|
||||
|
||||
__all__ = ["Colorspace", "Encoding", "FloatRect", "PageInfo", "PdfInfo"]
|
||||
__all__ = ["Colorspace", "Encoding", "FloatRect", "Ink", "PageInfo", "PdfInfo"]
|
||||
|
||||
@@ -11,11 +11,11 @@ from math import hypot, inf, isclose
|
||||
from typing import NamedTuple
|
||||
from warnings import warn
|
||||
|
||||
from pikepdf import Matrix, Object, PdfInlineImage, parse_content_stream
|
||||
from pikepdf import Matrix, Name, Object, PdfInlineImage, parse_content_stream
|
||||
|
||||
from ocrmypdf.exceptions import InputFileError
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pdfinfo._types import UNIT_SQUARE
|
||||
from ocrmypdf.pdfinfo._types import UNIT_SQUARE, Ink
|
||||
|
||||
|
||||
class XobjectSettings(NamedTuple):
|
||||
@@ -24,6 +24,7 @@ class XobjectSettings(NamedTuple):
|
||||
name: str
|
||||
shorthand: tuple[float, float, float, float, float, float]
|
||||
stack_depth: int
|
||||
fill_ink: Ink
|
||||
|
||||
|
||||
class InlineSettings(NamedTuple):
|
||||
@@ -32,6 +33,7 @@ class InlineSettings(NamedTuple):
|
||||
iimage: PdfInlineImage
|
||||
shorthand: tuple[float, float, float, float, float, float]
|
||||
stack_depth: int
|
||||
fill_ink: Ink
|
||||
|
||||
|
||||
class ContentsInfo(NamedTuple):
|
||||
@@ -67,6 +69,60 @@ def _is_unit_square(shorthand):
|
||||
return all(isclose(a, b, rel_tol=1e-3) for a, b in pairwise)
|
||||
|
||||
|
||||
_INK_EPSILON = 1e-3
|
||||
|
||||
# Maps a fill-colorspace name (set by the `cs` operator) to a device color
|
||||
# family we can classify. Names not present here (Separation, ICCBased,
|
||||
# Indexed, DeviceN, Pattern, resource names like /CS0) are treated as color.
|
||||
_DEVICE_FILL_SPACE = {
|
||||
'/DeviceGray': 'gray',
|
||||
'/CalGray': 'gray',
|
||||
'/G': 'gray',
|
||||
'/DeviceRGB': 'rgb',
|
||||
'/CalRGB': 'rgb',
|
||||
'/RGB': 'rgb',
|
||||
'/DeviceCMYK': 'cmyk',
|
||||
'/CMYK': 'cmyk',
|
||||
}
|
||||
|
||||
|
||||
def _ink_from_components(space: str, comps: list[float]) -> Ink:
|
||||
"""Classify a device-color fill into mono/gray/color.
|
||||
|
||||
``space`` is one of 'gray', 'rgb', 'cmyk'. Any other value is treated
|
||||
conservatively as color, since we cannot prove it is achromatic.
|
||||
"""
|
||||
eps = _INK_EPSILON
|
||||
if space == 'gray' and len(comps) == 1:
|
||||
return Ink.mono if comps[0] <= eps else Ink.gray
|
||||
if space == 'rgb' and len(comps) == 3:
|
||||
r, g, b = comps
|
||||
if max(r, g, b) <= eps:
|
||||
return Ink.mono
|
||||
if abs(r - g) <= eps and abs(g - b) <= eps:
|
||||
return Ink.gray
|
||||
return Ink.color
|
||||
if space == 'cmyk' and len(comps) == 4:
|
||||
c, m, y, k = comps
|
||||
if c <= eps and m <= eps and y <= eps:
|
||||
return Ink.mono if k <= eps else Ink.gray
|
||||
return Ink.color
|
||||
return Ink.color # conservative-to-color
|
||||
|
||||
|
||||
def _operand_floats(operands) -> list[float] | None:
|
||||
"""Convert color operands to floats, or None if any is non-numeric.
|
||||
|
||||
Color operators in a malformed content stream may carry the wrong number
|
||||
of operands or a non-numeric operand (e.g. a Name). Returning None lets
|
||||
the caller keep the prior fill state instead of raising.
|
||||
"""
|
||||
try:
|
||||
return [float(o) for o in operands]
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _normalize_stack(graphobjs):
|
||||
"""Convert runs of qQ's in the stack into single graphobjs."""
|
||||
for operands, operator in graphobjs:
|
||||
@@ -78,12 +134,15 @@ def _normalize_stack(graphobjs):
|
||||
yield (operands, operator)
|
||||
|
||||
|
||||
def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
def _interpret_contents(
|
||||
contentstream: Object, initial_shorthand=UNIT_SQUARE, initial_fill_ink=Ink.mono
|
||||
):
|
||||
"""Interpret the PDF content stream.
|
||||
|
||||
The stack represents the state of the PDF graphics stack. We are only
|
||||
interested in the current transformation matrix (CTM) so we only track
|
||||
this object; a full implementation would need to track many other items.
|
||||
The stack represents the state of the PDF graphics stack. We track the
|
||||
current transformation matrix (CTM) and the current fill color (so that
|
||||
image masks, which are painted with the fill color, can be classified);
|
||||
a full implementation would need to track many other items.
|
||||
|
||||
The CTM is initialized to the mapping from user space to device space.
|
||||
PDF units are 1/72". In a PDF viewer or printer this matrix is initialized
|
||||
@@ -102,26 +161,29 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
stack depth exceeds the spec limit and set a hard limit beyond this to
|
||||
bound our memory requirements. If the stack underflows behavior is
|
||||
undefined in the spec, but we just pretend nothing happened and leave the
|
||||
CTM unchanged.
|
||||
graphics state unchanged.
|
||||
"""
|
||||
stack = []
|
||||
ctm = Matrix(initial_shorthand)
|
||||
fill_ink = initial_fill_ink # PDF default fill color is black
|
||||
fill_space = '/DeviceGray' # current fill colorspace name (for sc/scn)
|
||||
xobject_settings: list[XobjectSettings] = []
|
||||
inline_images: list[InlineSettings] = []
|
||||
name_index = defaultdict(lambda: [])
|
||||
found_vector = False
|
||||
found_text = False
|
||||
vector_ops = set('S s f F f* B B* b b*'.split())
|
||||
text_showing_ops = set("""TJ Tj " '""".split())
|
||||
image_ops = set('BI ID EI q Q Do cm'.split())
|
||||
operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops)
|
||||
vector_ops = set(['S', 's', 'f', 'F', 'f*', 'B', 'B*', 'b', 'b*'])
|
||||
text_showing_ops = set(["TJ", "Tj", '"', "'"])
|
||||
image_ops = set(['BI', 'ID', 'EI', 'q', 'Q', 'Do', 'cm'])
|
||||
color_ops = set(['g', 'rg', 'k', 'cs', 'sc', 'scn'])
|
||||
operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops | color_ops)
|
||||
|
||||
for n, graphobj in enumerate(
|
||||
_normalize_stack(parse_content_stream(contentstream, operator_whitelist))
|
||||
):
|
||||
operands, operator = graphobj
|
||||
if operator == 'q':
|
||||
stack.append(ctm)
|
||||
stack.append((ctm, fill_ink, fill_space))
|
||||
if len(stack) > 32: # See docstring
|
||||
if len(stack) > 128:
|
||||
raise RuntimeError(
|
||||
@@ -130,9 +192,9 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
warn("PDF graphics stack overflowed spec limit")
|
||||
elif operator == 'Q':
|
||||
try:
|
||||
ctm = stack.pop()
|
||||
ctm, fill_ink, fill_space = stack.pop()
|
||||
except IndexError:
|
||||
# Keeping the ctm the same seems to be the only sensible thing
|
||||
# Keeping the state the same seems to be the only sensible thing
|
||||
# to do. Just pretend nothing happened, keep calm and carry on.
|
||||
warn("PDF graphics stack underflowed - PDF may be malformed")
|
||||
elif operator == 'cm':
|
||||
@@ -143,17 +205,51 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
"PDF content stream is corrupt - this PDF is malformed. "
|
||||
"Use a PDF editor that is capable of visually inspecting the PDF."
|
||||
) from e
|
||||
elif operator == 'g':
|
||||
if vals := _operand_floats(operands):
|
||||
fill_ink = _ink_from_components('gray', vals)
|
||||
fill_space = '/DeviceGray'
|
||||
elif operator == 'rg':
|
||||
if vals := _operand_floats(operands):
|
||||
fill_ink = _ink_from_components('rgb', vals)
|
||||
fill_space = '/DeviceRGB'
|
||||
elif operator == 'k':
|
||||
if vals := _operand_floats(operands):
|
||||
fill_ink = _ink_from_components('cmyk', vals)
|
||||
fill_space = '/DeviceCMYK'
|
||||
elif operator == 'cs':
|
||||
# Selecting a colorspace resets the fill color to that space's
|
||||
# initial value, which is black for all device colorspaces.
|
||||
fill_ink = Ink.mono
|
||||
if operands:
|
||||
fill_space = str(operands[0])
|
||||
elif operator in ('sc', 'scn'):
|
||||
if any(isinstance(o, Name) for o in operands):
|
||||
fill_ink = Ink.color # pattern fill
|
||||
else:
|
||||
space = _DEVICE_FILL_SPACE.get(fill_space)
|
||||
vals = _operand_floats(operands)
|
||||
if space is None or vals is None:
|
||||
fill_ink = Ink.color # conservative for non-device space
|
||||
else:
|
||||
fill_ink = _ink_from_components(space, vals)
|
||||
elif operator == 'Do':
|
||||
image_name = operands[0]
|
||||
settings = XobjectSettings(
|
||||
name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack)
|
||||
name=image_name,
|
||||
shorthand=ctm.shorthand,
|
||||
stack_depth=len(stack),
|
||||
fill_ink=fill_ink,
|
||||
)
|
||||
xobject_settings.append(settings)
|
||||
name_index[str(image_name)].append(settings)
|
||||
elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this
|
||||
iimage = operands[0]
|
||||
inline = InlineSettings(
|
||||
iimage=iimage, shorthand=ctm.shorthand, stack_depth=len(stack)
|
||||
iimage=iimage,
|
||||
shorthand=ctm.shorthand,
|
||||
stack_depth=len(stack),
|
||||
fill_ink=fill_ink,
|
||||
)
|
||||
inline_images.append(inline)
|
||||
elif operator in vector_ops:
|
||||
|
||||
@@ -7,6 +7,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
from collections.abc import Iterator
|
||||
from decimal import Decimal
|
||||
from typing import cast
|
||||
|
||||
from pikepdf import (
|
||||
Dictionary,
|
||||
@@ -20,7 +21,7 @@ from pikepdf import (
|
||||
UnsupportedImageTypeError,
|
||||
)
|
||||
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.helpers import Resolution, pikepdf_get_int
|
||||
from ocrmypdf.pdfinfo._contentstream import (
|
||||
ContentsInfo,
|
||||
TextMarker,
|
||||
@@ -36,6 +37,7 @@ from ocrmypdf.pdfinfo._types import (
|
||||
UNIT_SQUARE,
|
||||
Colorspace,
|
||||
Encoding,
|
||||
Ink,
|
||||
)
|
||||
|
||||
logger = logging.getLogger()
|
||||
@@ -53,6 +55,7 @@ class ImageInfo:
|
||||
|
||||
_comp: int | None
|
||||
_name: str
|
||||
_enc: Encoding | None
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
@@ -61,10 +64,12 @@ class ImageInfo:
|
||||
pdfimage: Object | None = None,
|
||||
inline: PdfInlineImage | None = None,
|
||||
shorthand=None,
|
||||
fill_ink: Ink | None = None,
|
||||
):
|
||||
"""Initialize an ImageInfo."""
|
||||
self._name = str(name)
|
||||
self._shorthand = shorthand
|
||||
self._fill_ink = fill_ink
|
||||
|
||||
pim: PdfInlineImage | PdfImage
|
||||
|
||||
@@ -87,8 +92,8 @@ class ImageInfo:
|
||||
# itself. Some PDF writers use this to create a grayscale stencil
|
||||
# mask. For our purposes, the effective size is the size of the
|
||||
# larger component (image or smask).
|
||||
self._width = max(smask.get(Name.Width, 0), self._width)
|
||||
self._height = max(smask.get(Name.Height, 0), self._height)
|
||||
self._width = max(pikepdf_get_int(smask, Name.Width), self._width)
|
||||
self._height = max(pikepdf_get_int(smask, Name.Height), self._height)
|
||||
if (mask := pim.obj.get(Name.Mask, None)) is not None and isinstance(
|
||||
mask, Stream | Dictionary
|
||||
):
|
||||
@@ -96,8 +101,8 @@ class ImageInfo:
|
||||
# /Mask can be a Stream or an Array. If it's a Stream,
|
||||
# use its /Width and /Height if they are larger than the main
|
||||
# image's.
|
||||
self._width = max(mask.get(Name.Width, 0), self._width)
|
||||
self._height = max(mask.get(Name.Height, 0), self._height)
|
||||
self._width = max(pikepdf_get_int(mask, Name.Width), self._width)
|
||||
self._height = max(pikepdf_get_int(mask, Name.Height), self._height)
|
||||
|
||||
# If /ImageMask is true, then this image is a stencil mask
|
||||
# (Images that draw with this stencil mask will have a reference to
|
||||
@@ -175,6 +180,17 @@ class ImageInfo:
|
||||
"""Type of image, either 'image' or 'stencil'."""
|
||||
return self._type
|
||||
|
||||
@property
|
||||
def ink(self) -> Ink | None:
|
||||
"""Fill-color classification for stencil masks, else None.
|
||||
|
||||
A stencil (image mask) is painted with the current fill color; this
|
||||
reports whether that color is mono/gray/color so the rasterizer can
|
||||
choose a device that does not discard the distinction. Non-stencil
|
||||
images return None.
|
||||
"""
|
||||
return self._fill_ink if self._type == 'stencil' else None
|
||||
|
||||
@property
|
||||
def width(self) -> int:
|
||||
"""Width of the image in pixels."""
|
||||
@@ -249,7 +265,10 @@ def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
||||
"""Find inline images in the contentstream."""
|
||||
for n, inline in enumerate(contentsinfo.inline_images):
|
||||
yield ImageInfo(
|
||||
name=f'inline-{n:02d}', shorthand=inline.shorthand, inline=inline.iimage
|
||||
name=f'inline-{n:02d}',
|
||||
shorthand=inline.shorthand,
|
||||
inline=inline.iimage,
|
||||
fill_ink=inline.fill_ink,
|
||||
)
|
||||
|
||||
|
||||
@@ -268,9 +287,15 @@ def _image_xobjects(container) -> Iterator[tuple[Object, str]]:
|
||||
if Name.Resources not in container:
|
||||
return
|
||||
resources = container[Name.Resources]
|
||||
if Name.XObject not in resources:
|
||||
# A malformed PDF may store a non-dictionary at /Resources or
|
||||
# /Resources /XObject; treat that as "no image XObjects" instead of
|
||||
# crashing when we try to iterate it.
|
||||
if not isinstance(resources, Dictionary):
|
||||
return
|
||||
for key, candidate in resources[Name.XObject].items():
|
||||
xobjects = resources.get(Name.XObject)
|
||||
if not isinstance(xobjects, Dictionary):
|
||||
return
|
||||
for key, candidate in xobjects.items():
|
||||
if candidate is None or Name.Subtype not in candidate:
|
||||
continue
|
||||
if candidate[Name.Subtype] == Name.Image:
|
||||
@@ -300,7 +325,12 @@ def _find_regular_images(
|
||||
# these from our DPI calculation for the page.
|
||||
continue
|
||||
|
||||
yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand)
|
||||
yield ImageInfo(
|
||||
name=draw.name,
|
||||
pdfimage=pdfimage,
|
||||
shorthand=draw.shorthand,
|
||||
fill_ink=draw.fill_ink,
|
||||
)
|
||||
|
||||
|
||||
def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo):
|
||||
@@ -312,9 +342,14 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
||||
if Name.Resources not in container:
|
||||
return
|
||||
resources = container[Name.Resources]
|
||||
if Name.XObject not in resources:
|
||||
# As in _image_xobjects, tolerate a non-dictionary /Resources or
|
||||
# /Resources /XObject in a malformed PDF rather than crashing.
|
||||
if not isinstance(resources, Dictionary):
|
||||
return
|
||||
xobjs = resources[Name.XObject].as_dict()
|
||||
xobject = resources.get(Name.XObject)
|
||||
if not isinstance(xobject, Dictionary):
|
||||
return
|
||||
xobjs = xobject.as_dict()
|
||||
for xobj in xobjs:
|
||||
candidate = xobjs[xobj]
|
||||
if candidate is None or candidate.get(Name.Subtype) != Name.Form:
|
||||
@@ -330,13 +365,19 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
||||
# but in practice both Form XObjects and multiple drawing of the
|
||||
# same object are both very rare.
|
||||
ctm_shorthand = settings.shorthand
|
||||
# A Form XObject inherits the graphics state (including fill color)
|
||||
# in effect at the Do that draws it, so a mask painted with an
|
||||
# inherited gray/color fill must carry that classification inward.
|
||||
yield from _process_content_streams(
|
||||
pdf=pdf, container=form_xobject, shorthand=ctm_shorthand
|
||||
pdf=pdf,
|
||||
container=form_xobject,
|
||||
shorthand=ctm_shorthand,
|
||||
initial_fill_ink=settings.fill_ink,
|
||||
)
|
||||
|
||||
|
||||
def _process_content_streams(
|
||||
*, pdf: Pdf, container: Object, shorthand=None
|
||||
*, pdf: Pdf, container: Object, shorthand=None, initial_fill_ink=Ink.mono
|
||||
) -> Iterator[VectorMarker | TextMarker | ImageInfo]:
|
||||
"""Find all individual instances of images drawn in the container.
|
||||
|
||||
@@ -368,7 +409,9 @@ def _process_content_streams(
|
||||
# A Form XObject may provide its own matrix to map form space into
|
||||
# user space. Get this if one exists
|
||||
form_shorthand = container.get(Name.Matrix, Matrix())
|
||||
form_matrix = Matrix(form_shorthand)
|
||||
# pikepdf's Matrix() stub omits the Object/Array overload, but the
|
||||
# underlying C++ implementation accepts any 6-element numeric array.
|
||||
form_matrix = Matrix(cast(Matrix, form_shorthand))
|
||||
|
||||
# Concatenate form matrix with CTM to ensure CTM is correct for
|
||||
# drawing this instance of the XObject
|
||||
@@ -377,7 +420,7 @@ def _process_content_streams(
|
||||
else:
|
||||
return
|
||||
|
||||
contentsinfo = _interpret_contents(container, initial_shorthand)
|
||||
contentsinfo = _interpret_contents(container, initial_shorthand, initial_fill_ink)
|
||||
|
||||
if contentsinfo.found_vector:
|
||||
yield VectorMarker()
|
||||
|
||||
@@ -39,6 +39,20 @@ class Encoding(Enum):
|
||||
flate_jpeg = auto()
|
||||
|
||||
|
||||
class Ink(Enum):
|
||||
"""Classification of the fill color used to paint a stencil image mask.
|
||||
|
||||
A stencil (image mask) is painted with the current fill color, so the
|
||||
color depth needed to rasterize it for OCR depends on that fill color,
|
||||
not on the mask's 1-bit data.
|
||||
"""
|
||||
|
||||
# pylint: disable=invalid-name
|
||||
mono = auto() # black (or no color information to preserve)
|
||||
gray = auto() # achromatic but not pure black
|
||||
color = auto() # chromatic, or a fill we cannot prove is achromatic
|
||||
|
||||
|
||||
FloatRect = tuple[float, float, float, float]
|
||||
|
||||
FRIENDLY_COLORSPACE: dict[str, Colorspace] = {
|
||||
|
||||
@@ -6,7 +6,7 @@ from __future__ import annotations
|
||||
|
||||
import atexit
|
||||
import logging
|
||||
from collections.abc import Container, Sequence
|
||||
from collections.abc import Container
|
||||
from contextlib import contextmanager
|
||||
from functools import partial
|
||||
from pathlib import Path
|
||||
@@ -28,7 +28,7 @@ logger = logging.getLogger()
|
||||
worker_pdf = None # pylint: disable=invalid-name
|
||||
|
||||
|
||||
def _pdf_pageinfo_sync_init(pdf: Pdf, infile: Path, pdfminer_loglevel):
|
||||
def _pdf_pageinfo_sync_init(pdf: Pdf | None, infile: Path, pdfminer_loglevel):
|
||||
global worker_pdf # pylint: disable=global-statement,invalid-name
|
||||
pikepdf_enable_mmap()
|
||||
|
||||
@@ -75,16 +75,16 @@ def _pdf_pageinfo_sync(
|
||||
|
||||
|
||||
def _pdf_pageinfo_concurrent(
|
||||
pdf,
|
||||
pdf: Pdf,
|
||||
executor: Executor,
|
||||
max_workers: int,
|
||||
max_workers: int | None,
|
||||
use_threads: bool,
|
||||
infile,
|
||||
progbar,
|
||||
check_pages,
|
||||
infile: Path,
|
||||
progbar: bool,
|
||||
check_pages: Container[int],
|
||||
detailed_analysis: bool = False,
|
||||
miner_state: PdfMinerState | None = None,
|
||||
) -> Sequence[PageInfo | None]:
|
||||
) -> list[PageInfo | None]:
|
||||
pages: list[PageInfo | None] = [None] * len(pdf.pages)
|
||||
|
||||
def update_pageinfo(page: PageInfo, pbar: ProgressBar):
|
||||
|
||||
@@ -16,11 +16,12 @@ from pathlib import Path
|
||||
from typing import NamedTuple
|
||||
|
||||
from pdfminer.layout import LTPage, LTTextBox
|
||||
from pikepdf import Name, Page, Pdf
|
||||
from pikepdf import Name, Object, Page, Pdf
|
||||
|
||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||
from ocrmypdf._pageboxes import coerce_box
|
||||
from ocrmypdf.exceptions import EncryptedPdfError
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.helpers import Resolution, pikepdf_get_bool, pikepdf_get_int
|
||||
from ocrmypdf.pdfinfo._contentstream import TextboxInfo, TextMarker, VectorMarker
|
||||
from ocrmypdf.pdfinfo._image import ImageInfo, _process_content_streams
|
||||
from ocrmypdf.pdfinfo._types import FloatRect
|
||||
@@ -34,6 +35,12 @@ from ocrmypdf.pdfinfo.layout import (
|
||||
logger = logging.getLogger()
|
||||
|
||||
|
||||
def _box_rect(values: Iterable) -> FloatRect:
|
||||
"""Coerce a page box to a normalized ``FloatRect`` (4-tuple)."""
|
||||
b = coerce_box(values)
|
||||
return (b[0], b[1], b[2], b[3])
|
||||
|
||||
|
||||
def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) -> bool:
|
||||
"""Smarter text detection that ignores text in margins."""
|
||||
pw, ph = float(page_width), float(page_height) # pylint: disable=invalid-name
|
||||
@@ -140,19 +147,23 @@ class PageInfo:
|
||||
miner_state: PdfMinerState | None,
|
||||
):
|
||||
page: Page = pdf.pages[pageno]
|
||||
mediabox = [Decimal(d) for d in page.mediabox.as_list()]
|
||||
mediabox = [Decimal(str(d)) for d in coerce_box(page.mediabox.as_list())]
|
||||
width_pt = mediabox[2] - mediabox[0]
|
||||
height_pt = mediabox[3] - mediabox[1]
|
||||
|
||||
self._artbox = [float(d) for d in page.artbox.as_list()]
|
||||
self._bleedbox = [float(d) for d in page.bleedbox.as_list()]
|
||||
self._cropbox = [float(d) for d in page.cropbox.as_list()]
|
||||
self._mediabox = [float(d) for d in page.mediabox.as_list()]
|
||||
self._trimbox = [float(d) for d in page.trimbox.as_list()]
|
||||
self._artbox = _box_rect(page.artbox.as_list())
|
||||
self._bleedbox = _box_rect(page.bleedbox.as_list())
|
||||
self._cropbox = _box_rect(page.cropbox.as_list())
|
||||
self._mediabox = _box_rect(page.mediabox.as_list())
|
||||
self._trimbox = _box_rect(page.trimbox.as_list())
|
||||
|
||||
check_this_page = pageno in check_pages
|
||||
|
||||
if check_this_page and detailed_analysis:
|
||||
# miner_state is only None when detailed_analysis is False (see
|
||||
# PdfInfo.__init__, which ties the two together), so it must be
|
||||
# set here.
|
||||
assert miner_state is not None
|
||||
page_analysis = miner_state.get_page_analysis(pageno)
|
||||
if page_analysis is not None:
|
||||
self._textboxes = list(
|
||||
@@ -168,7 +179,11 @@ class PageInfo:
|
||||
self._has_text = None # i.e. "no information"
|
||||
|
||||
userunit = page.get(Name.UserUnit, Decimal(1.0))
|
||||
if not isinstance(userunit, Decimal):
|
||||
if isinstance(userunit, Object):
|
||||
# Only reachable under pikepdf's explicit conversion mode; the
|
||||
# default (implicit) mode already unboxes to int/float/Decimal.
|
||||
userunit = Decimal(userunit.as_float())
|
||||
elif not isinstance(userunit, Decimal):
|
||||
userunit = Decimal(userunit)
|
||||
self._userunit = userunit
|
||||
self._width_inches = width_pt * userunit / Decimal(72.0)
|
||||
@@ -182,7 +197,7 @@ class PageInfo:
|
||||
self._has_text = False
|
||||
self._images = []
|
||||
for info in _process_content_streams(
|
||||
pdf=pdf, container=page, shorthand=userunit_shorthand
|
||||
pdf=pdf, container=page.obj, shorthand=userunit_shorthand
|
||||
):
|
||||
if isinstance(info, VectorMarker):
|
||||
self._has_vector = True
|
||||
@@ -398,6 +413,7 @@ class PdfInfo:
|
||||
_has_acroform: bool = False
|
||||
_has_signature: bool = False
|
||||
_needs_rendering: bool = False
|
||||
_has_structure_tree: bool = False
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
@@ -438,17 +454,23 @@ class PdfInfo:
|
||||
detailed_analysis=detailed_analysis,
|
||||
miner_state=miner_state,
|
||||
)
|
||||
self._needs_rendering = pdf.Root.get(Name.NeedsRendering, False)
|
||||
self._needs_rendering = pikepdf_get_bool(pdf.Root, Name.NeedsRendering)
|
||||
if Name.AcroForm in pdf.Root:
|
||||
if (
|
||||
len(pdf.Root.AcroForm.get(Name.Fields, [])) > 0
|
||||
or Name.XFA in pdf.Root.AcroForm
|
||||
):
|
||||
self._has_acroform = True
|
||||
self._has_signature = bool(pdf.Root.AcroForm.get(Name.SigFlags, 0) & 1)
|
||||
self._is_tagged = bool(
|
||||
pdf.Root.get(Name.MarkInfo, {}).get(Name.Marked, False)
|
||||
self._has_signature = bool(
|
||||
pikepdf_get_int(pdf.Root.AcroForm, Name.SigFlags) & 1
|
||||
)
|
||||
mark_info = pdf.Root.get(Name.MarkInfo)
|
||||
self._is_tagged = (
|
||||
pikepdf_get_bool(mark_info, Name.Marked)
|
||||
if mark_info is not None
|
||||
else False
|
||||
)
|
||||
self._has_structure_tree = Name.StructTreeRoot in pdf.Root
|
||||
|
||||
@property
|
||||
def pages(self) -> list[PageInfo | None]:
|
||||
@@ -481,6 +503,11 @@ class PdfInfo:
|
||||
"""Return True if the document catalog indicates this is a Tagged PDF."""
|
||||
return self._is_tagged
|
||||
|
||||
@property
|
||||
def has_structure_tree(self) -> bool:
|
||||
"""Return True if the document catalog has a logical structure tree."""
|
||||
return self._has_structure_tree
|
||||
|
||||
@property
|
||||
def filename(self) -> str | Path:
|
||||
"""Return filename of PDF."""
|
||||
@@ -523,6 +550,8 @@ def main(): # pragma: no cover
|
||||
pprint(pdfinfo)
|
||||
for page in pdfinfo.pages:
|
||||
pprint(page)
|
||||
if page is None:
|
||||
continue
|
||||
for im in page.images:
|
||||
pprint(im)
|
||||
|
||||
|
||||
@@ -5,20 +5,19 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import sys
|
||||
from collections.abc import Iterator, Mapping
|
||||
from contextlib import contextmanager
|
||||
from math import copysign
|
||||
from os import PathLike
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from typing import Any, BinaryIO
|
||||
from unittest.mock import patch
|
||||
|
||||
import pdfminer
|
||||
import pdfminer.encodingdb
|
||||
import pdfminer.pdfdevice
|
||||
import pdfminer.pdfinterp
|
||||
import pdfminer.psparser
|
||||
from deprecation import deprecated
|
||||
from pdfminer.converter import PDFLayoutAnalyzer
|
||||
from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox
|
||||
from pdfminer.pdfcolor import PDFColorSpace
|
||||
@@ -31,6 +30,11 @@ from pdfminer.utils import Matrix, bbox2str, matrix2str
|
||||
|
||||
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
||||
|
||||
if sys.version_info >= (3, 13):
|
||||
from warnings import deprecated
|
||||
else:
|
||||
from typing_extensions import deprecated
|
||||
|
||||
STRIP_NAME = re.compile(r'[0-9]+')
|
||||
|
||||
|
||||
@@ -58,12 +62,7 @@ def pdfsimplefont__init__(
|
||||
return
|
||||
|
||||
|
||||
PDFSimpleFont.__init__ = pdfsimplefont__init__
|
||||
|
||||
# Patch pdfminer.six buffer size
|
||||
# The parser doesn't properly handle keyword tokens are split across the end of the
|
||||
# buffer, so increase the buffer size something far larger than will ever be seen.
|
||||
pdfminer.psparser.PSBaseParser.BUFSIZ = 256 * 1024 * 1024
|
||||
PDFSimpleFont.__init__ = pdfsimplefont__init__ # type: ignore[method-assign]
|
||||
|
||||
|
||||
def pdftype3font__pscript5_get_height(self):
|
||||
@@ -290,7 +289,7 @@ def patch_pdfminer(pscript5_mode: bool):
|
||||
yield
|
||||
|
||||
|
||||
@deprecated(deprecated_in='16.6.0', details='Use PdfMinerState instead.')
|
||||
@deprecated('Deprecated since 16.6.0; use PdfMinerState instead.')
|
||||
def get_page_analysis(
|
||||
infile: PathLike, pageno: int, pscript5_mode: bool
|
||||
) -> LTPage | None:
|
||||
@@ -338,10 +337,10 @@ class PdfMinerState:
|
||||
self.infile = infile
|
||||
self.rman = pdfminer.pdfinterp.PDFResourceManager(caching=True)
|
||||
self.disable_boxes_flow = None
|
||||
self.page_iter = None
|
||||
self.page_iter: Iterator[PDFPage] | None = None
|
||||
self.page_cache: list[PDFPage] = []
|
||||
self.pscript5_mode = pscript5_mode
|
||||
self.file = None
|
||||
self.file: BinaryIO | None = None
|
||||
|
||||
def __enter__(self):
|
||||
"""Enter the context manager."""
|
||||
@@ -357,6 +356,7 @@ class PdfMinerState:
|
||||
|
||||
def get_page_analysis(self, pageno: int):
|
||||
"""Get the page analysis for a given page."""
|
||||
assert self.page_iter is not None, "must be used as a context manager"
|
||||
while len(self.page_cache) <= pageno:
|
||||
try:
|
||||
self.page_cache.append(next(self.page_iter))
|
||||
|
||||
@@ -38,6 +38,7 @@ class GhostscriptRasterDevice(StrEnum):
|
||||
JPEGGRAY = 'jpeggray'
|
||||
JPEGCOLOR = 'jpeg'
|
||||
PNGMONO = 'pngmono'
|
||||
PNGMONOD = 'pngmonod'
|
||||
PNGGRAY = 'pnggray'
|
||||
PNG256 = 'png256'
|
||||
PNG16M = 'png16m'
|
||||
|
||||
@@ -1,345 +1,31 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Wrappers to manage subprocess calls."""
|
||||
"""Wrappers to manage subprocess calls.
|
||||
|
||||
This package is split into three private submodules by concern:
|
||||
|
||||
- :mod:`ocrmypdf.subprocess._run` - low-level execution wrappers (``run``,
|
||||
``run_polling_stderr``) that add OCRmyPDF-aware logging and Windows PATH
|
||||
resolution. Useful as drop-in replacements for :func:`subprocess.run`.
|
||||
- :mod:`ocrmypdf.subprocess._version` - version probing (``get_version``).
|
||||
- :mod:`ocrmypdf.subprocess._check` - startup validation
|
||||
(``check_external_program``) with platform-aware error messages.
|
||||
|
||||
The names below are the stable public API. Importing from the private
|
||||
submodules directly is not supported for external code.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
from contextlib import suppress
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||
from subprocess import run as subprocess_run
|
||||
from ocrmypdf.subprocess._check import check_external_program
|
||||
from ocrmypdf.subprocess._run import Args, Environ, run, run_polling_stderr
|
||||
from ocrmypdf.subprocess._version import get_version
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
Args = Sequence[Path | str]
|
||||
Environ = Mapping[str, str] | os._Environ # pylint: disable=protected-access
|
||||
|
||||
|
||||
def run(
|
||||
args: Args,
|
||||
*,
|
||||
env: Environ | None = None,
|
||||
logs_errors_to_stdout: bool = False,
|
||||
check: bool = False,
|
||||
**kwargs,
|
||||
) -> CompletedProcess:
|
||||
"""Wrapper around :py:func:`subprocess.run`.
|
||||
|
||||
The main purpose of this wrapper is to log subprocess output in an orderly
|
||||
fashion that identifies the responsible subprocess. An additional
|
||||
task is that this function goes to greater lengths to find possible Windows
|
||||
locations of our dependencies when they are not on the system PATH.
|
||||
|
||||
Arguments should be identical to ``subprocess.run``, except for following:
|
||||
|
||||
Args:
|
||||
args: Positional arguments to pass to ``subprocess.run``.
|
||||
env: A set of environment variables. If None, the OS environment is used.
|
||||
logs_errors_to_stdout: If True, indicates that the process writes its error
|
||||
messages to stdout rather than stderr, so stdout should be logged
|
||||
if there is an error. If False, stderr is logged. Could be used with
|
||||
stderr=STDOUT, stdout=PIPE for example.
|
||||
check: If True, raise an exception if the process exits with a non-zero
|
||||
status code. If False, the return value will indicate success or failure.
|
||||
kwargs: Additional arguments to pass to ``subprocess.run``.
|
||||
"""
|
||||
args, env, process_log, _text = _fix_process_args(args, env, kwargs)
|
||||
|
||||
stderr = None
|
||||
stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout'
|
||||
try:
|
||||
proc = subprocess_run(args, env=env, check=check, **kwargs)
|
||||
except CalledProcessError as e:
|
||||
stderr = getattr(e, stderr_name, None)
|
||||
raise
|
||||
else:
|
||||
stderr = getattr(proc, stderr_name, None)
|
||||
finally:
|
||||
if process_log.isEnabledFor(logging.DEBUG) and stderr:
|
||||
with suppress(AttributeError, UnicodeDecodeError):
|
||||
stderr = stderr.decode('utf-8', 'replace')
|
||||
if logs_errors_to_stdout:
|
||||
process_log.debug("stdout/stderr = %s", stderr)
|
||||
else:
|
||||
process_log.debug("stderr = %s", stderr)
|
||||
return proc
|
||||
|
||||
|
||||
def run_polling_stderr(
|
||||
args: Args,
|
||||
*,
|
||||
callback: Callable[[str], None],
|
||||
check: bool = False,
|
||||
env: Environ | None = None,
|
||||
**kwargs,
|
||||
) -> CompletedProcess:
|
||||
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
||||
|
||||
Every line of produced by stderr will be forwarded to the callback function.
|
||||
The intended use is monitoring progress of subprocesses that output their
|
||||
own progress indicators. In addition, each line will be logged if debug
|
||||
logging is enabled.
|
||||
|
||||
Requires stderr to be opened in text mode for ease of handling errors. In
|
||||
addition the expected encoding= and errors= arguments should be set. Note
|
||||
that if stdout is already set up, it need not be binary.
|
||||
"""
|
||||
args, env, process_log, text = _fix_process_args(args, env, kwargs)
|
||||
assert text, "Must use text=True"
|
||||
|
||||
with Popen(args, env=env, **kwargs) as proc:
|
||||
lines = []
|
||||
while proc.poll() is None:
|
||||
if proc.stderr is None:
|
||||
continue
|
||||
for msg in iter(proc.stderr.readline, ''):
|
||||
if process_log.isEnabledFor(logging.DEBUG):
|
||||
process_log.debug(msg.strip())
|
||||
callback(msg)
|
||||
lines.append(msg)
|
||||
stderr = ''.join(lines)
|
||||
|
||||
if check and proc.returncode != 0:
|
||||
raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr)
|
||||
return CompletedProcess(args, proc.returncode, None, stderr=stderr)
|
||||
|
||||
|
||||
def _fix_process_args(
|
||||
args: Args, env: Environ | None, kwargs
|
||||
) -> tuple[Args, Environ, logging.Logger, bool]:
|
||||
if not env:
|
||||
env = os.environ
|
||||
|
||||
# Search in spoof path if necessary
|
||||
program = str(args[0])
|
||||
|
||||
if sys.platform == 'win32':
|
||||
# pylint: disable=import-outside-toplevel
|
||||
from ocrmypdf.subprocess._windows import fix_windows_args
|
||||
|
||||
args = fix_windows_args(program, args, env)
|
||||
|
||||
log.debug("Running: %s", args)
|
||||
process_log = log.getChild(os.path.basename(program))
|
||||
text = bool(kwargs.get('text', False))
|
||||
|
||||
return args, env, process_log, text
|
||||
|
||||
|
||||
def get_version(
|
||||
program: str,
|
||||
*,
|
||||
version_arg: str = '--version',
|
||||
regex=r'(\d+(\.\d+)*)',
|
||||
env: Environ | None = None,
|
||||
) -> str:
|
||||
"""Get the version of the specified program.
|
||||
|
||||
Arguments:
|
||||
program: The program to version check.
|
||||
version_arg: The argument needed to ask for its version, e.g. ``--version``.
|
||||
regex: A regular expression to parse the program's output and obtain the
|
||||
version.
|
||||
env: Custom ``os.environ`` in which to run program.
|
||||
"""
|
||||
args_prog = [program, version_arg]
|
||||
try:
|
||||
proc = run(
|
||||
args_prog,
|
||||
close_fds=True,
|
||||
text=True,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
check=True,
|
||||
env=env,
|
||||
)
|
||||
output: str = proc.stdout
|
||||
except FileNotFoundError as e:
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
except CalledProcessError as e:
|
||||
if e.returncode != 0:
|
||||
log.exception(e)
|
||||
raise MissingDependencyError(
|
||||
f"Ran program '{program}' but it exited with an error:\n{e.output}"
|
||||
) from e
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
|
||||
match = re.match(regex, output.strip())
|
||||
if not match:
|
||||
raise MissingDependencyError(
|
||||
f"The program '{program}' did not report its version. "
|
||||
f"Message was:\n{output}"
|
||||
)
|
||||
version = match.group(1)
|
||||
|
||||
return version
|
||||
|
||||
|
||||
MISSING_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH.
|
||||
'''
|
||||
|
||||
MISSING_OPTIONAL_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH. This program is required when you use the
|
||||
{required_for} arguments. You could try omitting these arguments, or install
|
||||
the package.
|
||||
'''
|
||||
|
||||
MISSING_RECOMMEND_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH. This program is recommended when using the {required_for} arguments,
|
||||
but not required, so we will proceed. For best results, install the program.
|
||||
'''
|
||||
|
||||
OLD_VERSION = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher. Your system appears
|
||||
to have {found_version}. Please update this program.
|
||||
'''
|
||||
|
||||
OLD_VERSION_REQUIRED_FOR = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
||||
{required_for} arguments. {program} {found_version} is installed.
|
||||
|
||||
If you omit these arguments, OCRmyPDF may be able to
|
||||
proceed. For best results, update the program.
|
||||
'''
|
||||
|
||||
OSX_INSTALL_ADVICE = '''
|
||||
If you have homebrew installed, try these command to install the missing
|
||||
package:
|
||||
brew install {package}
|
||||
'''
|
||||
|
||||
LINUX_INSTALL_ADVICE = '''
|
||||
On systems with the aptitude package manager (Debian, Ubuntu), try these
|
||||
commands:
|
||||
sudo apt update
|
||||
sudo apt install {package}
|
||||
|
||||
On RPM-based systems (Red Hat, Fedora), try this command:
|
||||
sudo dnf install {package}
|
||||
'''
|
||||
|
||||
WINDOWS_INSTALL_ADVICE = '''
|
||||
If not already installed, install the Chocolatey package manager. Then use
|
||||
a command prompt to install the missing package:
|
||||
choco install {package}
|
||||
'''
|
||||
|
||||
|
||||
def _get_platform() -> str:
|
||||
if sys.platform.startswith('freebsd'):
|
||||
return 'freebsd'
|
||||
elif sys.platform.startswith('linux'):
|
||||
return 'linux'
|
||||
elif sys.platform.startswith('win'):
|
||||
return 'windows'
|
||||
return sys.platform
|
||||
|
||||
|
||||
def _error_trailer(program: str, package: str | Mapping[str, str], **kwargs) -> None:
|
||||
del kwargs
|
||||
if isinstance(package, Mapping):
|
||||
package = package.get(_get_platform(), program)
|
||||
|
||||
if _get_platform() == 'darwin':
|
||||
log.info(OSX_INSTALL_ADVICE.format(**locals()))
|
||||
elif _get_platform() == 'linux':
|
||||
log.info(LINUX_INSTALL_ADVICE.format(**locals()))
|
||||
elif _get_platform() == 'windows':
|
||||
log.info(WINDOWS_INSTALL_ADVICE.format(**locals()))
|
||||
|
||||
|
||||
def _error_missing_program(
|
||||
program: str, package: str, required_for: str | None, recommended: bool
|
||||
) -> None:
|
||||
# pylint: disable=unused-argument
|
||||
if recommended:
|
||||
log.warning(MISSING_RECOMMEND_PROGRAM.format(**locals()))
|
||||
elif required_for:
|
||||
log.error(MISSING_OPTIONAL_PROGRAM.format(**locals()))
|
||||
else:
|
||||
log.error(MISSING_PROGRAM.format(**locals()))
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def _error_old_version(
|
||||
program: str,
|
||||
package: str,
|
||||
need_version: str,
|
||||
found_version: str,
|
||||
required_for: str | None,
|
||||
) -> None:
|
||||
# pylint: disable=unused-argument
|
||||
if required_for:
|
||||
log.error(OLD_VERSION_REQUIRED_FOR.format(**locals()))
|
||||
else:
|
||||
log.error(OLD_VERSION.format(**locals()))
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def check_external_program(
|
||||
*,
|
||||
program: str,
|
||||
package: str,
|
||||
version_checker: Callable[[], Version],
|
||||
need_version: str | Version,
|
||||
required_for: str | None = None,
|
||||
recommended: bool = False,
|
||||
version_parser: type[Version] = Version,
|
||||
) -> None:
|
||||
"""Check for required version of external program and raise exception if not.
|
||||
|
||||
Args:
|
||||
program: The name of the program to test.
|
||||
package: The name of a software package that typically supplies this program.
|
||||
Usually the same as program.
|
||||
version_checker: A callable without arguments that retrieves the installed
|
||||
version of program.
|
||||
need_version: The minimum required version.
|
||||
required_for: The name of an argument of feature that requires this program.
|
||||
recommended: If this external program is recommended, instead of raising
|
||||
an exception, log a warning and allow execution to continue.
|
||||
version_parser: A class that should be used to parse and compare version
|
||||
numbers. Used when version numbers do not follow standard conventions.
|
||||
"""
|
||||
if not isinstance(need_version, Version):
|
||||
need_version = version_parser(need_version)
|
||||
try:
|
||||
found_version = version_checker()
|
||||
except (CalledProcessError, FileNotFoundError) as e:
|
||||
_error_missing_program(program, package, required_for, recommended)
|
||||
if not recommended:
|
||||
raise MissingDependencyError(program) from e
|
||||
return
|
||||
except MissingDependencyError:
|
||||
_error_missing_program(program, package, required_for, recommended)
|
||||
if not recommended:
|
||||
raise
|
||||
return
|
||||
|
||||
if found_version and found_version < need_version:
|
||||
_error_old_version(
|
||||
program, package, str(need_version), str(found_version), required_for
|
||||
)
|
||||
if not recommended:
|
||||
raise MissingDependencyError(program)
|
||||
|
||||
log.debug('Found %s %s', program, found_version)
|
||||
__all__ = [
|
||||
'Args',
|
||||
'Environ',
|
||||
'check_external_program',
|
||||
'get_version',
|
||||
'run',
|
||||
'run_polling_stderr',
|
||||
]
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Validate that required external programs are installed and new enough."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import sys
|
||||
from collections.abc import Callable, Mapping
|
||||
from subprocess import CalledProcessError
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
|
||||
log = logging.getLogger('ocrmypdf.subprocess')
|
||||
|
||||
|
||||
MISSING_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH.
|
||||
'''
|
||||
|
||||
MISSING_OPTIONAL_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH. This program is required when you use the
|
||||
{required_for} arguments. You could try omitting these arguments, or install
|
||||
the package.
|
||||
'''
|
||||
|
||||
MISSING_RECOMMEND_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH. This program is recommended when using the {required_for} arguments,
|
||||
but not required, so we will proceed. For best results, install the program.
|
||||
'''
|
||||
|
||||
OLD_VERSION = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher. Your system appears
|
||||
to have {found_version}. Please update this program.
|
||||
'''
|
||||
|
||||
OLD_VERSION_REQUIRED_FOR = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
||||
{required_for} arguments. {program} {found_version} is installed.
|
||||
|
||||
If you omit these arguments, OCRmyPDF may be able to
|
||||
proceed. For best results, update the program.
|
||||
'''
|
||||
|
||||
OSX_INSTALL_ADVICE = '''
|
||||
If you have homebrew installed, try these command to install the missing
|
||||
package:
|
||||
brew install {package}
|
||||
'''
|
||||
|
||||
LINUX_INSTALL_ADVICE = '''
|
||||
On systems with the aptitude package manager (Debian, Ubuntu), try these
|
||||
commands:
|
||||
sudo apt update
|
||||
sudo apt install {package}
|
||||
|
||||
On RPM-based systems (Red Hat, Fedora), try this command:
|
||||
sudo dnf install {package}
|
||||
'''
|
||||
|
||||
WINDOWS_INSTALL_ADVICE = '''
|
||||
If not already installed, install the Chocolatey package manager. Then use
|
||||
a command prompt to install the missing package:
|
||||
choco install {package}
|
||||
'''
|
||||
|
||||
|
||||
def _get_platform() -> str:
|
||||
if sys.platform.startswith('freebsd'):
|
||||
return 'freebsd'
|
||||
elif sys.platform.startswith('linux'):
|
||||
return 'linux'
|
||||
elif sys.platform.startswith('win'):
|
||||
return 'windows'
|
||||
return sys.platform
|
||||
|
||||
|
||||
def _error_trailer(program: str, package: str | Mapping[str, str], **kwargs) -> None:
|
||||
del kwargs
|
||||
if isinstance(package, Mapping):
|
||||
package = package.get(_get_platform(), program)
|
||||
|
||||
if _get_platform() == 'darwin':
|
||||
log.info(OSX_INSTALL_ADVICE.format(**locals()))
|
||||
elif _get_platform() == 'linux':
|
||||
log.info(LINUX_INSTALL_ADVICE.format(**locals()))
|
||||
elif _get_platform() == 'windows':
|
||||
log.info(WINDOWS_INSTALL_ADVICE.format(**locals()))
|
||||
|
||||
|
||||
def _error_missing_program(
|
||||
program: str,
|
||||
package: str | Mapping[str, str],
|
||||
required_for: str | None,
|
||||
recommended: bool,
|
||||
) -> None:
|
||||
# pylint: disable=unused-argument
|
||||
if recommended:
|
||||
log.warning(MISSING_RECOMMEND_PROGRAM.format(**locals()))
|
||||
elif required_for:
|
||||
log.error(MISSING_OPTIONAL_PROGRAM.format(**locals()))
|
||||
else:
|
||||
log.error(MISSING_PROGRAM.format(**locals()))
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def _error_old_version(
|
||||
program: str,
|
||||
package: str | Mapping[str, str],
|
||||
need_version: str,
|
||||
found_version: str,
|
||||
required_for: str | None,
|
||||
) -> None:
|
||||
# pylint: disable=unused-argument
|
||||
if required_for:
|
||||
log.error(OLD_VERSION_REQUIRED_FOR.format(**locals()))
|
||||
else:
|
||||
log.error(OLD_VERSION.format(**locals()))
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def check_external_program(
|
||||
*,
|
||||
program: str,
|
||||
package: str | Mapping[str, str],
|
||||
version_checker: Callable[[], Version],
|
||||
need_version: str | Version,
|
||||
required_for: str | None = None,
|
||||
recommended: bool = False,
|
||||
version_parser: type[Version] = Version,
|
||||
) -> None:
|
||||
"""Check for required version of external program and raise exception if not.
|
||||
|
||||
Args:
|
||||
program: The name of the program to test.
|
||||
package: The name of a software package that typically supplies this program.
|
||||
Usually the same as program.
|
||||
version_checker: A callable without arguments that retrieves the installed
|
||||
version of program.
|
||||
need_version: The minimum required version.
|
||||
required_for: The name of an argument of feature that requires this program.
|
||||
recommended: If this external program is recommended, instead of raising
|
||||
an exception, log a warning and allow execution to continue.
|
||||
version_parser: A class that should be used to parse and compare version
|
||||
numbers. Used when version numbers do not follow standard conventions.
|
||||
"""
|
||||
if not isinstance(need_version, Version):
|
||||
need_version = version_parser(need_version)
|
||||
try:
|
||||
found_version = version_checker()
|
||||
except (CalledProcessError, FileNotFoundError) as e:
|
||||
_error_missing_program(program, package, required_for, recommended)
|
||||
if not recommended:
|
||||
raise MissingDependencyError(program) from e
|
||||
return
|
||||
except MissingDependencyError:
|
||||
_error_missing_program(program, package, required_for, recommended)
|
||||
if not recommended:
|
||||
raise
|
||||
return
|
||||
|
||||
if found_version and found_version < need_version:
|
||||
_error_old_version(
|
||||
program, package, str(need_version), str(found_version), required_for
|
||||
)
|
||||
if not recommended:
|
||||
raise MissingDependencyError(program)
|
||||
|
||||
log.debug('Found %s %s', program, found_version)
|
||||
@@ -0,0 +1,137 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Low-level wrappers around :py:mod:`subprocess`.
|
||||
|
||||
These functions exist to give OCRmyPDF child processes uniform logging
|
||||
behavior and to route through any platform-specific PATH fix-ups before
|
||||
invocation. They are intended as drop-in replacements for
|
||||
:py:func:`subprocess.run` in contexts where that routing is desirable
|
||||
(for example, plugin-provided tools).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
from contextlib import suppress
|
||||
from pathlib import Path
|
||||
from subprocess import CalledProcessError, CompletedProcess, Popen
|
||||
from subprocess import run as subprocess_run
|
||||
|
||||
log = logging.getLogger('ocrmypdf.subprocess')
|
||||
|
||||
Args = Sequence[Path | str]
|
||||
Environ = Mapping[str, str] | os._Environ # pylint: disable=protected-access
|
||||
|
||||
|
||||
def run(
|
||||
args: Args,
|
||||
*,
|
||||
env: Environ | None = None,
|
||||
logs_errors_to_stdout: bool = False,
|
||||
check: bool = False,
|
||||
**kwargs,
|
||||
) -> CompletedProcess:
|
||||
"""Wrapper around :py:func:`subprocess.run`.
|
||||
|
||||
The main purpose of this wrapper is to log subprocess output in an orderly
|
||||
fashion that identifies the responsible subprocess. An additional
|
||||
task is that this function goes to greater lengths to find possible Windows
|
||||
locations of our dependencies when they are not on the system PATH.
|
||||
|
||||
Arguments should be identical to ``subprocess.run``, except for following:
|
||||
|
||||
Args:
|
||||
args: Positional arguments to pass to ``subprocess.run``.
|
||||
env: A set of environment variables. If None, the OS environment is used.
|
||||
logs_errors_to_stdout: If True, indicates that the process writes its error
|
||||
messages to stdout rather than stderr, so stdout should be logged
|
||||
if there is an error. If False, stderr is logged. Could be used with
|
||||
stderr=STDOUT, stdout=PIPE for example.
|
||||
check: If True, raise an exception if the process exits with a non-zero
|
||||
status code. If False, the return value will indicate success or failure.
|
||||
kwargs: Additional arguments to pass to ``subprocess.run``.
|
||||
"""
|
||||
args, env, process_log, _text = _fix_process_args(args, env, kwargs)
|
||||
|
||||
stderr = None
|
||||
stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout'
|
||||
try:
|
||||
proc = subprocess_run(args, env=env, check=check, **kwargs)
|
||||
except CalledProcessError as e:
|
||||
stderr = getattr(e, stderr_name, None)
|
||||
raise
|
||||
else:
|
||||
stderr = getattr(proc, stderr_name, None)
|
||||
finally:
|
||||
if process_log.isEnabledFor(logging.DEBUG) and stderr:
|
||||
with suppress(AttributeError, UnicodeDecodeError):
|
||||
stderr = stderr.decode('utf-8', 'replace')
|
||||
if logs_errors_to_stdout:
|
||||
process_log.debug("stdout/stderr = %s", stderr)
|
||||
else:
|
||||
process_log.debug("stderr = %s", stderr)
|
||||
return proc
|
||||
|
||||
|
||||
def run_polling_stderr(
|
||||
args: Args,
|
||||
*,
|
||||
callback: Callable[[str], None],
|
||||
check: bool = False,
|
||||
env: Environ | None = None,
|
||||
**kwargs,
|
||||
) -> CompletedProcess:
|
||||
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
||||
|
||||
Every line of produced by stderr will be forwarded to the callback function.
|
||||
The intended use is monitoring progress of subprocesses that output their
|
||||
own progress indicators. In addition, each line will be logged if debug
|
||||
logging is enabled.
|
||||
|
||||
Requires stderr to be opened in text mode for ease of handling errors. In
|
||||
addition the expected encoding= and errors= arguments should be set. Note
|
||||
that if stdout is already set up, it need not be binary.
|
||||
"""
|
||||
args, env, process_log, text = _fix_process_args(args, env, kwargs)
|
||||
assert text, "Must use text=True"
|
||||
|
||||
with Popen(args, env=env, **kwargs) as proc:
|
||||
lines = []
|
||||
while proc.poll() is None:
|
||||
if proc.stderr is None:
|
||||
continue
|
||||
for msg in iter(proc.stderr.readline, ''):
|
||||
if process_log.isEnabledFor(logging.DEBUG):
|
||||
process_log.debug(msg.strip())
|
||||
callback(msg)
|
||||
lines.append(msg)
|
||||
stderr = ''.join(lines)
|
||||
|
||||
if check and proc.returncode != 0:
|
||||
raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr)
|
||||
return CompletedProcess(args, proc.returncode, None, stderr=stderr)
|
||||
|
||||
|
||||
def _fix_process_args(
|
||||
args: Args, env: Environ | None, kwargs
|
||||
) -> tuple[Args, Environ, logging.Logger, bool]:
|
||||
if not env:
|
||||
env = os.environ
|
||||
|
||||
# Search in spoof path if necessary
|
||||
program = str(args[0])
|
||||
|
||||
if sys.platform == 'win32':
|
||||
# pylint: disable=import-outside-toplevel
|
||||
from ocrmypdf.subprocess._windows import fix_windows_args
|
||||
|
||||
args = fix_windows_args(program, args, env)
|
||||
|
||||
log.debug("Running: %s", args)
|
||||
process_log = log.getChild(Path(program).name)
|
||||
text = bool(kwargs.get('text', False))
|
||||
|
||||
return args, env, process_log, text
|
||||
@@ -0,0 +1,79 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Extract version strings from external programs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess._run import Environ
|
||||
|
||||
log = logging.getLogger('ocrmypdf.subprocess')
|
||||
|
||||
|
||||
def get_version(
|
||||
program: str,
|
||||
*,
|
||||
version_arg: str = '--version',
|
||||
regex=r'(\d+(\.\d+)*)',
|
||||
env: Environ | None = None,
|
||||
) -> str:
|
||||
"""Get the version of the specified program.
|
||||
|
||||
Arguments:
|
||||
program: The program to version check.
|
||||
version_arg: The argument needed to ask for its version, e.g. ``--version``.
|
||||
regex: A regular expression to parse the program's output and obtain the
|
||||
version.
|
||||
env: Custom ``os.environ`` in which to run program.
|
||||
"""
|
||||
# Late import of the public ``run`` so that tests patching
|
||||
# ``ocrmypdf.subprocess.run`` affect this function. Binding ``run`` at
|
||||
# module load time would capture the real implementation and bypass the
|
||||
# patch.
|
||||
from ocrmypdf import subprocess as _sp
|
||||
|
||||
args_prog = [program, version_arg]
|
||||
try:
|
||||
proc = _sp.run(
|
||||
args_prog,
|
||||
close_fds=True,
|
||||
text=True,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
check=True,
|
||||
env=env,
|
||||
)
|
||||
output: str = proc.stdout
|
||||
except FileNotFoundError as e:
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
except CalledProcessError as e:
|
||||
if e.returncode != 0:
|
||||
log.exception(e)
|
||||
raise MissingDependencyError(
|
||||
f"Ran program '{program}' but it exited with an error:\n{e.output}"
|
||||
) from e
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
|
||||
# Some tools (e.g. veraPDF launched on a recent JDK) print warnings before
|
||||
# the version line, so scan each line rather than only the start of output.
|
||||
version = None
|
||||
for line in output.splitlines():
|
||||
match = re.match(regex, line.strip())
|
||||
if match:
|
||||
version = match.group(1)
|
||||
break
|
||||
if version is None:
|
||||
raise MissingDependencyError(
|
||||
f"The program '{program}' did not report its version. "
|
||||
f"Message was:\n{output}"
|
||||
)
|
||||
|
||||
return version
|
||||
@@ -0,0 +1,42 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MIT
|
||||
"""Test plugin that deliberately writes garbage to stdout.
|
||||
|
||||
Used to verify that OCRmyPDF's stdout protection diverts stray writes (from
|
||||
plugins or libraries) to stderr, so that a PDF written to stdout is never
|
||||
corrupted. Pollutes at three points: plugin import (main process), the
|
||||
``validate`` hook (main process), and the ``filter_ocr_image`` hook (worker
|
||||
process/thread).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
from ocrmypdf import hookimpl
|
||||
|
||||
POLLUTION = b'POLLUTION'
|
||||
|
||||
|
||||
def _pollute(where: bytes) -> None:
|
||||
# Write to file descriptor 1 directly (as a careless C library might) and
|
||||
# via Python's sys.stdout (as a stray print() might).
|
||||
os.write(1, POLLUTION + b'-fd1-' + where + b'\n')
|
||||
print(POLLUTION.decode() + '-stdout-' + where.decode())
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
# Pollute at import time, which happens while plugins are being loaded.
|
||||
_pollute(b'import')
|
||||
|
||||
|
||||
@hookimpl
|
||||
def validate(pdfinfo, options):
|
||||
_pollute(b'validate')
|
||||
|
||||
|
||||
@hookimpl
|
||||
def filter_ocr_image(page, image):
|
||||
_pollute(b'filter_ocr_image')
|
||||
return image
|
||||
@@ -74,11 +74,11 @@ class FixedRotateNoopOcrEngine(OcrEngine):
|
||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||
with (
|
||||
Image.open(input_file) as im,
|
||||
open(output_hocr, 'w', encoding='utf-8') as f,
|
||||
output_hocr.open('w', encoding='utf-8') as f,
|
||||
):
|
||||
w, h = im.size
|
||||
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
|
||||
with open(output_text, 'w') as f:
|
||||
with output_text.open('w') as f:
|
||||
f.write('')
|
||||
|
||||
@staticmethod
|
||||
|
||||
@@ -72,11 +72,11 @@ class NoopOcrEngine(OcrEngine):
|
||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||
with (
|
||||
Image.open(input_file) as im,
|
||||
open(output_hocr, 'w', encoding='utf-8') as f,
|
||||
output_hocr.open('w', encoding='utf-8') as f,
|
||||
):
|
||||
w, h = im.size
|
||||
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
|
||||
with open(output_text, 'w') as f:
|
||||
with output_text.open('w') as f:
|
||||
f.write('')
|
||||
|
||||
@staticmethod
|
||||
|
||||
@@ -76,6 +76,9 @@ the copyright holder(s) and license(s) applicable to these resources.
|
||||
* - missing_docinfo.pdf
|
||||
- synthetic
|
||||
- PDF file with no /DocumentInfo section
|
||||
* - docinfo_latin1_key.pdf
|
||||
- synthetic
|
||||
- PDF whose /DocumentInfo dictionary has a /Name key with Latin-1 bytes (/Saks#e5r) that is not valid UTF-8
|
||||
* - overlay.pdf
|
||||
- synthetic
|
||||
- PDF file generated by PDFPen pro that triggered content stream parse errors
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
%PDF-1.3
|
||||
%¿÷¢þ
|
||||
1 0 obj
|
||||
<< /Pages 3 0 R /Type /Catalog >>
|
||||
endobj
|
||||
2 0 obj
|
||||
<< /Author (Geomatikk AS) /Beskrivelse () /Creator (OCRmyPDF 16.10.0 / EasyOCR-PDF 1.7.2) /CreatorVersion (6.36.0.918) /Dokumentidplanreg () /Enhetsnavn () /Hyperlink (1) /Opprinnelse () /Producer (pikepdf 9.5.2) /RegistrationDate (N/A) /Saksansvarlig#20enhet () /Saksbehandler () /Saksnr () /Saks#e5r () /Status () >>
|
||||
endobj
|
||||
3 0 obj
|
||||
<< /Count 1 /Kids [ 4 0 R ] /Type /Pages >>
|
||||
endobj
|
||||
4 0 obj
|
||||
<< /Contents 5 0 R /MediaBox [ 0 0 612 792 ] /Parent 3 0 R /Resources << >> /Type /Page >>
|
||||
endobj
|
||||
5 0 obj
|
||||
<< /Length 0 /Filter /FlateDecode >>
|
||||
stream
|
||||
|
||||
endstream
|
||||
endobj
|
||||
xref
|
||||
0 6
|
||||
0000000000 65535 f
|
||||
0000000015 00000 n
|
||||
0000000064 00000 n
|
||||
0000000398 00000 n
|
||||
0000000457 00000 n
|
||||
0000000563 00000 n
|
||||
trailer << /Info 2 0 R /Root 1 0 R /Size 6 /ID [<c5231b8cfab9c82526c0da7475add5da><c5231b8cfab9c82526c0da7475add5da>] >>
|
||||
startxref
|
||||
633
|
||||
%%EOF
|
||||
+39
-1
@@ -28,7 +28,6 @@ def test_language_parameter_mapped_to_languages():
|
||||
Regression test for GitHub issue #1640: the Python API ignored the language
|
||||
parameter, always defaulting to 'eng'.
|
||||
"""
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf.api import create_options, setup_plugin_infrastructure
|
||||
from ocrmypdf.cli import get_parser
|
||||
|
||||
@@ -80,6 +79,45 @@ def test_language_parameter_mapped_to_languages():
|
||||
assert options.languages == ['eng', 'spa']
|
||||
|
||||
|
||||
def test_jpeg_quality_parameter_reaches_options():
|
||||
"""The canonical 'jpeg_quality' API parameter must reach OcrOptions.
|
||||
|
||||
Regression test for GitHub issue #1723: --jpeg-quality was silently
|
||||
dropped by the CLI's namespace_to_options() because the OcrOptions field
|
||||
was named jpg_quality. create_options(), used by the Python API, has the
|
||||
same field-name matching logic and is affected the same way when passed
|
||||
the alias name.
|
||||
"""
|
||||
from ocrmypdf.api import create_options, setup_plugin_infrastructure
|
||||
from ocrmypdf.cli import get_parser
|
||||
|
||||
setup_plugin_infrastructure()
|
||||
parser = get_parser()
|
||||
|
||||
options = create_options(
|
||||
input_file='test.pdf', output_file='output.pdf', parser=parser, jpeg_quality=10
|
||||
)
|
||||
assert options.jpeg_quality == 10
|
||||
|
||||
|
||||
def test_jpg_quality_parameter_deprecated_alias():
|
||||
"""The old 'jpg_quality' API parameter still works but warns."""
|
||||
from ocrmypdf.api import create_options, setup_plugin_infrastructure
|
||||
from ocrmypdf.cli import get_parser
|
||||
|
||||
setup_plugin_infrastructure()
|
||||
parser = get_parser()
|
||||
|
||||
with pytest.warns(UserWarning, match='jpg_quality'):
|
||||
options = create_options(
|
||||
input_file='test.pdf',
|
||||
output_file='output.pdf',
|
||||
parser=parser,
|
||||
jpg_quality=42,
|
||||
)
|
||||
assert options.jpeg_quality == 42
|
||||
|
||||
|
||||
def test_stream_api(resources: Path):
|
||||
in_ = (resources / 'graph.pdf').open('rb')
|
||||
out = BytesIO()
|
||||
|
||||
@@ -219,7 +219,7 @@ class TestFpdf2MultiPageRenderer:
|
||||
for i in range(3):
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
text=f"Page{i+1}",
|
||||
text=f"Page{i + 1}",
|
||||
bbox=BoundingBox(left=100, top=100, right=200, bottom=130),
|
||||
)
|
||||
line = OcrElement(
|
||||
|
||||
+327
-5
@@ -17,7 +17,11 @@ from PIL import Image, UnidentifiedImageError
|
||||
|
||||
from ocrmypdf._exec import ghostscript
|
||||
from ocrmypdf._exec.ghostscript import DuplicateFilter, rasterize_pdf
|
||||
from ocrmypdf.builtin_plugins.ghostscript import _repair_gs106_jpeg_corruption
|
||||
from ocrmypdf.builtin_plugins.ghostscript import (
|
||||
PdfaImageCompression,
|
||||
_repair_gs106_jpeg_corruption,
|
||||
_resolve_auto_compression,
|
||||
)
|
||||
from ocrmypdf.exceptions import ColorConversionNeededError, ExitCode, InputFileError
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
@@ -137,6 +141,185 @@ def test_rasterize_low_dpi_one_axis(francais, outdir):
|
||||
assert im.info['dpi'] == forced_dpi
|
||||
|
||||
|
||||
def _capture_rasterize_args(resources, outdir, raster_device):
|
||||
"""Run rasterize_pdf with the gs subprocess mocked; return the gs argv."""
|
||||
out = outdir / 'out.png'
|
||||
captured = {}
|
||||
|
||||
def fake_run(args, **kwargs):
|
||||
captured['args'] = list(args)
|
||||
# Produce a valid PNG so rasterize_pdf's post-processing succeeds.
|
||||
Image.new('RGB', (2, 2)).save(out)
|
||||
return subprocess.CompletedProcess(args, returncode=0, stdout=b'', stderr=b'')
|
||||
|
||||
with patch('ocrmypdf._exec.ghostscript.run', side_effect=fake_run):
|
||||
rasterize_pdf(
|
||||
resources / 'francais.pdf',
|
||||
out,
|
||||
raster_device=raster_device,
|
||||
raster_dpi=Resolution(150.0, 150.0),
|
||||
)
|
||||
return captured['args']
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'raster_device',
|
||||
[
|
||||
GhostscriptRasterDevice.PNGGRAY,
|
||||
GhostscriptRasterDevice.PNG256,
|
||||
GhostscriptRasterDevice.PNG16M,
|
||||
],
|
||||
)
|
||||
def test_rasterize_antialiases_contone_devices(resources, outdir, raster_device):
|
||||
"""Contone raster devices receive anti-aliasing flags to aid OCR.
|
||||
|
||||
Ghostscript 10.x renders aliased glyphs that OCR misreads as extra word
|
||||
breaks; -dTextAlphaBits/-dGraphicsAlphaBits markedly improve accuracy,
|
||||
especially for small fonts at moderate DPI (see issue #1439).
|
||||
"""
|
||||
args = _capture_rasterize_args(resources, outdir, raster_device)
|
||||
assert '-dTextAlphaBits=4' in args
|
||||
assert '-dGraphicsAlphaBits=4' in args
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'raster_device',
|
||||
[GhostscriptRasterDevice.PNGMONO, GhostscriptRasterDevice.PNGMONOD],
|
||||
)
|
||||
def test_rasterize_no_antialias_on_mono_devices(resources, outdir, raster_device):
|
||||
"""1-bit mono devices must not receive alpha-bit flags.
|
||||
|
||||
Older Ghostscript versions reject -dTextAlphaBits on 1-bit devices, and
|
||||
pngmonod performs its own anti-aliased downscaling.
|
||||
"""
|
||||
args = _capture_rasterize_args(resources, outdir, raster_device)
|
||||
assert not any(a.startswith('-dTextAlphaBits') for a in args)
|
||||
assert not any(a.startswith('-dGraphicsAlphaBits') for a in args)
|
||||
|
||||
|
||||
def test_generate_pdfa_default_jpeg_quality(outdir):
|
||||
"""When jpeg_quality is None, Ghostscript receives -dJPEGQ=95 (default)."""
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='auto',
|
||||
color_conversion_strategy='LeaveColorUnchanged',
|
||||
)
|
||||
|
||||
args = run_mock.call_args.args[0]
|
||||
assert '-dJPEGQ=95' in args
|
||||
# No downsample switches when jpeg_maxdpi is not set
|
||||
assert not any(a.startswith('-dDownsampleColorImages') for a in args)
|
||||
assert not any(a.startswith('-dColorImageResolution') for a in args)
|
||||
|
||||
|
||||
def test_generate_pdfa_uses_user_jpeg_quality(outdir):
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='jpeg',
|
||||
color_conversion_strategy='RGB',
|
||||
jpeg_quality=72,
|
||||
)
|
||||
|
||||
args = run_mock.call_args.args[0]
|
||||
assert '-dJPEGQ=72' in args
|
||||
assert '-dJPEGQ=95' not in args
|
||||
|
||||
|
||||
def test_generate_pdfa_jpeg_quality_zero_is_max_compression(outdir):
|
||||
"""Explicit jpeg_quality=0 must reach Ghostscript as -dJPEGQ=0.
|
||||
|
||||
Ghostscript accepts 0 as a valid quality value (maximum compression);
|
||||
it must not be silently replaced by the default 95.
|
||||
"""
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='jpeg',
|
||||
color_conversion_strategy='RGB',
|
||||
jpeg_quality=0,
|
||||
)
|
||||
|
||||
args = run_mock.call_args.args[0]
|
||||
assert '-dJPEGQ=0' in args
|
||||
assert '-dJPEGQ=95' not in args
|
||||
|
||||
|
||||
def test_generate_pdfa_honors_jpeg_maxdpi(outdir):
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='auto',
|
||||
color_conversion_strategy='LeaveColorUnchanged',
|
||||
jpeg_maxdpi=300,
|
||||
)
|
||||
|
||||
args = run_mock.call_args.args[0]
|
||||
assert '-dJPEGQ=95' in args
|
||||
assert '-dDownsampleColorImages=true' in args
|
||||
assert '-dColorImageDownsampleThreshold=1.0' in args
|
||||
assert '-dDownsampleGrayImages=true' in args
|
||||
assert '-dGrayImageDownsampleThreshold=1.0' in args
|
||||
assert '-dDownsampleMonoImages=true' in args
|
||||
assert '-dMonoImageDownsampleThreshold=1.0' in args
|
||||
assert '-dColorImageResolution=300' in args
|
||||
assert '-dGrayImageResolution=300' in args
|
||||
assert '-dMonoImageResolution=300' in args
|
||||
|
||||
|
||||
def test_ghostscript_jpeg_options_via_cli(resources, outpdf):
|
||||
"""End-to-end: CLI flags reach the ghostscript plugin namespace."""
|
||||
with patch(
|
||||
'ocrmypdf._exec.ghostscript.generate_pdfa',
|
||||
wraps=ghostscript.generate_pdfa,
|
||||
) as gen_mock:
|
||||
run_ocrmypdf_api(
|
||||
resources / 'francais.pdf',
|
||||
outpdf,
|
||||
'--output-type',
|
||||
'pdfa',
|
||||
'--ghostscript-jpeg-quality',
|
||||
'60',
|
||||
'--ghostscript-jpeg-maxdpi',
|
||||
'150',
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
assert gen_mock.called
|
||||
call_kwargs = gen_mock.call_args.kwargs
|
||||
assert call_kwargs['jpeg_quality'] == 60
|
||||
assert call_kwargs['jpeg_maxdpi'] == 150
|
||||
|
||||
|
||||
def test_gs_render_failure(resources, outpdf, caplog):
|
||||
exitcode = run_ocrmypdf_api(
|
||||
resources / 'blank.pdf',
|
||||
@@ -176,9 +359,9 @@ def test_ghostscript_pdfa_failure(resources, outpdf, caplog):
|
||||
'--plugin',
|
||||
'tests/plugins/gs_pdfa_failure.py',
|
||||
)
|
||||
assert (
|
||||
exitcode == ExitCode.pdfa_conversion_failed
|
||||
), "Unexpected return when PDF/A fails"
|
||||
assert exitcode == ExitCode.pdfa_conversion_failed, (
|
||||
"Unexpected return when PDF/A fails"
|
||||
)
|
||||
|
||||
|
||||
def test_ghostscript_feature_elision(resources, outpdf):
|
||||
@@ -204,6 +387,88 @@ def test_ghostscript_mandatory_color_conversion(resources, outpdf):
|
||||
)
|
||||
|
||||
|
||||
def _run_generate_pdfa_with_devicen_warning(outdir, color_conversion_strategy):
|
||||
"""Invoke generate_pdfa with Ghostscript mocked to emit the DeviceN warning.
|
||||
|
||||
Ghostscript emits this warning when it writes a DeviceN colorspace with an
|
||||
inappropriate alternate, i.e. when it could not normalize the colorspace for
|
||||
PDF/A. The output is then liable to render blank in viewers such as Adobe
|
||||
Reader (see issue #1187), regardless of which conversion strategy was
|
||||
requested.
|
||||
"""
|
||||
(outdir / 'input.pdf').write_bytes(b'%PDF-1.5\n%fake\n')
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'],
|
||||
returncode=0,
|
||||
stdout='',
|
||||
stderr='Attempting to write a DeviceN space with an inappropriate '
|
||||
'alternate, reverting to the alternate color space.',
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='auto',
|
||||
color_conversion_strategy=color_conversion_strategy,
|
||||
)
|
||||
|
||||
|
||||
def test_devicen_warning_default_strategy_raises_with_guidance(outdir):
|
||||
"""Default (no conversion): raise and tell the user to pick a strategy."""
|
||||
with pytest.raises(ColorConversionNeededError) as exc_info:
|
||||
_run_generate_pdfa_with_devicen_warning(outdir, 'LeaveColorUnchanged')
|
||||
message = str(exc_info.value)
|
||||
assert '--color-conversion-strategy' in message
|
||||
assert 'RGB' in message
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'strategy',
|
||||
[
|
||||
# A strategy that genuinely cannot fix the colorspace; confirmed in #1187.
|
||||
'UseDeviceIndependentColor',
|
||||
# A normally-effective strategy that nonetheless failed on this input:
|
||||
# if Ghostscript still warns, the output is still broken and we must not
|
||||
# silently pass it through (the behaviour PR #1692 would have introduced).
|
||||
'RGB',
|
||||
],
|
||||
)
|
||||
def test_devicen_warning_persists_despite_strategy_still_raises(outdir, strategy):
|
||||
"""If the warning survives the requested conversion, the output is broken.
|
||||
|
||||
We must still raise rather than silently emit a PDF/A that may render blank.
|
||||
The guidance should acknowledge that the chosen strategy did not work and
|
||||
point at strategies that do (or --output-type pdf).
|
||||
"""
|
||||
with pytest.raises(ColorConversionNeededError) as exc_info:
|
||||
_run_generate_pdfa_with_devicen_warning(outdir, strategy)
|
||||
message = str(exc_info.value)
|
||||
assert strategy in message
|
||||
assert '--output-type pdf' in message
|
||||
|
||||
|
||||
def test_no_devicen_warning_does_not_raise(outdir):
|
||||
"""When Ghostscript does not warn, conversion succeeded; never raise."""
|
||||
(outdir / 'input.pdf').write_bytes(b'%PDF-1.5\n%fake\n')
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
# Must not raise for any strategy when there is no DeviceN warning.
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='auto',
|
||||
color_conversion_strategy='RGB',
|
||||
)
|
||||
|
||||
|
||||
def test_rasterize_pdf_errors(resources, no_outpdf, caplog):
|
||||
with patch('ocrmypdf._exec.ghostscript.run') as mock:
|
||||
# ghostscript can produce empty files with return code 0
|
||||
@@ -439,7 +704,9 @@ class TestGs106JpegCorruptionRepair:
|
||||
repaired_bytes_list.append(obj.read_raw_bytes())
|
||||
|
||||
assert len(repaired_bytes_list) == len(original_bytes_list)
|
||||
for orig, repaired_bytes in zip(original_bytes_list, repaired_bytes_list, strict=False):
|
||||
for orig, repaired_bytes in zip(
|
||||
original_bytes_list, repaired_bytes_list, strict=False
|
||||
):
|
||||
assert orig == repaired_bytes, "Repaired bytes should match original"
|
||||
|
||||
# Check that error/warning was logged
|
||||
@@ -468,3 +735,58 @@ class TestGs106JpegCorruptionRepair:
|
||||
repaired = _repair_gs106_jpeg_corruption(source_path, damaged_path)
|
||||
assert repaired is False, "Should not repair truncation > 15 bytes"
|
||||
assert "JPEG corruption detected" not in caplog.text
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
('compression', 'optimize', 'expected'),
|
||||
[
|
||||
# auto coerces to lossless only at -O0; -O1 is a historical exception
|
||||
# that keeps Ghostscript's (possibly lossy) heuristic, as do -O2/-O3
|
||||
(PdfaImageCompression.AUTO, 0, PdfaImageCompression.LOSSLESS),
|
||||
(PdfaImageCompression.AUTO, 1, PdfaImageCompression.AUTO),
|
||||
(PdfaImageCompression.AUTO, 2, PdfaImageCompression.AUTO),
|
||||
(PdfaImageCompression.AUTO, 3, PdfaImageCompression.AUTO),
|
||||
# explicit choices are always respected, regardless of optimize level
|
||||
(PdfaImageCompression.JPEG, 0, PdfaImageCompression.JPEG),
|
||||
(PdfaImageCompression.JPEG, 1, PdfaImageCompression.JPEG),
|
||||
(PdfaImageCompression.LOSSLESS, 1, PdfaImageCompression.LOSSLESS),
|
||||
(PdfaImageCompression.LOSSLESS, 3, PdfaImageCompression.LOSSLESS),
|
||||
],
|
||||
)
|
||||
def test_resolve_auto_compression(compression, optimize, expected):
|
||||
assert _resolve_auto_compression(compression, optimize) == expected
|
||||
|
||||
|
||||
def _capture_generate_pdfa_args(tmp_path, compression):
|
||||
"""Run generate_pdfa with a mocked Ghostscript and return the argv it built."""
|
||||
from subprocess import CompletedProcess
|
||||
|
||||
captured = {}
|
||||
|
||||
def fake_run(args, **kwargs):
|
||||
captured['args'] = list(args)
|
||||
return CompletedProcess(args, 0, None, stderr='')
|
||||
|
||||
out = tmp_path / 'out.pdf'
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', side_effect=fake_run):
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=['dummy.pdf'],
|
||||
output_file=out,
|
||||
compression=compression,
|
||||
color_conversion_strategy='RGB',
|
||||
)
|
||||
return captured['args']
|
||||
|
||||
|
||||
def test_lossless_compression_passes_through_jpegs(tmp_path):
|
||||
# Re-encoding an existing JPEG losslessly only bloats it (the lossy data is
|
||||
# already baked in), so lossless mode must let Ghostscript pass JPEGs through
|
||||
# untouched while still keeping lossless images lossless.
|
||||
args = _capture_generate_pdfa_args(tmp_path, 'lossless')
|
||||
assert '-dPassThroughJPEGImages=true' in args
|
||||
assert '-dColorImageFilter=/FlateEncode' in args
|
||||
|
||||
|
||||
def test_jpeg_compression_does_not_force_passthrough(tmp_path):
|
||||
args = _capture_generate_pdfa_args(tmp_path, 'jpeg')
|
||||
assert '-dPassThroughJPEGImages=true' not in args
|
||||
|
||||
+6
-6
@@ -78,9 +78,9 @@ def test_redo_ocr_with_offset_mediabox(resources, outdir):
|
||||
mediabox = list(page.MediaBox)
|
||||
|
||||
# MediaBox origin should be preserved
|
||||
assert (
|
||||
float(mediabox[1]) == y_offset
|
||||
), f"MediaBox Y origin should be preserved at {y_offset}, got {mediabox[1]}"
|
||||
assert float(mediabox[1]) == y_offset, (
|
||||
f"MediaBox Y origin should be preserved at {y_offset}, got {mediabox[1]}"
|
||||
)
|
||||
|
||||
# The content stream should include a CTM with the Y origin translation.
|
||||
# Without the fix, the CTM was omitted for rotation==0, causing a shift.
|
||||
@@ -153,7 +153,7 @@ def test_strip_invisble_text():
|
||||
nr_visible_pre = count('visible', page)
|
||||
ocrmypdf._graft.strip_invisible_text(pdf, page)
|
||||
nr_visible_post = count('visible', page)
|
||||
assert (
|
||||
nr_visible_pre == nr_visible_post
|
||||
), 'Number of visible text elements did not change'
|
||||
assert nr_visible_pre == nr_visible_post, (
|
||||
'Number of visible text elements did not change'
|
||||
)
|
||||
assert count('invisible', page) == 0, 'No invisible elems left'
|
||||
|
||||
@@ -121,8 +121,8 @@ def test_shim_paths(tmp_path):
|
||||
results = result_str.split(os.pathsep)
|
||||
assert results[0] == str(syspath), results
|
||||
assert results[-3].endswith('tesseract-ocr'), results
|
||||
assert results[-2].endswith(os.path.join('gs9.52.3', 'bin')), results
|
||||
assert results[-1].endswith(os.path.join('gs', '9.51', 'bin')), results
|
||||
assert results[-2].endswith(str(Path('gs9.52.3', 'bin'))), results
|
||||
assert results[-1].endswith(str(Path('gs', '9.51', 'bin'))), results
|
||||
|
||||
|
||||
def test_resolution():
|
||||
|
||||
@@ -27,7 +27,7 @@ from .conftest import check_ocrmypdf
|
||||
|
||||
def text_from_pdf(filename):
|
||||
output_string = StringIO()
|
||||
with open(filename, 'rb') as in_file:
|
||||
with filename.open('rb') as in_file:
|
||||
parser = PDFParser(in_file)
|
||||
doc = PDFDocument(parser)
|
||||
rsrcmgr = PDFResourceManager()
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
"""Test JSON serialization of OcrOptions for multiprocessing compatibility."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import multiprocessing
|
||||
|
||||
+32
-11
@@ -100,14 +100,14 @@ def test_redo_ocr(resources, outpdf):
|
||||
out = check_ocrmypdf(in_, out, '--redo-ocr')
|
||||
after = PdfInfo(out, detailed_analysis=True)
|
||||
assert before[0].has_text and after[0].has_text
|
||||
assert (
|
||||
before[0].get_textareas() != after[0].get_textareas()
|
||||
), "Expected text to be different after re-OCR"
|
||||
assert before[0].get_textareas() != after[0].get_textareas(), (
|
||||
"Expected text to be different after re-OCR"
|
||||
)
|
||||
|
||||
|
||||
def test_argsfile(resources, outdir):
|
||||
path_argsfile = outdir / 'test_argsfile.txt'
|
||||
with open(str(path_argsfile), 'w') as argsfile:
|
||||
with path_argsfile.open('w') as argsfile:
|
||||
print(
|
||||
'--title',
|
||||
'ArgsFile Test',
|
||||
@@ -646,7 +646,7 @@ def test_compression_preserved(ocrmypdf_exec, resources, image, outpdf):
|
||||
|
||||
im = Image.open(input_file)
|
||||
# Runs: ocrmypdf - output.pdf < testfile
|
||||
with open(input_file, 'rb') as input_stream:
|
||||
with Path(input_file).open('rb') as input_stream:
|
||||
p_args = ocrmypdf_exec + [
|
||||
'--optimize',
|
||||
'0',
|
||||
@@ -704,7 +704,7 @@ def test_compression_changed(ocrmypdf_exec, resources, image, compression, outpd
|
||||
im = Image.open(input_file)
|
||||
|
||||
# Runs: ocrmypdf - output.pdf < testfile
|
||||
with open(input_file, 'rb') as input_stream:
|
||||
with Path(input_file).open('rb') as input_stream:
|
||||
p_args = ocrmypdf_exec + [
|
||||
'--image-dpi',
|
||||
'150',
|
||||
@@ -763,14 +763,14 @@ def test_sidecar_pagecount(resources, outpdf):
|
||||
pdfinfo = PdfInfo(resources / '3small.pdf')
|
||||
num_pages = len(pdfinfo)
|
||||
|
||||
with open(sidecar, encoding='utf-8') as f:
|
||||
with sidecar.open(encoding='utf-8') as f:
|
||||
ocr_text = f.read()
|
||||
|
||||
# There should a formfeed between each pair of pages, so the count of
|
||||
# formfeeds is the page count less one
|
||||
assert (
|
||||
ocr_text.count('\f') == num_pages - 1
|
||||
), "Sidecar page count does not match PDF page count"
|
||||
assert ocr_text.count('\f') == num_pages - 1, (
|
||||
"Sidecar page count does not match PDF page count"
|
||||
)
|
||||
|
||||
|
||||
def test_sidecar_nonempty(resources, outpdf):
|
||||
@@ -784,7 +784,7 @@ def test_sidecar_nonempty(resources, outpdf):
|
||||
'tests/plugins/tesseract_cache.py',
|
||||
)
|
||||
|
||||
with open(sidecar, encoding='utf-8') as f:
|
||||
with sidecar.open(encoding='utf-8') as f:
|
||||
ocr_text = f.read()
|
||||
assert 'the' in ocr_text
|
||||
|
||||
@@ -889,6 +889,27 @@ def test_version_check():
|
||||
get_version('echo')
|
||||
|
||||
|
||||
def test_get_version_skips_leading_warning_lines(monkeypatch):
|
||||
"""VeraPDF 1.30.0 prints JVM warnings before its version line."""
|
||||
from subprocess import CompletedProcess
|
||||
|
||||
import ocrmypdf.subprocess as sp
|
||||
|
||||
output = (
|
||||
"WARNING: Final field flavour has been mutated reflectively\n"
|
||||
"WARNING: Use --enable-final-field-mutation=ALL-UNNAMED to avoid this\n"
|
||||
"veraPDF 1.30.0\n"
|
||||
"Built: Wed Jun 03 13:29:00 PDT 2026\n"
|
||||
)
|
||||
|
||||
def fake_run(args, **kwargs):
|
||||
return CompletedProcess(args, 0, stdout=output, stderr="")
|
||||
|
||||
monkeypatch.setattr(sp, 'run', fake_run)
|
||||
version = get_version('verapdf', regex=r'veraPDF (\d+(\.\d+)*)')
|
||||
assert version == '1.30.0'
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'threshold, optimize, output_type, expected',
|
||||
[
|
||||
|
||||
+29
-2
@@ -6,13 +6,14 @@ from __future__ import annotations
|
||||
import datetime as dt
|
||||
import warnings
|
||||
from shutil import copyfile
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from pikepdf.models.metadata import decode_pdf_date
|
||||
|
||||
from ocrmypdf._jobcontext import PdfContext
|
||||
from ocrmypdf._metadata import metadata_fixup
|
||||
from ocrmypdf._metadata import metadata_fixup, repair_docinfo_nuls
|
||||
from ocrmypdf._pipeline import convert_to_pdfa
|
||||
from ocrmypdf.api import setup_plugin_infrastructure
|
||||
from ocrmypdf.cli import get_options_and_plugins
|
||||
@@ -43,6 +44,32 @@ def test_preserve_docinfo(output_type, resources, outpdf):
|
||||
assert pdfa_info['output'] == output_type
|
||||
|
||||
|
||||
def test_repair_docinfo_nuls_undecodable_key(caplog):
|
||||
"""A DocumentInfo key with bytes that don't decode must not crash.
|
||||
|
||||
Some PDFs use a /Name dictionary key in DocumentInfo whose bytes are not
|
||||
valid PDFDocEncoding/UTF-8 (e.g. Latin-1 ``/Saks#e5r``). Older pikepdf
|
||||
raised UnicodeDecodeError while iterating such a dictionary. The repair
|
||||
must log and continue rather than propagate the exception. See #1540.
|
||||
"""
|
||||
pdf = MagicMock()
|
||||
pdf.docinfo.items.side_effect = UnicodeDecodeError(
|
||||
'utf-8', b'Saks\xe5r', 4, 5, 'invalid continuation byte'
|
||||
)
|
||||
# Make isinstance(pdf.docinfo, Dictionary) succeed so we reach the loop.
|
||||
with patch('ocrmypdf._metadata.Dictionary', MagicMock):
|
||||
result = repair_docinfo_nuls(pdf)
|
||||
assert result is False
|
||||
assert 'malformed DocumentInfo' in caplog.text
|
||||
|
||||
|
||||
def test_repair_docinfo_nuls_undecodable_key_real_file(resources):
|
||||
"""Opening a real file with a Latin-1 DocumentInfo key must not crash."""
|
||||
with pikepdf.open(resources / 'docinfo_latin1_key.pdf') as pdf:
|
||||
# Should return without raising regardless of pikepdf's decode behavior.
|
||||
repair_docinfo_nuls(pdf)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("output_type", ['pdfa', 'pdf'])
|
||||
def test_override_metadata(output_type, resources, outpdf, caplog):
|
||||
input_file = resources / 'c02-22.pdf'
|
||||
@@ -110,7 +137,7 @@ def test_unset_metadata(output_type, field, resources, outpdf, caplog):
|
||||
# isn't contained anywhere in the output pdf. We'll also check to ensure
|
||||
# it's in the input pdf and that any values not unset are still in the
|
||||
# output pdf.
|
||||
with open(input_file, 'rb') as before, open(outpdf, 'rb') as after:
|
||||
with input_file.open('rb') as before, outpdf.open('rb') as after:
|
||||
before_data = before.read()
|
||||
after_data = after.read()
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user