Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8de7b05fb9 | ||
|
|
72ce05768e | ||
|
|
3dc68778fc | ||
|
|
1aec92b919 | ||
|
|
43d3448709 | ||
|
|
7512b1042a | ||
|
|
efe83e8c54 | ||
|
|
a13d27bfb5 | ||
|
|
ea7ad7d683 | ||
|
|
8b20bb3c5b | ||
|
|
320876a6d1 | ||
|
|
dfbb4c9275 | ||
|
|
d4f5c2d160 | ||
|
|
263d6034be | ||
|
|
de403f6d5e | ||
|
|
86b6f2c907 | ||
|
|
e6fab76918 | ||
|
|
334918d0f7 | ||
|
|
d6329489ce | ||
|
|
e6d240ee93 | ||
|
|
ff45e54c07 | ||
|
|
e0ee0882ef | ||
|
|
3d17419a6c | ||
|
|
476ec12383 | ||
|
|
e99177ada7 | ||
|
|
e95ec9c497 | ||
|
|
82f30bfbec | ||
|
|
d1437e6bbc | ||
|
|
c669d30642 | ||
|
|
3613b30ca8 | ||
|
|
0d4c3bcdcf | ||
|
|
8a8d515933 | ||
|
|
11de13ecfe | ||
|
|
58642d8411 | ||
|
|
7e42d3c771 | ||
|
|
5cb5d7a682 | ||
|
|
37e71dece6 | ||
|
|
df84945773 | ||
|
|
b5a6a9f9f1 | ||
|
|
ed36aefe48 | ||
|
|
32013f4294 | ||
|
|
8f2bcc2c64 | ||
|
|
015b53ae30 | ||
|
|
164cf2dc8a | ||
|
|
98d6d02704 | ||
|
|
5efb98931d | ||
|
|
2f4e47213f | ||
|
|
94c8123bd7 | ||
|
|
0db130e1c3 | ||
|
|
91b6a818f5 | ||
|
|
6bc9499e68 | ||
|
|
09f2d6c386 | ||
|
|
87f918f58c | ||
|
|
80e77fb021 | ||
|
|
fa9c5b3fae | ||
|
|
3d17a60a54 | ||
|
|
c33f073d4f | ||
|
|
5d7b5742e4 | ||
|
|
c391b2b7d0 | ||
|
|
0250929150 | ||
|
|
9748208e68 | ||
|
|
e4b0c04be4 | ||
|
|
efb83ad64f | ||
|
|
08e40f96e8 | ||
|
|
3f6feb1dcc | ||
|
|
ab6553f4ff | ||
|
|
cedca9fa1f | ||
|
|
3f40118022 | ||
|
|
b18b1da6d0 | ||
|
|
14fb9f56e8 | ||
|
|
8709cf506b | ||
|
|
9a92eb40df | ||
|
|
0a59c210f9 | ||
|
|
0b370fdd15 | ||
|
|
1c16dd26f7 | ||
|
|
c355d927ba | ||
|
|
c993857752 | ||
|
|
84f5fe9ee0 |
+29
-7
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM ubuntu:25.04 AS base
|
||||
FROM ubuntu:26.04 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -40,7 +40,7 @@ RUN \
|
||||
WORKDIR /app
|
||||
|
||||
# Copy uv from ghcr
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9.8 /uv /uvx /bin/
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.21 /uv /uvx /bin/
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
@@ -60,10 +60,8 @@ RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
|
||||
FROM base
|
||||
|
||||
RUN apt-get update && apt-get install -y software-properties-common
|
||||
|
||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr5
|
||||
|
||||
# Tesseract 5 ships in the Ubuntu archive as of 24.04, so no third-party PPA is
|
||||
# needed. (Previously this used ppa:alex-p/tesseract-ocr5.)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ghostscript \
|
||||
fonts-droid-fallback \
|
||||
@@ -81,6 +79,18 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
unpaper \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Create a non-root user to run the application (defense in depth). The build
|
||||
# stages above need root to install packages, but the entrypoint should not.
|
||||
# A fixed uid/gid of 1000 keeps `--user`/`--userns keep-id` mappings predictable
|
||||
# and matches the --chown below. See docs/docker.md for the volume/permissions
|
||||
# implications under rootless vs rootful Docker.
|
||||
# The Ubuntu base ships a default "ubuntu" user at uid/gid 1000; remove it so
|
||||
# "app" can claim that uid for parity with the Alpine image.
|
||||
RUN userdel -r ubuntu 2>/dev/null; groupdel ubuntu 2>/dev/null; \
|
||||
groupadd -g 1000 app \
|
||||
&& useradd -u 1000 -g app -m -d /home/app app
|
||||
ENV HOME=/home/app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
||||
@@ -90,9 +100,21 @@ COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
ln -s /app/misc/watcher.py /app/watcher.py && \
|
||||
chown app:app /app
|
||||
|
||||
# Default working directory for bind-mounted data, so relative input/output
|
||||
# paths work without passing --workdir (e.g. `-v "$PWD:/data" in.pdf out.pdf`).
|
||||
# The webservice/watcher are run by absolute path (/app/*.py), unaffected by this.
|
||||
RUN mkdir -p /data && chown app:app /data
|
||||
WORKDIR /data
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
# Drop privileges: run the entrypoint (ocrmypdf, or the webservice/watcher when
|
||||
# overridden) as the unprivileged app user. Override with `--user root` if you
|
||||
# need root inside a running container (e.g. to apt install extra packages).
|
||||
USER app
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM alpine:3.23 AS base
|
||||
FROM alpine:3.24 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -22,7 +22,7 @@ RUN apk add --no-cache \
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9.8 /uv /uvx /bin/
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.21 /uv /uvx /bin/
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
@@ -62,14 +62,35 @@ RUN apk add --no-cache \
|
||||
unpaper \
|
||||
&& rm -rf /var/cache/apk/*
|
||||
|
||||
# Create a non-root user to run the application (defense in depth). The build
|
||||
# stages above need root to install packages, but the entrypoint should not.
|
||||
# A fixed uid/gid of 1000 keeps `--user`/`--userns keep-id` mappings predictable
|
||||
# and matches the --chown below. See docs/docker.md for the volume/permissions
|
||||
# implications under rootless vs rootful Docker.
|
||||
RUN addgroup -g 1000 app \
|
||||
&& adduser -u 1000 -G app -D -h /home/app app
|
||||
ENV HOME=/home/app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
ln -s /app/misc/watcher.py /app/watcher.py && \
|
||||
chown app:app /app
|
||||
|
||||
# Default working directory for bind-mounted data, so relative input/output
|
||||
# paths work without passing --workdir (e.g. `-v "$PWD:/data" in.pdf out.pdf`).
|
||||
# The webservice/watcher are run by absolute path (/app/*.py), unaffected by this.
|
||||
RUN mkdir -p /data && chown app:app /data
|
||||
WORKDIR /data
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
# Drop privileges: run the entrypoint (ocrmypdf, or the webservice/watcher when
|
||||
# overridden) as the unprivileged app user. Override with `--user root` if you
|
||||
# need root inside a running container (e.g. to apk add extra packages).
|
||||
USER app
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
|
||||
+10
-10
@@ -31,7 +31,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -87,7 +87,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v6
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -107,7 +107,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install Homebrew deps
|
||||
continue-on-error: true
|
||||
@@ -149,7 +149,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v6
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -169,7 +169,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -196,7 +196,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v6
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -207,7 +207,7 @@ jobs:
|
||||
name: Build sdist and wheels
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -233,7 +233,7 @@ jobs:
|
||||
permissions:
|
||||
contents: write
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/download-artifact@v8
|
||||
with:
|
||||
@@ -281,7 +281,7 @@ jobs:
|
||||
- name: Set image name
|
||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
||||
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4
|
||||
@@ -327,7 +327,7 @@ jobs:
|
||||
- name: Set image name
|
||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf-alpine" >> $GITHUB_ENV
|
||||
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4
|
||||
|
||||
@@ -19,7 +19,7 @@ jobs:
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Download artifacts from draft release
|
||||
env:
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
|
||||
# PyPI doesn't support sigstore publishing, so generate after publishing to PyPI
|
||||
- name: Sign the dists with Sigstore
|
||||
uses: sigstore/gh-action-sigstore-python@v3.3.0
|
||||
uses: sigstore/gh-action-sigstore-python@v3.4.0
|
||||
with:
|
||||
inputs: |
|
||||
./dist/*.tar.gz
|
||||
|
||||
+90
-76
@@ -1,8 +1,3 @@
|
||||
# OCRmyPDF 中文说明
|
||||
|
||||
我将为您创建一个中文版的 README.md 文件。以下是基于原始英文版本的中文翻译:
|
||||
|
||||
```markdown:/Users/huapai/PycharmProjects/OCRmyPDF/README.md
|
||||
<!-- SPDX-FileCopyrightText: 2014 Julien Pfefferkorn -->
|
||||
<!-- SPDX-FileCopyrightText: 2015 James R. Barlow -->
|
||||
<!-- SPDX-License-Identifier: CC-BY-SA-4.0 -->
|
||||
@@ -16,98 +11,109 @@
|
||||
[docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD"
|
||||
[pyversions]: https://img.shields.io/pypi/pyversions/ocrmypdf "支持的 Python 版本"
|
||||
|
||||
OCRmyPDF 为扫描的 PDF 文件添加 OCR 文本层,使其可以被搜索或复制粘贴。
|
||||
OCRmyPDF 会为扫描版 PDF 文件添加 OCR 文本层,使其可以搜索或复制粘贴。
|
||||
|
||||
```bash
|
||||
ocrmypdf # 这是一个可脚本化的命令行程序
|
||||
-l eng+fra # 支持多种语言
|
||||
--rotate-pages # 可以修正旋转错误的页面
|
||||
--deskew # 可以校正倾斜的 PDF!
|
||||
--title "My PDF" # 可以更改输出元数据
|
||||
--jobs 4 # 默认使用多核心处理
|
||||
--output-type pdfa # 默认生成 PDF/A 格式
|
||||
ocrmypdf # 它是一个可脚本化的命令行程序
|
||||
-l eng+fra # 它支持多种语言
|
||||
--rotate-pages # 它可以修正旋转方向错误的页面
|
||||
--deskew # 它可以校正歪斜的 PDF!
|
||||
--title "My PDF" # 它可以更改输出元数据
|
||||
--jobs 4 # 它默认使用多个 CPU 核心
|
||||
--output-type pdfa # 它默认生成 PDF/A
|
||||
input_scanned.pdf # 接受 PDF 输入(或图像)
|
||||
output_searchable.pdf # 生成经过验证的 PDF 输出
|
||||
```
|
||||
|
||||
[查看发布说明了解最新变更的详情](https://ocrmypdf.readthedocs.io/en/latest/release_notes.html)。
|
||||
[查看发布说明,了解最新变更详情](https://ocrmypdf.readthedocs.io/en/latest/release_notes.html)。
|
||||
|
||||
## 主要特点
|
||||
## 主要功能
|
||||
|
||||
- 从普通 PDF 生成可搜索的 [PDF/A](https://en.wikipedia.org/?title=PDF/A) 文件
|
||||
- 准确地将 OCR 文本放置在图像下方,便于复制/粘贴
|
||||
- 将 OCR 文本准确放置在图像下方,便于复制/粘贴
|
||||
- 保持原始嵌入图像的精确分辨率
|
||||
- 在可能的情况下,以"无损"操作方式插入 OCR 信息,不破坏任何其他内容
|
||||
- 在可能时,以“无损”操作插入 OCR 信息,不干扰任何其他内容
|
||||
- 优化 PDF 图像,通常生成比输入文件更小的文件
|
||||
- 如果需要,在执行 OCR 前对图像进行校正和/或清理
|
||||
- 按需在执行 OCR 前校正和/或清理图像
|
||||
- 验证输入和输出文件
|
||||
- 在所有可用的 CPU 核心上分配工作
|
||||
- 在所有可用 CPU 核心间分配工作
|
||||
- 使用 [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) 引擎识别超过 [100 种语言](https://github.com/tesseract-ocr/tessdata)
|
||||
- 保护您的私人数据安全
|
||||
- 适当扩展以处理包含数千页的文件
|
||||
- 在数百万 PDF 上经过实战测试
|
||||
- 保护你的私有数据。
|
||||
- 可以妥善扩展,处理包含数千页的文件。
|
||||
- 已在数百万份 PDF 上经过实战检验。
|
||||
|
||||
<img src="misc/screencast/demo.svg" alt="终端会话中的 OCRmyPDF 演示">
|
||||
<img src="misc/screencast/demo.svg" alt="OCRmyPDF 在终端会话中的演示">
|
||||
|
||||
详情请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/)。
|
||||
|
||||
## 开发动机
|
||||
## 动机
|
||||
|
||||
我在网上搜索免费的命令行工具来对 PDF 文件进行 OCR:我找到了很多,但没有一个真正令人满意:
|
||||
我曾在网上寻找一款免费的命令行工具来对 PDF 文件执行 OCR:我找到了很多,但没有一个真正令人满意:
|
||||
|
||||
- 要么它们生成的 PDF 文件中文本位置错误(使复制/粘贴变得不可能)
|
||||
- 要么它们不处理重音和多语言字符
|
||||
- 要么它们改变了嵌入图像的分辨率
|
||||
- 要么它们生成了体积巨大的 PDF 文件
|
||||
- 要么它们在尝试 OCR 时崩溃
|
||||
- 要么它们不生成有效的 PDF 文件
|
||||
- 最重要的是,它们都不生成 PDF/A 文件(专为长期存储设计的格式)
|
||||
- 要么生成的 PDF 文件中文本位于图像下方的错误位置(导致无法复制/粘贴)
|
||||
- 要么无法处理重音字符和多语言字符
|
||||
- 要么会改变嵌入图像的分辨率
|
||||
- 要么生成的 PDF 文件大得离谱
|
||||
- 要么在尝试 OCR 时崩溃
|
||||
- 要么无法生成有效的 PDF 文件
|
||||
- 除此之外,它们都不能生成 PDF/A 文件(专为长期存储设计的格式)
|
||||
|
||||
...所以我决定开发自己的工具。
|
||||
……所以我决定开发自己的工具。
|
||||
|
||||
## 安装
|
||||
|
||||
支持 Linux、Windows、macOS 和 FreeBSD。Docker 镜像也可用,同时支持 x64 和 ARM。
|
||||
支持 Linux、Windows、macOS 和 FreeBSD。也提供 Docker 镜像,同时支持 x64 和 ARM。
|
||||
|
||||
| 操作系统 | 安装命令 |
|
||||
| --------------------------- | ----------------------------- |
|
||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||
| 操作系统 | 安装命令 |
|
||||
| ----------------------------- | ------------------------------ |
|
||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| OpenBSD | ``pkg_add ocrmypdf`` |
|
||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||
|
||||
对于其他用户,[请参阅我们的文档](https://ocrmypdf.readthedocs.io/en/latest/installation.html)了解安装步骤。
|
||||
其他用户请[参阅我们的文档](https://ocrmypdf.readthedocs.io/en/latest/installation.html)了解安装步骤。
|
||||
|
||||
## 语言
|
||||
|
||||
OCRmyPDF 使用 Tesseract 进行 OCR,并依赖其语言包。对于 Linux 用户,您通常可以找到提供语言包的软件包:
|
||||
OCRmyPDF 使用 Tesseract 执行 OCR,并依赖其语言包。对于 Linux 用户,通常可以找到提供语言包的软件包:
|
||||
|
||||
```bash
|
||||
# 显示所有 Tesseract 语言包的列表
|
||||
apt-cache search tesseract-ocr
|
||||
|
||||
# Debian/Ubuntu 用户
|
||||
apt-get install tesseract-ocr-chi-sim # 示例:安装中文简体语言包
|
||||
apt-cache search tesseract-ocr # 显示所有 Tesseract 语言包列表
|
||||
apt-get install tesseract-ocr-chi-sim # 示例:安装简体中文语言包
|
||||
|
||||
|
||||
# Arch Linux 用户
|
||||
pacman -S tesseract-data-eng tesseract-data-deu # 示例:安装英语和德语语言包
|
||||
|
||||
# OpenBSD 用户
|
||||
pkg_info -aQ tesseract # 显示所有 Tesseract 语言包列表
|
||||
pkg_add tesseract-cym # 示例:安装威尔士语语言包
|
||||
|
||||
# brew macOS 用户
|
||||
brew install tesseract-lang
|
||||
|
||||
# Fedora 用户
|
||||
dnf search tesseract-langpack # 显示所有 Tesseract 语言包列表
|
||||
dnf install tesseract-langpack-ita # 示例:安装意大利语语言包
|
||||
|
||||
|
||||
```
|
||||
|
||||
然后,您可以向 OCRmyPDF 传递 `-l LANG` 参数,提示它应该搜索哪些语言。可以请求多种语言。
|
||||
随后可以向 OCRmyPDF 传递 `-l LANG` 参数,提示它应搜索哪些语言。可以同时请求多种语言。
|
||||
|
||||
OCRmyPDF 支持 Tesseract 4.1.1+。它会自动使用在 `PATH` 环境变量中首先找到的版本。在 Windows 上,如果 `PATH` 不提供 Tesseract 二进制文件,我们会根据 Windows 注册表使用已安装的最高版本号。
|
||||
OCRmyPDF 支持 Tesseract 4.1.1+。它会自动使用 `PATH` 环境变量中首先找到的版本。在 Windows 上,如果 `PATH` 中没有 Tesseract 二进制文件,我们会根据 Windows 注册表使用已安装的最高版本号。
|
||||
|
||||
## 文档和支持
|
||||
|
||||
安装 OCRmyPDF 后,可以通过以下方式访问内置帮助,解释命令语法和选项:
|
||||
安装 OCRmyPDF 后,可以通过以下命令访问内置帮助,了解命令语法和选项:
|
||||
|
||||
```bash
|
||||
ocrmypdf --help
|
||||
@@ -115,13 +121,13 @@ ocrmypdf --help
|
||||
|
||||
我们的[文档托管在 Read the Docs 上](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
|
||||
请在我们的 [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) 页面上报告问题,并遵循问题模板以获得快速响应。
|
||||
请在我们的 [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) 页面报告问题,并遵循 issue 模板以便快速获得响应。
|
||||
|
||||
## 功能演示
|
||||
|
||||
```bash
|
||||
# 添加 OCR 层并转换为 PDF/A
|
||||
ocrmypdf input.pdf output.pdf
|
||||
# 添加 OCR 层并要求输出 PDF/A
|
||||
ocrmypdf --output-type pdfa input.pdf output.pdf
|
||||
|
||||
# 将图像转换为单页 PDF
|
||||
ocrmypdf input.jpg output.pdf
|
||||
@@ -129,45 +135,53 @@ ocrmypdf input.jpg output.pdf
|
||||
# 就地为文件添加 OCR(仅在成功时修改文件)
|
||||
ocrmypdf myfile.pdf myfile.pdf
|
||||
|
||||
# 使用非英语语言进行 OCR(查找您语言的 ISO 639-3 代码)
|
||||
# 使用非英语语言执行 OCR(请查找对应语言的 ISO 639-3 代码)
|
||||
ocrmypdf -l fra LeParisien.pdf LeParisien.pdf
|
||||
|
||||
# OCR 多语言文档
|
||||
ocrmypdf -l eng+fra Bilingual-English-French.pdf Bilingual-English-French.pdf
|
||||
|
||||
# 校正(矫正倾斜的页面)
|
||||
# 校正歪斜页面
|
||||
ocrmypdf --deskew input.pdf output.pdf
|
||||
```
|
||||
|
||||
更多功能,请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
更多功能请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
|
||||
## 要求
|
||||
|
||||
除了所需的 Python 版本外,OCRmyPDF 还需要外部程序安装 Ghostscript 和 Tesseract OCR。OCRmyPDF 是纯 Python 编写的,几乎可以在所有平台上运行:Linux、macOS、Windows 和 FreeBSD。
|
||||
除所需的 Python 版本外,OCRmyPDF 还需要安装 Ghostscript 和 Tesseract OCR 这两个外部程序。OCRmyPDF 是纯 Python 项目,几乎可以在所有平台上运行:Linux、macOS、Windows 和 FreeBSD。
|
||||
|
||||
## 媒体报道
|
||||
## 插件
|
||||
|
||||
- [使用 OCRmyPDF 实现无纸化](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a)
|
||||
- [将扫描文档转换为带有编辑的压缩可搜索 PDF](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c)
|
||||
- [c't 1-2014, 第 59 页](https://heise.de/-2279695):在德国领先的 IT 杂志 c't 中详细介绍 OCRmyPDF v1.0
|
||||
- [heise Open Source, 09/2014: 使用 OCRmyPDF 进行文本识别](https://heise.de/-2356670)
|
||||
- [heise 使用 OCRmyPDF 创建可搜索的 PDF 文档](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [优秀实用工具:OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser 使用 OCRmyPDF 和 Scanbd 自动化文本识别](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [Y Combinator 讨论](https://news.ycombinator.com/item?id=32028752)
|
||||
OCRmyPDF 提供插件接口,允许扩展或替换其能力。以下是我们知道的一些插件:
|
||||
|
||||
## 商业咨询
|
||||
- [OCRmyPDF-AppleOCR](https://github.com/mkyt/ocrmypdf-AppleOCR):用 Apple Vision Framework 替换标准 Tesseract OCR 引擎。需要 macOS。
|
||||
- [OCRmyPDF-EasyOCR](https://github.com/ocrmypdf/OCRmyPDF-EasyOCR):用 EasyOCR 替换标准 Tesseract OCR 引擎;EasyOCR 是基于 PyTorch 的较新 OCR 引擎。强烈建议使用 GPU。
|
||||
- [OCRmyPDF-PaddleOCR](https://github.com/clefru/ocrmypdf-paddleocr):用 PaddleOCR 替换标准 Tesseract OCR 引擎;PaddleOCR 是功能强大的 GPU 加速 OCR 引擎。
|
||||
|
||||
如果没有公司和用户选择为功能开发和咨询提供支持,OCRmyPDF 就不会成为今天的软件。我们很乐意讨论所有咨询,无论是扩展现有功能集,还是将 OCRmyPDF 集成到更大的系统中。
|
||||
[paperless-ngx](https://docs.paperless-ngx.com/) 将 OCRmyPDF 集成到可搜索的文档管理系统中。
|
||||
|
||||
## 新闻与媒体
|
||||
|
||||
- [Going paperless with OCRmyPDF](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a)
|
||||
- [Converting a scanned document into a compressed searchable PDF with redactions](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c)
|
||||
- [c't 1-2014,第 59 页](https://heise.de/-2279695):德国领先 IT 杂志 c't 对 OCRmyPDF v1.0 的详细介绍
|
||||
- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670)
|
||||
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [Y Combinator discussion](https://news.ycombinator.com/item?id=32028752)
|
||||
|
||||
## 商务咨询
|
||||
|
||||
如果没有公司和用户选择支持功能开发与咨询服务,OCRmyPDF 不会成为今天的软件。无论是扩展现有功能集,还是将 OCRmyPDF 集成到更大的系统中,我们都很乐意讨论各类咨询需求。
|
||||
|
||||
## 许可证
|
||||
|
||||
OCRmyPDF 软件根据 Mozilla 公共许可证 2.0 (MPL-2.0) 授权。此许可证允许将 OCRmyPDF 与其他代码集成,包括商业和闭源代码,但要求您发布对 OCRmyPDF 所做的源代码级修改。
|
||||
OCRmyPDF 软件采用 Mozilla Public License 2.0 (MPL-2.0) 授权。该许可证允许将 OCRmyPDF 与其他代码集成,包括商业代码和闭源代码,但要求你发布对 OCRmyPDF 所做的源代码级修改。
|
||||
|
||||
OCRmyPDF 的某些组件有其他许可证,如标准 SPDX 许可证标识符或 DEP5 版权和许可信息文件所示。一般来说,非核心代码根据 MIT 许可,文档和测试文件根据 Creative Commons ShareAlike 4.0 (CC-BY-SA 4.0) 许可。
|
||||
OCRmyPDF 的某些组件采用其他许可证,具体由标准 SPDX 许可证标识符或 DEP5 版权与许可信息文件标明。一般来说,非核心代码采用 MIT 许可证,文档和测试文件采用 Creative Commons ShareAlike 4.0 (CC-BY-SA 4.0) 许可证。
|
||||
|
||||
## 免责声明
|
||||
|
||||
本软件按"原样"分发,不提供任何明示或暗示的保证或条件。
|
||||
|
||||
这份中文版 README.md 保留了原始文档的所有重要信息,包括功能介绍、安装说明、语言支持、使用示例等内容,同时保持了原始格式和结构。
|
||||
本软件按“原样”分发,不提供任何明示或暗示的保证或条件。
|
||||
|
||||
+4
-3
@@ -18,8 +18,9 @@ import cyclopts
|
||||
from packaging.version import InvalidVersion, Version
|
||||
|
||||
try:
|
||||
from github import Github, GithubException
|
||||
from github import Auth, Github, GithubException
|
||||
except ImportError:
|
||||
Auth = None # type: ignore
|
||||
Github = None # type: ignore
|
||||
GithubException = Exception # type: ignore
|
||||
|
||||
@@ -68,7 +69,7 @@ def validate_release_notes(new_version: str) -> bool:
|
||||
|
||||
def get_github_client():
|
||||
"""Get an authenticated GitHub client."""
|
||||
if Github is None:
|
||||
if Github is None or Auth is None:
|
||||
print(f"{RED}error:{OFF} PyGithub is not installed")
|
||||
print(" Install with: pip install PyGithub")
|
||||
return None
|
||||
@@ -92,7 +93,7 @@ def get_github_client():
|
||||
return None
|
||||
|
||||
try:
|
||||
return Github(token)
|
||||
return Github(auth=Auth.Token(token))
|
||||
except GithubException as e:
|
||||
print(f"{RED}error:{OFF} Failed to authenticate with GitHub: {e}")
|
||||
return None
|
||||
|
||||
+131
-3
@@ -121,6 +121,30 @@ representation. This is useful for redoing OCR, for fixing OCR text
|
||||
with a damaged character map (text is selectable but not searchable),
|
||||
and destroying redacted information.
|
||||
|
||||
### Tagged PDFs and structural markup
|
||||
|
||||
Some PDFs carry a logical structure tree (`/StructTreeRoot`), the markup that
|
||||
makes a "Tagged PDF" — typically the result of layout analysis or a born-digital
|
||||
export. By default OCRmyPDF treats this as a signal that the document may not need
|
||||
OCR and exits, in the same way it stops on PDFs that already contain text. Use
|
||||
`--tagged-pdf-mode ignore`, or one of `--mode skip`/`redo`/`force`, to process
|
||||
such a file anyway.
|
||||
|
||||
OCRmyPDF cannot rebuild a structure tree to match newly recognized text. When
|
||||
`--force-ocr` rasterizes pages, or `--redo-ocr` strips and rewrites the text layer,
|
||||
the structure tree no longer corresponds to the page content, so it is discarded.
|
||||
`--mode skip` leaves text pages untouched, so their structural markup is preserved.
|
||||
|
||||
:::{note}
|
||||
Preservation under `--mode skip` only holds when the output is not converted to
|
||||
PDF/A. PDF/A conversion is performed by Ghostscript, and Ghostscript 10.x discards
|
||||
the structure tree during conversion (Ghostscript 9.x preserved it). Because the
|
||||
default `--output-type auto` may fall back to Ghostscript, use
|
||||
`--output-type pdf` if you need to guarantee that a Tagged PDF's structural markup
|
||||
survives. For best results, install veraPDF so that speculative PDF/A
|
||||
conversion can sidestep this issue entirely in most real cases.
|
||||
:::
|
||||
|
||||
### Time and image size limits
|
||||
|
||||
By default, OCRmyPDF permits tesseract to run for three minutes (180
|
||||
@@ -187,6 +211,13 @@ include:
|
||||
Overrides the path to Tesseract's data files. This can allow
|
||||
simultaneous installation of the "best" and "fast" training data
|
||||
sets. OCRmyPDF does not manage this environment variable.
|
||||
|
||||
If you point ``TESSDATA_PREFIX`` at a hand-assembled ``tessdata``
|
||||
folder (for example, individual ``.traineddata`` files downloaded
|
||||
from tessdata_best), make sure it also contains the ``configs/``
|
||||
subdirectory with the ``hocr`` and ``txt`` files. OCRmyPDF requires
|
||||
these; without them Tesseract produces no output. See
|
||||
:ref:`Tesseract cannot open its config file <tesseract-config-missing>`.
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
@@ -419,6 +450,70 @@ curves. In this case, you may want to use a different color conversion
|
||||
strategy. The `--color-conversion-strategy` option allows you to select a
|
||||
different strategy, such as `RGB`.
|
||||
|
||||
## Advanced Ghostscript tuning
|
||||
|
||||
:::{versionadded} 17.5.0
|
||||
:::
|
||||
|
||||
OCRmyPDF intentionally hides most Ghostscript controls because Ghostscript
|
||||
is a legacy code path. The preferred PDF/A pipeline in v17+ uses pypdfium2
|
||||
as the rasterizer and verapdf to validate speculative PDF/A output, with
|
||||
Ghostscript reserved as a fallback for PDFs that cannot be made compliant
|
||||
without it. OCRmyPDF's separate optimizer (controlled by `--optimize`,
|
||||
`--jpeg-quality`, `--png-quality`, etc.) is the supported way to shrink
|
||||
output PDFs: it gives consistent results across input files, and isolates
|
||||
Ghostscript so it can focus on producing a PDF/A with as few image
|
||||
transformations as possible.
|
||||
|
||||
The two options below are exposed for advanced users who want to tune
|
||||
Ghostscript's intermediate PDF/A output directly. Most users will get
|
||||
more predictable results from the optimizer.
|
||||
|
||||
### `--ghostscript-jpeg-quality Q`
|
||||
|
||||
Sets Ghostscript's `-dJPEGQ` switch for images that Ghostscript chooses
|
||||
to recompress to JPEG while building a PDF/A. `Q=0` requests maximum
|
||||
compression and `Q=100` requests best quality; if the flag is omitted,
|
||||
OCRmyPDF passes `95` (the historical default). This only affects images
|
||||
Ghostscript transcodes — existing JPEGs pass through unchanged on modern
|
||||
Ghostscript releases. For end-to-end JPEG quality tuning, prefer
|
||||
`--jpeg-quality`, which is implemented by the OCRmyPDF optimizer and is
|
||||
applied independently of whatever Ghostscript decides to do.
|
||||
|
||||
Note: setting both `--ghostscript-jpeg-quality` and `--jpeg-quality` can
|
||||
result in double JPEG recompression, since the optimizer may re-encode
|
||||
images that Ghostscript already recompressed. This can degrade quality
|
||||
in subtle ways.
|
||||
|
||||
### `--ghostscript-jpeg-maxdpi DPI`
|
||||
|
||||
Enables Ghostscript's image downsampling and caps color, grayscale, and
|
||||
monochrome image resolution to `DPI`. The downsample threshold is set to
|
||||
`1.0`, so any image whose effective DPI exceeds the cap will be
|
||||
downsampled.
|
||||
|
||||
Reducing JPEG quality is almost always a better trade than downsampling
|
||||
at the same compression budget: a 400 DPI JPEG at modest quality usually
|
||||
looks much better than a 200 DPI JPEG, because the JPEG codec can spend
|
||||
bits where they count. Downsampling is also dangerous for PDFs that
|
||||
combine a low-resolution color image with a high-resolution monochrome
|
||||
mask — capping the mask resolution can produce visible quality loss.
|
||||
For these reasons, prefer `--jpeg-quality` over `--ghostscript-jpeg-maxdpi`
|
||||
unless you specifically want to force a hard DPI cap.
|
||||
|
||||
Example:
|
||||
|
||||
```bash
|
||||
ocrmypdf --output-type pdfa \
|
||||
--ghostscript-jpeg-quality 80 \
|
||||
--ghostscript-jpeg-maxdpi 150 \
|
||||
in.pdf out.pdf
|
||||
```
|
||||
|
||||
These options only take effect when Ghostscript is invoked for PDF/A
|
||||
conversion (`--output-type pdfa`, `pdfa-1`, `pdfa-2`, or `pdfa-3`, or
|
||||
when `--output-type auto` falls back to Ghostscript).
|
||||
|
||||
## PDF/A output modes
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
@@ -438,6 +533,36 @@ OCRmyPDF can produce PDF/A compliant output for long-term archival. The
|
||||
| `pdf` | Standard PDF, no PDF/A conversion |
|
||||
| `none` | No output file (useful with `--sidecar`) |
|
||||
|
||||
### Non-embedded fonts and PDF/A
|
||||
|
||||
:::{versionadded} 17.8.0
|
||||
OCRmyPDF now refuses to corrupt non-embedded CID text layers during PDF/A
|
||||
conversion.
|
||||
:::
|
||||
|
||||
PDF/A requires every font to be embedded. If your input already has a text
|
||||
layer that uses *non-embedded* CID fonts — most commonly a CJK
|
||||
(Chinese-Japanese-Korean) OCR layer
|
||||
produced by Adobe Acrobat, which relies on the reader's system fonts —
|
||||
Ghostscript would have to substitute and re-embed a replacement font to make
|
||||
the file PDF/A. For CID-keyed (CJK) fonts this routinely corrupts the
|
||||
character-to-Unicode mapping, so the text silently becomes garbage or stops
|
||||
being searchable even though the page still *looks* correct.
|
||||
|
||||
Rather than emit corrupted output, OCRmyPDF detects this situation and:
|
||||
|
||||
- with `--output-type auto` (the default), produces a regular PDF instead of
|
||||
PDF/A, preserving the existing text layer exactly;
|
||||
- with an explicit `--output-type pdfa` (or `pdfa-1`/`pdfa-2`/`pdfa-3`), stops
|
||||
with an error.
|
||||
|
||||
This is a Ghostscript limitation that OCRmyPDF cannot repair, because a
|
||||
non-embedded font cannot be made PDF/A-compliant without re-embedding it. To
|
||||
keep the existing text layer, use `--output-type pdf`. To produce PDF/A anyway,
|
||||
re-run OCR with `--force-ocr`, which discards the original text layer and
|
||||
rebuilds it with embedded fonts. Text layers whose fonts are *already embedded*
|
||||
are converted to PDF/A normally.
|
||||
|
||||
### Speculative PDF/A conversion
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
@@ -451,9 +576,12 @@ fast "speculative" PDF/A conversion that avoids Ghostscript when possible:
|
||||
3. If validation passes, Ghostscript is skipped entirely
|
||||
4. If validation fails or verapdf is unavailable, falls back to Ghostscript
|
||||
|
||||
This approach is faster and avoids some Ghostscript limitations (such as
|
||||
image transcoding), but only works for PDFs that are already "mostly"
|
||||
PDF/A compliant.
|
||||
This fast path avoids some Ghostscript limitations (such as image
|
||||
transcoding) and is used whenever it can produce valid PDF/A. When it
|
||||
cannot — for example when veraPDF is not installed, or the input needs real
|
||||
conversion — `auto` falls back to Ghostscript so that it still produces
|
||||
PDF/A by default, matching OCRmyPDF 16 and earlier. If even Ghostscript
|
||||
cannot safely produce PDF/A, `auto` outputs a regular PDF instead of failing.
|
||||
|
||||
### PDF/A conversion flow
|
||||
|
||||
|
||||
+9
-1
@@ -174,7 +174,15 @@ docker run \
|
||||
--env PYTHONUNBUFFERED=1 \
|
||||
--interactive --tty --entrypoint python3 \
|
||||
jbarlow83/ocrmypdf \
|
||||
watcher.py
|
||||
/app/watcher.py
|
||||
:::
|
||||
|
||||
:::{note}
|
||||
The image runs as the non-root `app` user (uid 1000) by default, so it
|
||||
may not be able to write to the `/output` and `/processed` volumes unless
|
||||
you add a `--user` argument. The correct value depends on whether you use
|
||||
rootful Docker, rootless Docker, or Podman -- see
|
||||
{ref}`Bind-mounted volumes <docker-volumes>` for details.
|
||||
:::
|
||||
|
||||
This service will watch for a file that matches `/input/\*.pdf`, convert
|
||||
|
||||
@@ -178,6 +178,20 @@ html_theme = 'sphinx_rtd_theme'
|
||||
#
|
||||
html_theme_options = {}
|
||||
|
||||
# ReadTheDocs used to inject the "Edit on GitHub" context automatically, but
|
||||
# dropped it when it switched to Addons, so set it explicitly here. This makes
|
||||
# sphinx_rtd_theme add an "Edit on GitHub" link to each page that points at the
|
||||
# corresponding source file in the repository, replacing the static
|
||||
# "View page source" (_sources/*.txt) link. See
|
||||
# https://github.com/ocrmypdf/OCRmyPDF/issues/1490
|
||||
html_context = {
|
||||
'display_github': True,
|
||||
'github_user': 'ocrmypdf',
|
||||
'github_repo': 'OCRmyPDF',
|
||||
'github_version': 'main',
|
||||
'conf_py_path': '/docs/',
|
||||
}
|
||||
|
||||
# Add any paths that contain custom themes here, relative to this directory.
|
||||
# html_theme_path = []
|
||||
|
||||
|
||||
+47
-12
@@ -31,6 +31,16 @@ ocrmypdf --output-type pdf input.pdf output.pdf
|
||||
ocrmypdf --output-type pdfa --pdfa-image-compression jpeg input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Reduce JPEG quality with the optimizer
|
||||
|
||||
This is the recommended way to shrink JPEG content in the output. The
|
||||
optimizer applies regardless of `--output-type`, so it works on both
|
||||
plain PDFs and Ghostscript-produced PDF/A files.
|
||||
|
||||
```bash
|
||||
ocrmypdf --optimize 2 --jpeg-quality 60 input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Modify a file in place
|
||||
|
||||
The file will only be overwritten if OCRmyPDF is successful.
|
||||
@@ -239,19 +249,34 @@ case. Use `--tesseract-non-ocr-timeout` to control the timeout for
|
||||
non-OCR operations, if needed.
|
||||
:::
|
||||
|
||||
### Remove all text or OCR from my PDF
|
||||
### Remove the OCR text layer from my PDF
|
||||
|
||||
This is getting ridiculous, but OCRmyPDF can complete strip all textual
|
||||
information from a PDF and reconstruct it as a \"bag of images\" PDF.
|
||||
To remove the invisible OCR text layer while keeping the original pages
|
||||
exactly as they are -- no rasterizing, no change to images or visible
|
||||
content, and a smaller output file -- use `--mode strip`:
|
||||
|
||||
```bash
|
||||
ocrmypdf --mode strip input.pdf output.pdf
|
||||
```
|
||||
|
||||
Why would you want to do this? Perhaps you have a PDF where OCR failed to
|
||||
produce useful results and you simply want to get rid of it.
|
||||
|
||||
`--mode strip` removes only text drawn as *invisible* (PDF text render
|
||||
mode 3), which is how OCRmyPDF and most OCR tools add a searchable layer
|
||||
over a scanned page. Some OCR products -- and OCRmyPDF v2.2 and earlier --
|
||||
instead draw *visible* text and paint an opaque image on top of it. That
|
||||
text is part of the visible page, so `--mode strip` cannot remove it
|
||||
without altering the page's appearance.
|
||||
|
||||
To strip *all* text, including such visible text, rasterize the whole page
|
||||
into a \"bag of images\" PDF instead (this rebuilds every page as an image,
|
||||
so the file usually grows and vector content is lost):
|
||||
|
||||
```bash
|
||||
ocrmypdf --ocr-engine none --force-ocr input.pdf output.pdf
|
||||
```
|
||||
|
||||
Why would you want to do this? Perhaps you have a PDF where OCR fails to
|
||||
produce useful results, and just want to get rid of all OCR information.
|
||||
This command also removes OCR generated by third party tools.
|
||||
|
||||
### Optimize images without performing OCR
|
||||
|
||||
You can also optimize all images without performing any OCR:
|
||||
@@ -333,12 +358,22 @@ Hyphens denote a range of pages and commas separate page numbers. If you
|
||||
prefer to use spaces, quote all of the page numbers:
|
||||
`--pages '2, 3, 5, 7'`.
|
||||
|
||||
The token `end` (case-insensitive) is an alias for the last page in the
|
||||
document. For example, `--pages 3-end` OCRs from page 3 through the
|
||||
final page, and `--pages end` OCRs only the last page:
|
||||
|
||||
```bash
|
||||
ocrmypdf --pages 3-end input.pdf output.pdf
|
||||
ocrmypdf --pages end input.pdf output.pdf
|
||||
```
|
||||
|
||||
OCRmyPDF will warn if your list of page numbers contains duplicates or
|
||||
overlapping pages. OCRmyPDF does not currently account for document page
|
||||
numbers, such as an introduction section of a book that uses Roman
|
||||
numerals. It simply counts the number of virtual pieces of paper since
|
||||
the start. If your list of pages is out of numerical order, OCRmyPDF
|
||||
will sort it for you.
|
||||
overlapping pages. (Repeated page numbers are de-duplicated automatically,
|
||||
since the underlying set of pages is what matters.) OCRmyPDF does not
|
||||
currently account for document page numbers, such as an introduction
|
||||
section of a book that uses Roman numerals. It simply counts the number
|
||||
of virtual pieces of paper since the start. If your list of pages is out
|
||||
of numerical order, OCRmyPDF will sort it for you.
|
||||
|
||||
Regardless of the argument to `--pages`, OCRmyPDF will optimize all
|
||||
pages/images in the file and convert it to PDF/A, unless you disable
|
||||
|
||||
+84
-23
@@ -71,15 +71,29 @@ application (as opposed to the more conventional case, where a Docker
|
||||
container runs as a server). For that reason we usually use the `--rm`
|
||||
argument to delete the container when it exits.
|
||||
|
||||
:::{note}
|
||||
The image runs as a non-root user (`app`, uid/gid 1000) by default,
|
||||
rather than as root. This is a defense-in-depth measure: a flaw in
|
||||
OCRmyPDF or one of its dependencies cannot trivially act as root inside
|
||||
the container. The examples below assume **rootless Docker** or
|
||||
**Podman**; the differences for traditional *rootful* Docker are
|
||||
described separately under *Special case: rootful Docker* below.
|
||||
:::
|
||||
|
||||
To start a Docker container (instance of the image):
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm -i jbarlow83/ocrmypdf-alpine (... all other arguments here...) - -
|
||||
:::
|
||||
|
||||
For convenience, create a shell alias to hide the Docker command. It is
|
||||
easier to send the input file as stdin and read the output from stdout
|
||||
-- **this avoids the messy permission issues with Docker entirely**.
|
||||
### Recommended: pipe through stdin and stdout
|
||||
|
||||
The easiest and most portable way to use the image is to send the input
|
||||
file on stdin and read the output from stdout. This **avoids file
|
||||
permission issues entirely** -- nothing is written to a mounted
|
||||
directory, so it does not matter which user the container runs as, nor
|
||||
whether you use rootless or rootful Docker. For convenience, create a
|
||||
shell alias to hide the Docker command:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i jbarlow83/ocrmypdf-alpine'
|
||||
@@ -90,28 +104,42 @@ docker_ocrmypdf - - <input.pdf >output.pdf
|
||||
Or in the wonderful [fish shell](https://fishshell.com/):
|
||||
|
||||
:::{code} fish
|
||||
alias docker_ocrmypdf 'docker run --rm jbarlow83/ocrmypdf-alpine'
|
||||
alias docker_ocrmypdf 'docker run --rm -i jbarlow83/ocrmypdf-alpine'
|
||||
funcsave docker_ocrmypdf
|
||||
:::
|
||||
|
||||
Alternately, you could mount the local current working directory as a
|
||||
Docker volume:
|
||||
{#docker-volumes}
|
||||
### Bind-mounted volumes
|
||||
|
||||
If you would rather mount a directory and pass file paths, you need to
|
||||
consider which user owns the files OCRmyPDF writes back into that
|
||||
directory. The image's default working directory is `/data`, so mounting
|
||||
your files there lets you pass plain relative paths without an explicit
|
||||
`--workdir`. Because the container runs as the non-root `app` user, the
|
||||
right invocation otherwise depends on your container runtime.
|
||||
|
||||
**Rootless Docker (the assumed default).** Your own account runs the
|
||||
daemon, so the container's `root` maps back to *your* unprivileged host
|
||||
user, while every other container uid -- including the image's default
|
||||
`app`/1000 -- maps to a *subordinate* uid. A directory you own on the
|
||||
host therefore appears owned by `root` inside the container, so the
|
||||
default `app` user usually **cannot write to it at all**. Run the job as
|
||||
container-`root`, which under rootless Docker is still your ordinary host
|
||||
user, so the write succeeds and the output is owned by you:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" --workdir /data -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias docker_ocrmypdf='docker run --rm -i --user 0:0 -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
## Podman
|
||||
|
||||
Especially if you use [Podman](https://podman.io/) (or use Docker in
|
||||
rootless mode), you may need to add `--userns keep-id` there,
|
||||
otherwise you may get access errors, because the user ID is otherwise not
|
||||
mapped to the same UID as on the host:
|
||||
**Podman.** Podman provides `--userns keep-id`, which maps your host uid
|
||||
straight through into the container. Combined with `--user`, you run as
|
||||
your own uid and own the output directly, otherwise you may get access
|
||||
errors because the user ID is not mapped to the same UID as on the host:
|
||||
|
||||
:::{code} bash
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id --workdir /data -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
If you have SELinux enabled, you may additionally need to add the `:Z` [suffix to
|
||||
@@ -124,10 +152,27 @@ the end of the linked podman documentation for details. This results in
|
||||
the following full command:
|
||||
|
||||
:::{code} bash
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id --workdir /data -v "$PWD:/data" --security-opt label=disable jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id -v "$PWD:/data" --security-opt label=disable jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
{#docker-rootful}
|
||||
### Special case: rootful Docker
|
||||
|
||||
With a traditional root daemon, container uid *N* is the *same* uid *N*
|
||||
on the host. Running the container as root would therefore fill your
|
||||
mounted directory with root-owned files and -- more importantly -- a
|
||||
container escape would run as real host root. Drop to your own uid so the
|
||||
output is owned by you and the process stays unprivileged:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
The non-root default and the `--user` override both reduce the risk here,
|
||||
but rootless Docker or Podman remain the safer choice when available.
|
||||
|
||||
{#docker-lang-packs}
|
||||
## Adding languages to the Docker image
|
||||
|
||||
@@ -139,8 +184,12 @@ creating a new Dockerfile based on the public one.
|
||||
:::{code} dockerfile
|
||||
FROM jbarlow83/ocrmypdf
|
||||
|
||||
# The image runs as the non-root "app" user, so switch back to root for
|
||||
# build steps that install packages, then drop back to "app".
|
||||
USER root
|
||||
# Example: add Italian
|
||||
RUN apt install tesseract-ocr-ita
|
||||
RUN apt-get update && apt-get install -y tesseract-ocr-ita
|
||||
USER app
|
||||
:::
|
||||
|
||||
To install language packs (training data) such as the
|
||||
@@ -179,7 +228,11 @@ Extending the Docker image
|
||||
--------------------------
|
||||
|
||||
You can extend the Docker image with your own customizations, similar to
|
||||
the way it is extended to add language packs.
|
||||
the way it is extended to add language packs. Because the image runs as
|
||||
the non-root `app` user, switch to `USER root` for any build steps that
|
||||
require root (installing packages, writing to system directories) and
|
||||
back to `USER app` afterwards, as shown in the language pack example
|
||||
above.
|
||||
|
||||
Note that the Docker image is subject to change at any time. For
|
||||
example, the base image may be updated to a newer version of Ubuntu or
|
||||
@@ -196,7 +249,7 @@ Executing the test suite
|
||||
The OCRmyPDF test suite is installed with image. To run it:
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm --entrypoint python jbarlow83/ocrmypdf -m pytest
|
||||
docker run --rm --workdir /app --entrypoint python jbarlow83/ocrmypdf -m pytest
|
||||
:::
|
||||
|
||||
Accessing the shell
|
||||
@@ -205,7 +258,15 @@ Accessing the shell
|
||||
To use the shell in the Docker image:
|
||||
|
||||
:::{code} bash
|
||||
docker run -it --entrypoint sh jbarlow83/ocrmypdf
|
||||
docker run -it --entrypoint sh jbarlow83/ocrmypdf-alpine
|
||||
:::
|
||||
|
||||
This shell runs as the non-root `app` user. If you need root inside the
|
||||
container -- for example to install extra packages with `apk` or `apt` --
|
||||
add `--user root`:
|
||||
|
||||
:::{code} bash
|
||||
docker run -it --user root --entrypoint sh jbarlow83/ocrmypdf-alpine
|
||||
:::
|
||||
|
||||
Using the OCRmyPDF web service wrapper
|
||||
@@ -215,7 +276,7 @@ The OCRmyPDF Docker image includes an example, barebones HTTP web
|
||||
service. The webservice may be launched as follows:
|
||||
|
||||
:::{code} bash
|
||||
docker run --entrypoint python -p 5000:5000 jbarlow83/ocrmypdf webservice.py
|
||||
docker run --entrypoint python -p 5000:5000 jbarlow83/ocrmypdf /app/webservice.py
|
||||
:::
|
||||
|
||||
We omit the `--rm` parameter so that the container will not be
|
||||
|
||||
@@ -49,3 +49,31 @@ pdftk input.pdf cat output output.pdf
|
||||
|
||||
Sometimes Acrobat can repair PDFs with its [Preflight
|
||||
tool](https://helpx.adobe.com/acrobat/using/correcting-problem-areas-preflight-tool.html).
|
||||
|
||||
(tesseract-config-missing)=
|
||||
|
||||
## Tesseract cannot open its config file \'hocr\' or \'txt\'
|
||||
|
||||
:::{code}
|
||||
ERROR - Tesseract cannot open its config file 'hocr'.
|
||||
:::
|
||||
|
||||
OCRmyPDF asks Tesseract to produce `hocr` and `txt` output. Tesseract
|
||||
reads the instructions for these output formats from configuration files
|
||||
named `hocr` and `txt` that live in the `configs/` subdirectory of its
|
||||
`tessdata` folder. If those files are missing, Tesseract prints
|
||||
`read_params_file: Can't open hocr`, exits without error, and produces no
|
||||
output.
|
||||
|
||||
This usually happens when a `tessdata` directory was assembled by hand --
|
||||
for example, by downloading individual `.traineddata` files from
|
||||
[tessdata_best](https://github.com/tesseract-ocr/tessdata_best) and
|
||||
pointing `TESSDATA_PREFIX` at them -- because those repositories do not
|
||||
include the `configs/` directory. A complete Tesseract installation from
|
||||
your operating system\'s package manager includes it.
|
||||
|
||||
To fix this, ensure the `configs/hocr` and `configs/txt` files exist in
|
||||
the `tessdata` directory that Tesseract is using. Copying the `configs/`
|
||||
directory from a full Tesseract installation is sufficient. See
|
||||
{envvar}`TESSDATA_PREFIX` for more on selecting an alternate `tessdata`
|
||||
folder.
|
||||
|
||||
+16
-4
@@ -178,11 +178,23 @@ v17 addresses through alternative codepaths. When Ghostscript is used:
|
||||
encoding, which may introduce compression artifacts, if Ghostscript
|
||||
PDF/A is enabled.
|
||||
- Ghostscript may transcode grayscale and color images, potentially
|
||||
lossily, based on an internal algorithm. This
|
||||
behavior can be suppressed by setting `--pdfa-image-compression` to
|
||||
`jpeg` or `lossless` to set all images to one type or the other.
|
||||
Ghostscript lacks an option to maintain the input image's format.
|
||||
lossily, based on an internal algorithm. By default
|
||||
(`--pdfa-image-compression=auto`) OCRmyPDF selects lossless image
|
||||
compression at `-O0` so Ghostscript will not transcode lossless images
|
||||
to JPEG. At `-O1` (the default optimization level) and above, `auto`
|
||||
defers to Ghostscript's heuristic instead; `-O1` is a historical
|
||||
exception, kept for backwards compatibility because coercing it to
|
||||
lossless can substantially bloat output. You can override this by
|
||||
setting `--pdfa-image-compression` to `jpeg` or `lossless` to force all
|
||||
images to one type or the other. `lossless` passes existing JPEGs
|
||||
through untouched (re-encoding them losslessly would only inflate them)
|
||||
while encoding non-JPEG images losslessly.
|
||||
(Modern Ghostscript can copy JPEG images without transcoding them.)
|
||||
Advanced users can also tune Ghostscript's image recompression with
|
||||
`--ghostscript-jpeg-quality` and `--ghostscript-jpeg-maxdpi`; see
|
||||
[Advanced Ghostscript tuning](advanced.md#advanced-ghostscript-tuning).
|
||||
Most users should prefer `--jpeg-quality` (applied by the OCRmyPDF
|
||||
optimizer) over those Ghostscript-scoped controls.
|
||||
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
||||
one of the standard XMP metadata namespaces for PDFs. In particular,
|
||||
PRISM Metadata is removed.
|
||||
|
||||
+10
-1
@@ -98,7 +98,16 @@ If `pngquant` is installed, OCRmyPDF will use it to perform quantize
|
||||
paletted images to reduce their size.
|
||||
|
||||
The quality of JPEGs may be lowered, on the assumption that a lower
|
||||
quality image may be suitable for storage after OCR.
|
||||
quality image may be suitable for storage after OCR. Use `--jpeg-quality`
|
||||
to control the optimizer's JPEG quality target. The optimizer is the
|
||||
recommended way to reduce JPEG image sizes: it applies consistently
|
||||
regardless of whether Ghostscript was used to produce a PDF/A.
|
||||
|
||||
If you specifically need to tune Ghostscript's own PDF/A image handling
|
||||
(for example, to force a hard DPI cap), see
|
||||
[Advanced Ghostscript tuning](advanced.md#advanced-ghostscript-tuning)
|
||||
for the separate `--ghostscript-jpeg-quality` and
|
||||
`--ghostscript-jpeg-maxdpi` options.
|
||||
|
||||
It is not possible to optimize all image types. Uncommon image types may
|
||||
be skipped by the optimizer.
|
||||
|
||||
@@ -3,6 +3,208 @@
|
||||
|
||||
# v17
|
||||
|
||||
## v17.8.0
|
||||
|
||||
- `--output-type auto` (the default) again produces PDF/A whenever it can,
|
||||
matching OCRmyPDF 16's "PDF/A by default" behavior. It first tries the fast
|
||||
Ghostscript-free conversion (validated by veraPDF when available) and now
|
||||
falls back to Ghostscript when that cannot produce PDF/A, only emitting a
|
||||
regular PDF when even Ghostscript cannot safely convert (for example, an
|
||||
input with non-embedded CID/CJK fonts, per {issue}`1561`). A consequence is
|
||||
that the default path may once again invoke Ghostscript, which is slower and
|
||||
may transcode images; use `--output-type pdf` to skip PDF/A conversion
|
||||
entirely.
|
||||
- Fixed detection of veraPDF 1.30.0 and newer: recent builds print JVM
|
||||
warnings before their version string, which caused OCRmyPDF to report
|
||||
veraPDF as unavailable and skip the fast PDF/A path.
|
||||
- OCRmyPDF no longer silently corrupts a non-embedded CID (CJK) text layer when
|
||||
producing PDF/A ({issue}`1561`). PDF/A requires all fonts to be embedded, so
|
||||
Ghostscript substitutes and re-embeds non-embedded CID fonts — such as the OCR
|
||||
text layer Adobe Acrobat adds to scanned CJK documents — which mangles the
|
||||
text and destroys searchability. OCRmyPDF now detects non-embedded CID fonts
|
||||
before conversion: with `--output-type auto` (the default) it produces a
|
||||
regular PDF and preserves the existing text layer, and with an explicit
|
||||
`--output-type pdfa*` it stops with an error rather than emit corrupted
|
||||
output. Use `--output-type pdf` to keep the text layer, or `--force-ocr` to
|
||||
rebuild it with embedded fonts.
|
||||
- Writing the output PDF to standard output (`ocrmypdf input.pdf -`) is now
|
||||
protected against corruption at the operating system level. Previously
|
||||
OCRmyPDF relied on no in-process code — third-party libraries, plugins, or
|
||||
stray `print()` calls — ever writing to stdout; a single accidental write
|
||||
would silently corrupt the PDF. The command line program now saves the real
|
||||
stdout at startup, before plugins are loaded or any worker process/thread is
|
||||
started, and redirects file descriptor 1 to stderr, so that only OCRmyPDF's
|
||||
final PDF output can reach stdout. A consequence is that a plugin which
|
||||
intentionally prints to stdout will have that output redirected to stderr.
|
||||
- Added the public API function {func}`ocrmypdf.configure_stdout_protection`,
|
||||
which installs this same protection. Like {func}`ocrmypdf.configure_logging`,
|
||||
it is optional and intended for callers that want command-line-like behavior;
|
||||
applications that manage their own standard output should not call it.
|
||||
- Fixed an uncaught `UnicodeDecodeError` when processing a PDF whose
|
||||
`/DocumentInfo` dictionary contains a `/Name` key encoded in Latin-1 (or
|
||||
another non-UTF-8 encoding), such as `/Saks#e5r`. `repair_docinfo_nuls` now
|
||||
treats such a block as malformed, logs a message, and continues instead of
|
||||
crashing the pipeline ({issue}`1540`). Current pikepdf releases tolerate these
|
||||
keys by surrogate-escaping them, but older versions raised while iterating the
|
||||
dictionary.
|
||||
|
||||
## v17.7.1
|
||||
|
||||
- Fixed a severe, Windows-specific performance regression in the "Scanning
|
||||
contents" phase, most visible with `--redo-ocr` ({issue}`1662`). Since
|
||||
v16.4.3, OCRmyPDF forced pdfminer's read buffer to 256 MiB to work around a
|
||||
pdfminer bug that mishandled tokens split across the buffer boundary
|
||||
({issue}`1361`). On Windows, CPython's `BufferedReader.read()` eagerly
|
||||
allocates a buffer of the requested size on every read, so the oversized
|
||||
buffer made each of pdfminer's thousands of reads cost tens of milliseconds
|
||||
(this allocation is lazy, and effectively free, on Linux). The underlying
|
||||
pdfminer bug was fixed upstream in pdfminer.six 20250327
|
||||
([#1030](https://github.com/pdfminer/pdfminer.six/pull/1030)), with a
|
||||
follow-up for tokens split across streams in 20260107
|
||||
([#1158](https://github.com/pdfminer/pdfminer.six/pull/1158)), so the
|
||||
workaround has been removed and the minimum pdfminer.six version raised to
|
||||
20260107.
|
||||
- The font discovery used to build the OCR text layer now finds variable fonts
|
||||
such as `NotoSansArabic[wdth,wght].ttf`, the form shipped by Homebrew casks
|
||||
and current Google Fonts releases. Previously only static `-Regular.ttf`/`.otf`
|
||||
files were matched, so users who had installed the correct Noto font still got
|
||||
the glyphless fallback and a "No font found" warning ({issue}`1652`).
|
||||
- Font discovery is now language-aware for CJK: each Chinese, Japanese, and
|
||||
Korean language maps to its own per-language Noto family (NotoSansSC, TC, HK,
|
||||
JP, KR), with the pan-CJK super font kept as a shared fallback, since the
|
||||
per-language fonts are region subsets that may lack glyphs from other scripts.
|
||||
- The warning shown when no installed font has glyphs for some text was reworded
|
||||
to explain the consequence — the text is still added as a searchable, copyable
|
||||
layer but appears blank when highlighted in a viewer — and to name the specific
|
||||
font family to install.
|
||||
|
||||
## v17.7.0
|
||||
|
||||
- The Docker images now run as a non-root user (`app`, uid/gid 1000) by default
|
||||
rather than as root, as a defense-in-depth measure. If you bind-mount a
|
||||
directory for input and output, you may now need to add a `--user` argument so
|
||||
the container can write to it; the correct value differs for rootless Docker,
|
||||
Podman, and rootful Docker, and is described in the Docker documentation.
|
||||
Piping the input and output through stdin/stdout still works with no
|
||||
permission setup.
|
||||
- The Docker images now default their working directory to `/data`, so files in
|
||||
a directory mounted there can be given as relative paths without an explicit
|
||||
`--workdir`.
|
||||
- The Ubuntu Docker image now installs Tesseract 5 from the Ubuntu archive
|
||||
instead of the third-party `alex-p/tesseract-ocr5` PPA, and the base images
|
||||
were updated to Ubuntu 26.04 and Alpine 3.24.
|
||||
- Fixed a missing space in the error message shown when OCRmyPDF cannot access
|
||||
its working directory inside a Docker container.
|
||||
- Updated packaged dependencies, including the optional web service stack
|
||||
(starlette, tornado, python-multipart) and cryptography.
|
||||
|
||||
## v17.6.0
|
||||
|
||||
- When the optimizer encounters an image it cannot process (for example, an
|
||||
exotic colorspace that cannot be transcoded), it now logs a concise warning
|
||||
that the image was left unchanged rather than printing an alarming
|
||||
traceback. The output file was already valid in these cases; only the
|
||||
reporting was misleading. The full traceback is still available at debug
|
||||
verbosity (`-v 1`) ({issue}`846`).
|
||||
- `--pdfa-image-compression=auto` (the default) now selects lossless image
|
||||
compression at `-O0` so Ghostscript no longer transcodes lossless images to
|
||||
JPEG during PDF/A generation. At `-O1` and above, `auto` continues to defer
|
||||
to Ghostscript's heuristic, which may recompress images lossily. `-O1` (the
|
||||
default level) is kept as a historical exception because coercing it to
|
||||
lossless can substantially bloat output; users who want guaranteed lossless
|
||||
image handling should pass `--pdfa-image-compression=lossless` or use `-O0`
|
||||
({issue}`1124`).
|
||||
- `--pdfa-image-compression=lossless` now passes existing JPEG images through
|
||||
unchanged rather than re-encoding them with a lossless codec. Re-encoding an
|
||||
already-lossy JPEG losslessly cannot recover quality and only inflates the
|
||||
file, so JPEGs are preserved while non-JPEG images are encoded losslessly.
|
||||
- OCRmyPDF now validates and repairs malformed page-boundary boxes
|
||||
(``/MediaBox``, ``/CropBox``, ``/TrimBox``, ``/ArtBox``, ``/BleedBox``) in its
|
||||
input, following the PDF 2.0 specification. Coordinates written in invalid
|
||||
exponential notation are reinterpreted ({issue}`1398`); rectangles whose
|
||||
corners are given in reversed order are normalized, which previously crashed
|
||||
with ``NegativeDimensionError`` ({issue}`1526`); and a crop/trim/art/bleed box
|
||||
that falls outside the MediaBox is clamped to their intersection, or discarded
|
||||
when that intersection is empty, which previously produced an output with a
|
||||
zero-height effective page that some viewers refused to open ({issue}`1400`).
|
||||
When a box is discarded, clamped, or reinterpreted, OCRmyPDF logs a warning
|
||||
recommending visual inspection of the output. Thanks @ajdlinux for the initial
|
||||
fix in PR #1691.
|
||||
- OCRmyPDF now discards an embedded Adobe full-text search index
|
||||
(``/Root/PieceInfo/SearchIndex``) from its output. This proprietary index,
|
||||
produced by Acrobat's "Embed Index" feature, is read only by Adobe Acrobat;
|
||||
other viewers ignore it and search the text on the fly. Because any change to
|
||||
a PDF invalidates the index, retaining it after OCRmyPDF rewrites the document
|
||||
would leave a stale index that returns incorrect search results in Acrobat.
|
||||
Modern viewers rebuild a search index on demand, so there is no loss of
|
||||
search capability.
|
||||
- OCRmyPDF now discards embedded per-page thumbnail images (the optional
|
||||
``/Thumb`` image XObject on a page) from its output. OCRmyPDF alters page
|
||||
appearance (deskew, clean, rasterize, re-render) and plugins may edit pages
|
||||
arbitrarily, so a retained thumbnail would be stale and no longer match its
|
||||
page. Embedded thumbnails are a navigation aid that modern viewers generate
|
||||
on demand, so there is no loss of functionality.
|
||||
- Fixed a regression in OCR quality for PDFs that paint a 1-bit image mask
|
||||
(stencil) with a gray or colored fill color. Previously such pages were
|
||||
rasterized as 1-bit black-and-white before OCR, so Ghostscript dithered
|
||||
mid-tone text into an unreadable stipple and Tesseract failed to recognize
|
||||
it. The rasterizer now inspects the fill color used to paint a mask and
|
||||
promotes the page to grayscale or full color as needed, so the distinction
|
||||
is preserved for the OCR engine. This applies to both the Ghostscript and
|
||||
pypdfium rasterizers. {issue}`1688`
|
||||
- The default 1-bit raster device for Ghostscript is now ``pngmonod``
|
||||
(error-diffusion) instead of ``pngmono`` (ordered dithering). It produces
|
||||
better input for OCR on faint or anti-aliased scans at negligible cost and
|
||||
no change to output file size, since the rasterized image is an
|
||||
intermediate that is discarded after OCR.
|
||||
- When rasterizing pages with Ghostscript, OCRmyPDF now enables text and
|
||||
graphics anti-aliasing (``-dTextAlphaBits=4 -dGraphicsAlphaBits=4``) for the
|
||||
grayscale and color raster devices. Ghostscript 10.x renders aliased glyphs
|
||||
that OCR frequently misreads as extra word breaks or substituted characters;
|
||||
anti-aliasing materially improves OCR accuracy on the Ghostscript
|
||||
rasterization path, especially for small fonts at moderate resolution. The
|
||||
1-bit monochrome devices are unaffected, since they perform their own
|
||||
anti-aliased downscaling and older Ghostscript versions reject alpha-bit
|
||||
options on them. Note that the default rasterizer (``--rasterizer auto``)
|
||||
prefers pypdfium2, which already anti-aliases; this change benefits users who
|
||||
select ``--rasterizer ghostscript`` or do not have pypdfium2 installed.
|
||||
OCRmyPDF now also logs which rasterizer rendered each page at debug verbosity
|
||||
(``-v 1``), and the ``--rasterizer`` help text explains the OCR-quality
|
||||
trade-off, to make such reports easier to diagnose. {issue}`1439`
|
||||
- When Tesseract reports a page with many diacritics, OCRmyPDF still logs its
|
||||
interpreted "lots of diacritics - possibly poor OCR" hint, but now also emits
|
||||
Tesseract's raw message at debug verbosity (``-v 1``) so the original wording
|
||||
is available for diagnosis. {issue}`1566`
|
||||
- Added ``--mode strip``, which removes the invisible OCR text layer from a PDF
|
||||
in place. Unlike ``--ocr-engine none --force-ocr``, it does not rasterize the
|
||||
page, so images and visible content are preserved unchanged and the output is
|
||||
smaller rather than larger. Only text drawn as invisible (PDF text render mode
|
||||
3) is removed; some OCR engines -- and OCRmyPDF v2.2 and earlier -- express
|
||||
text as visible glyphs covered by an opaque image, and that text cannot be
|
||||
removed this way. {issue}`1435`
|
||||
|
||||
## v17.5.0
|
||||
|
||||
- Added support for the ``end`` alias in ``--pages``, denoting the last page
|
||||
of the document. For example, ``--pages 3-end`` OCRs from page 3 through
|
||||
the final page. {issue}`1615`
|
||||
- Added ``--ghostscript-jpeg-quality`` and ``--ghostscript-jpeg-maxdpi``
|
||||
advanced options for tuning Ghostscript's PDF/A output. The optimizer's
|
||||
``--jpeg-quality`` remains the recommended file-size control.
|
||||
- Fixed pypdfium2 rasterizer clipping content when the CropBox was smaller
|
||||
than the MediaBox (e.g. JSTOR or cropped PDFs). {issue}`1685`
|
||||
- Fixed Form XObject cycle detection in the optimizer's image xref scan.
|
||||
Self-referential or DAG-shaped Form graphs (notably from PowerPoint
|
||||
exports) previously produced floods of recursion warnings and could hang
|
||||
for minutes. {issue}`1321`
|
||||
- Tesseract config errors are now surfaced as ``TesseractConfigError`` with
|
||||
actionable guidance, instead of crashing later with a confusing
|
||||
``FileNotFoundError`` on the missing hOCR output. {issue}`1687`
|
||||
- Refreshed the Chinese README translation. Thanks @cislunarspace.
|
||||
- Internal refactoring of the ``_exec`` and ``subprocess`` modules to
|
||||
separate probing from execution.
|
||||
- CI dependency updates.
|
||||
|
||||
## v17.4.2
|
||||
|
||||
- Fixed Python API unconditionally overriding ``PIL.Image.MAX_IMAGE_PIXELS``
|
||||
|
||||
@@ -46,6 +46,8 @@ __ocrmypdf_arguments()
|
||||
--rasterizer (PDF page rasterizer)
|
||||
--rotate-pages-threshold (page rotation confidence)
|
||||
--pdfa-image-compression (set PDF/A image compression options)
|
||||
--ghostscript-jpeg-quality (Ghostscript JPEG quality during PDF/A [0..100])
|
||||
--ghostscript-jpeg-maxdpi (cap Ghostscript image DPI during PDF/A)
|
||||
--fast-web-view (if file size if above this amount in MB linearize PDF)
|
||||
--continue-on-soft-render-error (continue after recoverable render errors)
|
||||
--plugin (name of plugin to import)
|
||||
@@ -337,6 +339,7 @@ __ocrmypdf_check_previous()
|
||||
|
||||
--title|--author|--subject|--keywords|--unpaper-args|--pages|--plugin|\
|
||||
--jpeg-quality|--png-quality|--image-dpi|--oversample|--skip-big|--max-image-mpixels|\
|
||||
--ghostscript-jpeg-quality|--ghostscript-jpeg-maxdpi|\
|
||||
--tesseract-timeout|--tesseract-non-ocr-timeout|--tesseract-downsample-above|\
|
||||
--rotate-pages-threshold|--fast-web-view)
|
||||
# argument required but no completions available
|
||||
|
||||
@@ -102,6 +102,8 @@ function __fish_ocrmypdf_pdfa_compression
|
||||
echo -e "lossless\t"(_ "convert color and grayscale images to lossless (PNG)")
|
||||
end
|
||||
complete -c ocrmypdf -x -l pdfa-image-compression -a '(__fish_ocrmypdf_pdfa_compression)' -d "set PDF/A image compression options"
|
||||
complete -c ocrmypdf -x -l ghostscript-jpeg-quality -d "Ghostscript JPEG quality during PDF/A [0..100]"
|
||||
complete -c ocrmypdf -x -l ghostscript-jpeg-maxdpi -d "cap Ghostscript image DPI during PDF/A"
|
||||
|
||||
complete -c ocrmypdf -x -s j -l jobs -d "how many worker processes to use"
|
||||
complete -c ocrmypdf -x -l title -d "set metadata"
|
||||
|
||||
@@ -6,12 +6,19 @@ services:
|
||||
ocrmypdf:
|
||||
restart: always
|
||||
container_name: ocrmypdf
|
||||
image: jbarlow83/ocrmypdf
|
||||
image: jbarlow83/ocrmypdf-alpine
|
||||
volumes:
|
||||
- "/media/scan:/input"
|
||||
- "/mnt/scan:/output"
|
||||
environment:
|
||||
- OCR_OUTPUT_DIRECTORY_YEAR_MONTH=0
|
||||
# The image runs as the non-root "app" user (uid 1000) by default. The
|
||||
# correct value here depends on your runtime, so that the watcher can write
|
||||
# to the /output bind mount and the files end up owned by you:
|
||||
# rootful Docker -> your host uid:gid
|
||||
# rootless Docker -> "0:0" (container root maps to your host user)
|
||||
# Podman -> your host uid:gid, plus `userns_mode: "keep-id"`
|
||||
# See docs/docker.md ("Bind-mounted volumes") for the reasoning.
|
||||
user: "<SET TO YOUR USER ID>:<SET TO YOUR GROUP ID>"
|
||||
entrypoint: python3
|
||||
command: watcher.py
|
||||
command: /app/watcher.py
|
||||
|
||||
+2
-2
@@ -6,7 +6,7 @@ build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "ocrmypdf"
|
||||
version = "17.4.2"
|
||||
version = "17.8.0"
|
||||
description = "OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched"
|
||||
readme = "README.md"
|
||||
license = "MPL-2.0"
|
||||
@@ -16,7 +16,7 @@ dependencies = [
|
||||
"fpdf2>=2.8.0",
|
||||
"img2pdf>=0.5",
|
||||
"packaging>=20",
|
||||
"pdfminer.six>=20220319",
|
||||
"pdfminer.six>=20260107", # fixes parsing of tokens split across the read buffer/streams (gh #1361)
|
||||
"pi-heif", # Heif image format - maintainers: if this is removed, it will NOT break
|
||||
"pikepdf>=10",
|
||||
"Pillow>=10.0.1",
|
||||
|
||||
@@ -19,6 +19,7 @@ from ocrmypdf._version import __version__
|
||||
from ocrmypdf.api import (
|
||||
Verbosity,
|
||||
configure_logging,
|
||||
configure_stdout_protection,
|
||||
ocr,
|
||||
)
|
||||
from ocrmypdf.exceptions import (
|
||||
@@ -53,6 +54,7 @@ __all__ = [
|
||||
'BoundingBox',
|
||||
'configure_debug_logging',
|
||||
'configure_logging',
|
||||
'configure_stdout_protection',
|
||||
'DpiError',
|
||||
'EncryptedPdfError',
|
||||
'Executor',
|
||||
|
||||
@@ -16,7 +16,7 @@ from contextlib import suppress
|
||||
from ocrmypdf import __version__
|
||||
from ocrmypdf._pipelines.ocr import run_pipeline_cli
|
||||
from ocrmypdf._validation import check_options
|
||||
from ocrmypdf.api import Verbosity, configure_logging
|
||||
from ocrmypdf.api import Verbosity, configure_logging, configure_stdout_protection
|
||||
from ocrmypdf.cli import get_options_and_plugins
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
@@ -39,6 +39,11 @@ def sigbus(*args):
|
||||
|
||||
def run(args=None):
|
||||
"""Run the ocrmypdf command line interface."""
|
||||
# Protect the real stdout before loading plugins or starting any worker
|
||||
# processes/threads, so that only our final PDF output can reach it and
|
||||
# stray writes from plugins or libraries are diverted to stderr.
|
||||
configure_stdout_protection()
|
||||
|
||||
options, plugin_manager = get_options_and_plugins(args=args)
|
||||
|
||||
with suppress(AttributeError, PermissionError):
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Probe helper for external executables.
|
||||
|
||||
Each ``ocrmypdf._exec.<tool>`` module describes its external program with a
|
||||
module-level :class:`ToolProbe` and delegates ``version()`` / ``available()``
|
||||
to it. This separates the "is the tool installed and suitable?" question
|
||||
(probing) from the "run the tool" question (execution). Work functions stay
|
||||
as pure module-level functions so they are trivially picklable for use in
|
||||
subprocess workers.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolProbe:
|
||||
"""Describes how to detect an external executable and its version.
|
||||
|
||||
Attributes:
|
||||
program: The program name as it appears on PATH (or a full path).
|
||||
version_arg: The argument that elicits a version string.
|
||||
version_regex: A regex with a capturing group that extracts the
|
||||
version from the program's output.
|
||||
version_cls: A :class:`packaging.version.Version` subclass, used for
|
||||
tools with non-standard version strings (e.g. Tesseract).
|
||||
env: Optional environment overrides applied when probing the version.
|
||||
also_catch: Additional exception types that should be treated as
|
||||
"not available" by :meth:`available`. :class:`OSError` is useful
|
||||
for tools like verapdf whose launcher may fail with non-standard
|
||||
errors when the JVM is missing.
|
||||
"""
|
||||
|
||||
program: str
|
||||
version_arg: str = '--version'
|
||||
version_regex: str = r'(\d+(\.\d+)*)'
|
||||
version_cls: type[Version] = Version
|
||||
env: Mapping[str, str] | None = None
|
||||
also_catch: tuple[type[BaseException], ...] = ()
|
||||
|
||||
def version(self) -> Version:
|
||||
"""Return the installed version of the program.
|
||||
|
||||
Raises:
|
||||
MissingDependencyError: if the program cannot be found or its
|
||||
version string cannot be parsed.
|
||||
"""
|
||||
raw = get_version(
|
||||
self.program,
|
||||
version_arg=self.version_arg,
|
||||
regex=self.version_regex,
|
||||
env=self.env,
|
||||
)
|
||||
return self.version_cls(raw)
|
||||
|
||||
def available(self) -> bool:
|
||||
"""Return whether a usable version of the program is installed."""
|
||||
try:
|
||||
self.version()
|
||||
except MissingDependencyError:
|
||||
return False
|
||||
except self.also_catch:
|
||||
return False
|
||||
return True
|
||||
@@ -16,6 +16,7 @@ from subprocess import PIPE, CalledProcessError
|
||||
from packaging.version import Version
|
||||
from PIL import Image, UnidentifiedImageError
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import (
|
||||
ColorConversionNeededError,
|
||||
InputFileError,
|
||||
@@ -23,7 +24,7 @@ from ocrmypdf.exceptions import (
|
||||
)
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
||||
from ocrmypdf.subprocess import run, run_polling_stderr
|
||||
|
||||
COLOR_CONVERSION_STRATEGIES = frozenset(
|
||||
[
|
||||
@@ -69,11 +70,19 @@ class DuplicateFilter(logging.Filter):
|
||||
return True
|
||||
|
||||
|
||||
log.addFilter(DuplicateFilter(log))
|
||||
PROBE = ToolProbe(program=GS)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version(GS))
|
||||
def _ensure_log_filter_installed() -> None:
|
||||
"""Idempotently attach the duplicate-suppressing filter to the GS logger.
|
||||
|
||||
Called at the top of each work function so the filter is present in the
|
||||
main process *and* in any subprocess worker that calls Ghostscript.
|
||||
"""
|
||||
if not any(isinstance(f, DuplicateFilter) for f in log.filters):
|
||||
log.addFilter(DuplicateFilter(log))
|
||||
|
||||
|
||||
def _gs_error_reported(stream) -> bool:
|
||||
@@ -123,6 +132,7 @@ def rasterize_pdf(
|
||||
use_cropbox: If True, rasterize the CropBox instead of MediaBox.
|
||||
Default is False (use MediaBox).
|
||||
"""
|
||||
_ensure_log_filter_installed()
|
||||
raster_dpi = raster_dpi.round(6)
|
||||
if not page_dpi:
|
||||
page_dpi = raster_dpi
|
||||
@@ -140,6 +150,19 @@ def rasterize_pdf(
|
||||
else:
|
||||
effective_dpi = raster_dpi
|
||||
|
||||
# Anti-alias text and vector graphics when rendering to a contone device.
|
||||
# Ghostscript 10.x renders aliased glyphs that OCR frequently misreads as
|
||||
# extra word breaks; anti-aliasing empirically improves OCR accuracy on the
|
||||
# Ghostscript path, especially for small fonts at moderate DPI (#1439).
|
||||
# The 1-bit mono devices do not accept alpha bits (older Ghostscript
|
||||
# rejects them) and pngmonod performs its own anti-aliased downscaling.
|
||||
mono_devices = (GhostscriptRasterDevice.PNGMONO, GhostscriptRasterDevice.PNGMONOD)
|
||||
antialias_args = (
|
||||
[]
|
||||
if raster_device in mono_devices
|
||||
else ['-dTextAlphaBits=4', '-dGraphicsAlphaBits=4']
|
||||
)
|
||||
|
||||
args_gs = (
|
||||
[
|
||||
GS,
|
||||
@@ -152,6 +175,7 @@ def rasterize_pdf(
|
||||
f'-dLastPage={pageno}',
|
||||
f'-r{effective_dpi.x:f}x{effective_dpi.y:f}',
|
||||
]
|
||||
+ antialias_args
|
||||
+ (['-dUseCropBox'] if use_cropbox else [])
|
||||
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
||||
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||
@@ -268,11 +292,14 @@ def generate_pdfa(
|
||||
*,
|
||||
compression: str,
|
||||
color_conversion_strategy: str,
|
||||
jpeg_quality: int | None = None,
|
||||
jpeg_maxdpi: int | None = None,
|
||||
pdf_version: str = '1.5',
|
||||
pdfa_part: str = '2',
|
||||
progressbar_class=None,
|
||||
stop_on_error: bool = False,
|
||||
):
|
||||
_ensure_log_filter_installed()
|
||||
# Ghostscript's compression is all or nothing. We can either force all images
|
||||
# to JPEG, force all to Flate/PNG, or let it decide how to encode the images.
|
||||
# In most case it's best to let it decide.
|
||||
@@ -286,6 +313,11 @@ def generate_pdfa(
|
||||
]
|
||||
elif compression == 'lossless':
|
||||
compression_args = [
|
||||
# Re-encoding an existing JPEG with a lossless codec only inflates
|
||||
# its size: the lossy data is already baked in, so there is nothing
|
||||
# to gain. Pass JPEGs through untouched and apply lossless (Flate)
|
||||
# encoding only to images that are not already JPEG.
|
||||
"-dPassThroughJPEGImages=true",
|
||||
"-dAutoFilterColorImages=false",
|
||||
"-dColorImageFilter=/FlateEncode",
|
||||
"-dAutoFilterGrayImages=false",
|
||||
@@ -307,6 +339,35 @@ def generate_pdfa(
|
||||
# Windows has lots of fatal "permission denied" errors
|
||||
stop_on_error = False
|
||||
|
||||
# `-dJPEGQ=N` tells Ghostscript to use a JPEG quality of N, IF it decides
|
||||
# to transcode an image to JPEG. When there are existing JPEG images,
|
||||
# Ghostscript uses passthrough mode, so the quality level is not changed.
|
||||
# OCRmyPDF's optimizer separately uses the `--jpeg-quality` command line
|
||||
# option to potentially re-encode JPEG images, regardless of whether
|
||||
# Ghostscript decided to transcode them to JPEG or not.
|
||||
# `jpeg_quality=0` is meaningful to Ghostscript (maximum compression), so
|
||||
# only fall back to the default when the value is None.
|
||||
effective_jpeg_quality = jpeg_quality if jpeg_quality is not None else 95
|
||||
|
||||
# Downsampling images is a blunt-force way to reduce file size and almost
|
||||
# always degrades quality more than lowering JPEG quality at the original
|
||||
# resolution. We expose this for users with very specific needs (e.g.
|
||||
# producing very small files for screen-only viewing); the optimizer is
|
||||
# usually a better choice.
|
||||
downsample_args: list[str] = []
|
||||
if jpeg_maxdpi is not None:
|
||||
downsample_args = [
|
||||
"-dDownsampleColorImages=true",
|
||||
"-dColorImageDownsampleThreshold=1.0",
|
||||
"-dDownsampleGrayImages=true",
|
||||
"-dGrayImageDownsampleThreshold=1.0",
|
||||
"-dDownsampleMonoImages=true",
|
||||
"-dMonoImageDownsampleThreshold=1.0",
|
||||
f"-dColorImageResolution={jpeg_maxdpi}",
|
||||
f"-dGrayImageResolution={jpeg_maxdpi}",
|
||||
f"-dMonoImageResolution={jpeg_maxdpi}",
|
||||
]
|
||||
|
||||
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
||||
# is set; see:
|
||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699392
|
||||
@@ -323,8 +384,9 @@ def generate_pdfa(
|
||||
]
|
||||
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||
+ compression_args
|
||||
+ downsample_args
|
||||
+ [
|
||||
"-dJPEGQ=95",
|
||||
f"-dJPEGQ={effective_jpeg_quality}", # See note above on JPEG quality
|
||||
"-dSubsetFonts=false", # Prevents GS from messing up some encodings
|
||||
f"-dPDFA={pdfa_part}",
|
||||
"-dPDFACompatibilityPolicy=1",
|
||||
@@ -362,4 +424,8 @@ def generate_pdfa(
|
||||
for part in stderr.split('****'):
|
||||
log.error(part)
|
||||
if _gs_devicen_reported(stderr):
|
||||
raise ColorConversionNeededError()
|
||||
# Ghostscript could not normalize the DeviceN colorspace for PDF/A,
|
||||
# even if the user requested a conversion strategy. The output is
|
||||
# liable to render blank in some viewers, so raise regardless of the
|
||||
# strategy and tailor the guidance to what was attempted.
|
||||
raise ColorConversionNeededError(color_conversion_strategy)
|
||||
|
||||
@@ -9,21 +9,23 @@ from subprocess import PIPE, CalledProcessError
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
_PROBE = ToolProbe(program='jbig2', version_regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
try:
|
||||
version = get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
return _PROBE.version()
|
||||
except CalledProcessError as e:
|
||||
# TeX Live for Windows provides an incompatible jbig2.EXE which may
|
||||
# be on the PATH.
|
||||
raise MissingDependencyError('jbig2enc') from e
|
||||
return Version(version)
|
||||
|
||||
|
||||
def available():
|
||||
def available() -> bool:
|
||||
try:
|
||||
version()
|
||||
except MissingDependencyError:
|
||||
|
||||
@@ -8,22 +8,12 @@ from __future__ import annotations
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE
|
||||
|
||||
from packaging.version import Version
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('pngquant', regex=r'(\d+(\.\d+)*).*'))
|
||||
|
||||
|
||||
def available():
|
||||
try:
|
||||
version()
|
||||
except MissingDependencyError:
|
||||
return False
|
||||
return True
|
||||
PROBE = ToolProbe(program='pngquant', version_regex=r'(\d+(\.\d+)*).*')
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max: int):
|
||||
|
||||
@@ -17,13 +17,14 @@ from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import (
|
||||
MissingDependencyError,
|
||||
SubprocessOutputError,
|
||||
TesseractConfigError,
|
||||
)
|
||||
from ocrmypdf.pluginspec import OrientationConfidence
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -115,8 +116,13 @@ class TesseractVersion(Version):
|
||||
)
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return TesseractVersion(get_version('tesseract', regex=r'tesseract\s(.+)'))
|
||||
PROBE = ToolProbe(
|
||||
program='tesseract',
|
||||
version_regex=r'tesseract\s(.+)',
|
||||
version_cls=TesseractVersion,
|
||||
)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def has_thresholding() -> bool:
|
||||
@@ -287,12 +293,14 @@ def tesseract_log_output(stream: bytes) -> None:
|
||||
|
||||
lines = text.splitlines()
|
||||
for line in lines:
|
||||
if line.startswith(
|
||||
("Tesseract Open Source", "Warning in pixReadMem")
|
||||
):
|
||||
if line.startswith(("Tesseract Open Source", "Warning in pixReadMem")):
|
||||
continue
|
||||
elif 'diacritics' in line:
|
||||
tlog.warning("lots of diacritics - possibly poor OCR")
|
||||
# Surface the raw Tesseract message at debug level so users can see
|
||||
# exactly what Tesseract reported (e.g. the affected count) without
|
||||
# losing the interpreted hint above (#1566).
|
||||
tlog.debug(line.strip())
|
||||
elif line.startswith('OSD: Weak margin'):
|
||||
tlog.warning("unsure about page orientation")
|
||||
elif 'Error in pixScanForForeground' in line:
|
||||
@@ -309,6 +317,23 @@ def tesseract_log_output(stream: bytes) -> None:
|
||||
tlog.warning(line.strip())
|
||||
elif 'read_params_file' in line.lower():
|
||||
tlog.error(line.strip())
|
||||
# Tesseract emits "read_params_file: Can't open <name>" when it
|
||||
# cannot locate a config file (e.g. 'hocr', 'txt') in its
|
||||
# tessdata configs/ directory, then exits 0 without producing
|
||||
# the requested output. Promote to a hard error so the user
|
||||
# sees the root cause instead of a downstream FileNotFoundError.
|
||||
if "Can't open" in line:
|
||||
missing = line.split("Can't open", 1)[1].strip()
|
||||
else:
|
||||
missing = line.strip()
|
||||
raise TesseractConfigError(
|
||||
f"Tesseract cannot open its config file '{missing}'. "
|
||||
"This usually means Tesseract is installed but its config "
|
||||
"files are missing from the tessdata configs/ directory. "
|
||||
"On Debian/Ubuntu, ensure the 'tesseract-ocr' package is "
|
||||
"fully installed. If you set TESSDATA_PREFIX, verify its "
|
||||
"configs/ subdirectory contains the required files."
|
||||
)
|
||||
else:
|
||||
tlog.info(line.strip())
|
||||
|
||||
@@ -389,6 +414,12 @@ def generate_hocr(
|
||||
raise SubprocessOutputError() from e
|
||||
else:
|
||||
tesseract_log_output(stdout)
|
||||
if not output_hocr.exists():
|
||||
raise SubprocessOutputError(
|
||||
"Tesseract exited successfully but did not produce the "
|
||||
f"expected hOCR output at {output_hocr}. Tesseract output:\n"
|
||||
+ (stdout.decode(errors='replace') if stdout else '(empty)')
|
||||
)
|
||||
# The sidecar text file will get the suffix .txt; rename it to
|
||||
# whatever caller wants it named
|
||||
with suppress(FileNotFoundError):
|
||||
@@ -457,6 +488,12 @@ def generate_pdf(
|
||||
stdout = p.stdout
|
||||
with suppress(FileNotFoundError):
|
||||
prefix.with_suffix('.txt').replace(output_text)
|
||||
if not output_pdf.exists():
|
||||
raise SubprocessOutputError(
|
||||
"Tesseract exited successfully but did not produce the "
|
||||
f"expected PDF output at {output_pdf}. Tesseract output:\n"
|
||||
+ (stdout.decode(errors='replace') if stdout else '(empty)')
|
||||
)
|
||||
except TimeoutExpired:
|
||||
page_timedout(timeout)
|
||||
use_skip_page(output_pdf, output_text)
|
||||
|
||||
@@ -14,11 +14,11 @@ from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
from packaging.version import Version
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import SubprocessOutputError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
# unpaper documentation:
|
||||
# https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md
|
||||
@@ -46,8 +46,9 @@ class UnpaperImageTooLargeError(Exception):
|
||||
super().__init__(self.message)
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('unpaper', regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)'))
|
||||
PROBE = ToolProbe(program='unpaper', version_regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)')
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
@contextmanager
|
||||
|
||||
@@ -11,10 +11,9 @@ from pathlib import Path
|
||||
from subprocess import PIPE
|
||||
from typing import NamedTuple
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -27,18 +26,13 @@ class ValidationResult(NamedTuple):
|
||||
message: str
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
"""Get verapdf version."""
|
||||
return Version(get_version('verapdf', regex=r'veraPDF (\d+(\.\d+)*)'))
|
||||
|
||||
|
||||
def available() -> bool:
|
||||
"""Check if verapdf is available."""
|
||||
try:
|
||||
version()
|
||||
except (MissingDependencyError, OSError):
|
||||
return False
|
||||
return True
|
||||
PROBE = ToolProbe(
|
||||
program='verapdf',
|
||||
version_regex=r'veraPDF (\d+(\.\d+)*)',
|
||||
also_catch=(OSError,),
|
||||
)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def output_type_to_flavour(output_type: str) -> str:
|
||||
|
||||
+104
-3
@@ -211,6 +211,95 @@ def strip_invisible_text(pdf: Pdf, page: Page):
|
||||
page.Contents = Stream(pdf, content_stream)
|
||||
|
||||
|
||||
def discard_text_search_index(pdf: Pdf) -> bool:
|
||||
"""Discard an embedded Adobe full-text search index from the catalog.
|
||||
|
||||
Adobe Acrobat can embed a full-text search index in the document catalog at
|
||||
``/Root/PieceInfo/SearchIndex``. It is built from the page text, and only
|
||||
Acrobat reads it; other viewers ignore it and search the text on the fly.
|
||||
Any change to the PDF invalidates the index, so once OCRmyPDF rewrites the
|
||||
document (editing the text layer, rasterizing, optimizing) a retained index
|
||||
would be stale and return incorrect search results in Acrobat. We cannot
|
||||
update this vendor-private data, so we discard it; modern viewers rebuild a
|
||||
search index on demand. Returns True if the catalog was modified.
|
||||
"""
|
||||
try:
|
||||
pieceinfo = pdf.Root.get(Name.PieceInfo)
|
||||
if not isinstance(pieceinfo, Dictionary) or Name.SearchIndex not in pieceinfo:
|
||||
return False
|
||||
del pieceinfo[Name.SearchIndex]
|
||||
log.debug(
|
||||
"Discarded embedded text search index "
|
||||
"(/Root/PieceInfo/SearchIndex) because the PDF was rewritten; "
|
||||
"it would otherwise be stale."
|
||||
)
|
||||
# Drop an empty PieceInfo rather than leave a husk behind.
|
||||
if len(pieceinfo) == 0:
|
||||
del pdf.Root.PieceInfo
|
||||
return True
|
||||
except (KeyError, TypeError, AttributeError):
|
||||
return False
|
||||
|
||||
|
||||
def discard_page_thumbnails(pdf: Pdf) -> int:
|
||||
"""Discard embedded per-page thumbnail images.
|
||||
|
||||
A page object may carry an optional ``/Thumb`` image XObject — a miniature
|
||||
rendering of the page (ISO 32000-2, 12.3.4). It is only a navigation aid and
|
||||
modern viewers generate page thumbnails on demand. OCRmyPDF alters page
|
||||
appearance (deskew, clean, rasterize, re-render) and plugins may edit pages
|
||||
arbitrarily, so any retained thumbnail would be stale and misrepresent its
|
||||
page. We discard them; viewers rebuild thumbnails as needed. Returns the
|
||||
number of thumbnails removed.
|
||||
"""
|
||||
removed = 0
|
||||
for page in pdf.pages:
|
||||
pageobj = page.obj
|
||||
if Name.Thumb in pageobj:
|
||||
del pageobj[Name.Thumb]
|
||||
removed += 1
|
||||
if removed:
|
||||
log.debug(
|
||||
"Discarded %d embedded page thumbnail(s) (/Thumb) because the PDF "
|
||||
"was rewritten; they would otherwise be stale.",
|
||||
removed,
|
||||
)
|
||||
return removed
|
||||
|
||||
|
||||
def discard_structure_tree(pdf: Pdf) -> bool:
|
||||
"""Discard the logical structure (tagged-PDF) tree from the document.
|
||||
|
||||
The structure tree (``/Root/StructTreeRoot``, ``/Root/MarkInfo``) maps
|
||||
marked content in the page content streams to semantic elements via MCIDs.
|
||||
When OCRmyPDF rasterizes pages (force) or strips and rewrites the text layer
|
||||
(redo), those MCIDs are destroyed or renumbered, leaving the tree dangling
|
||||
and inconsistent with the new content. We cannot rebuild it to match, so we
|
||||
discard it; the page-level ``/StructParents`` keys go too. Returns True if
|
||||
the catalog was modified.
|
||||
"""
|
||||
modified = False
|
||||
try:
|
||||
if Name.StructTreeRoot in pdf.Root:
|
||||
del pdf.Root.StructTreeRoot
|
||||
modified = True
|
||||
if Name.MarkInfo in pdf.Root:
|
||||
del pdf.Root.MarkInfo
|
||||
modified = True
|
||||
for page in pdf.pages:
|
||||
if Name.StructParents in page.obj:
|
||||
del page.obj[Name.StructParents]
|
||||
modified = True
|
||||
except (KeyError, TypeError, AttributeError):
|
||||
return modified
|
||||
if modified:
|
||||
log.debug(
|
||||
"Discarded the logical structure tree (/Root/StructTreeRoot) "
|
||||
"because the PDF was re-OCR'd; it would otherwise be stale."
|
||||
)
|
||||
return modified
|
||||
|
||||
|
||||
class OcrGrafter:
|
||||
"""Manages grafting text-only PDFs onto regular PDFs."""
|
||||
|
||||
@@ -253,6 +342,14 @@ class OcrGrafter:
|
||||
ocr_tree: OCR tree for fpdf2 renderer.
|
||||
autorotate_correction: Orientation correction in degrees (0, 90, 180, 270).
|
||||
"""
|
||||
if self.context.options.mode == ProcessingMode.strip_text:
|
||||
# Strip mode: remove the invisible OCR text layer in place without
|
||||
# rasterizing or grafting anything. Honor --pages if specified.
|
||||
options = self.context.options
|
||||
if not options.pages or pageno in options.pages:
|
||||
strip_invisible_text(self.pdf_base, self.pdf_base.pages[pageno])
|
||||
return
|
||||
|
||||
if ocr_output and ocr_tree:
|
||||
raise ValueError(
|
||||
'Cannot specify both ocr_output and ocr_tree for fpdf2 renderer'
|
||||
@@ -319,9 +416,9 @@ class OcrGrafter:
|
||||
|
||||
def finalize(self):
|
||||
# Can have hocr OR parsed pages OR neither (no OCR), but not both
|
||||
assert not (
|
||||
self.fpdf2_hocr_pages and self.fpdf2_parsed_pages
|
||||
), "Can't have both hocr and ocrtree pages"
|
||||
assert not (self.fpdf2_hocr_pages and self.fpdf2_parsed_pages), (
|
||||
"Can't have both hocr and ocrtree pages"
|
||||
)
|
||||
|
||||
if self.fpdf2_hocr_pages:
|
||||
# Render all pages with fpdf2, then graft
|
||||
@@ -331,6 +428,10 @@ class OcrGrafter:
|
||||
if self.fpdf2_parsed_pages:
|
||||
self._render_and_graft_fpdf2_pages()
|
||||
|
||||
discard_text_search_index(self.pdf_base)
|
||||
discard_page_thumbnails(self.pdf_base)
|
||||
if self.context.options.mode in (ProcessingMode.force, ProcessingMode.redo):
|
||||
discard_structure_tree(self.pdf_base)
|
||||
self.pdf_base.save(self.output_file)
|
||||
self.pdf_base.close()
|
||||
return self.output_file
|
||||
|
||||
@@ -88,8 +88,12 @@ def repair_docinfo_nuls(pdf):
|
||||
if isinstance(v, str) and b'\x00' in bytes(v):
|
||||
pdf.docinfo[k] = bytes(v).replace(b'\x00', b'')
|
||||
modified = True
|
||||
except TypeError:
|
||||
# TypeError can also be raised if dictionary items are unexpected types
|
||||
except (TypeError, UnicodeDecodeError):
|
||||
# TypeError: DocumentInfo is not a dictionary, or its items are
|
||||
# unexpected types.
|
||||
# UnicodeDecodeError: a DocumentInfo key or value contains bytes that
|
||||
# are not valid PDFDocEncoding/UTF-16, e.g. a Latin-1 /Name key such as
|
||||
# /Saks#e5r. Older pikepdf raised while iterating such a block (#1540).
|
||||
log.error("File contains a malformed DocumentInfo block - continuing anyway.")
|
||||
return modified
|
||||
|
||||
|
||||
@@ -43,12 +43,16 @@ class ProcessingMode(StrEnum):
|
||||
- ``force``: Rasterize all content and run OCR regardless of existing text
|
||||
- ``skip``: Skip OCR on pages that already have text
|
||||
- ``redo``: Re-OCR pages, stripping old invisible text layer
|
||||
- ``strip``: Remove the invisible OCR text layer in place; do not OCR
|
||||
"""
|
||||
|
||||
default = 'default'
|
||||
force = 'force'
|
||||
skip = 'skip'
|
||||
redo = 'redo'
|
||||
# User-facing value is '--mode strip'; the member is named strip_text to
|
||||
# avoid shadowing str.strip on this str-based enum.
|
||||
strip_text = 'strip'
|
||||
|
||||
|
||||
class TaggedPdfMode(StrEnum):
|
||||
@@ -65,8 +69,33 @@ class TaggedPdfMode(StrEnum):
|
||||
ignore = 'ignore'
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges: str) -> set[int]:
|
||||
"""Convert page range string to set of page numbers."""
|
||||
def _has_end_alias(ranges: str) -> bool:
|
||||
"""Return True if the page range string uses the ``end`` alias."""
|
||||
return 'end' in ranges.lower()
|
||||
|
||||
|
||||
def _resolve_page_token(token: str, total_pages: int | None) -> int:
|
||||
"""Convert a single page-number token to a 1-based integer.
|
||||
|
||||
The literal ``end`` (case-insensitive) is resolved to ``total_pages``. If
|
||||
``total_pages`` is None, an error is raised.
|
||||
"""
|
||||
if token.lower() == 'end':
|
||||
if total_pages is None:
|
||||
raise BadArgsError(
|
||||
"'end' was used in --pages but the total page count is not yet known"
|
||||
)
|
||||
return total_pages
|
||||
return int(token)
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges: str, total_pages: int | None = None) -> set[int]:
|
||||
"""Convert page range string to set of 0-based page numbers.
|
||||
|
||||
The token ``end`` (case-insensitive) is an alias for the last page of the
|
||||
document. It is resolved using ``total_pages``; if ``end`` appears in the
|
||||
string and ``total_pages`` is None, a :class:`BadArgsError` is raised.
|
||||
"""
|
||||
pages: list[int] = []
|
||||
page_groups = ranges.replace(' ', '').split(',')
|
||||
for group in page_groups:
|
||||
@@ -75,10 +104,15 @@ def _pages_from_ranges(ranges: str) -> set[int]:
|
||||
try:
|
||||
start, end = group.split('-')
|
||||
except ValueError:
|
||||
pages.append(int(group) - 1)
|
||||
try:
|
||||
pages.append(_resolve_page_token(group, total_pages) - 1)
|
||||
except ValueError:
|
||||
raise BadArgsError(f"invalid page number '{group}'") from None
|
||||
else:
|
||||
try:
|
||||
new_pages = list(range(int(start) - 1, int(end)))
|
||||
start_n = _resolve_page_token(start, total_pages)
|
||||
end_n = _resolve_page_token(end, total_pages)
|
||||
new_pages = list(range(start_n - 1, end_n))
|
||||
if not new_pages:
|
||||
raise BadArgsError(
|
||||
f"invalid page subrange '{start}-{end}'"
|
||||
@@ -332,11 +366,19 @@ class OcrOptions(BaseModel):
|
||||
@field_validator('pages')
|
||||
@classmethod
|
||||
def validate_pages_format(cls, v):
|
||||
"""Convert page ranges string to set of page numbers."""
|
||||
"""Convert page ranges string to set of page numbers.
|
||||
|
||||
If the string uses the ``end`` alias, the original string is preserved
|
||||
so that resolution can happen later, once the document's page count is
|
||||
known.
|
||||
"""
|
||||
if v is None:
|
||||
return v
|
||||
if isinstance(v, set):
|
||||
return v # Already processed
|
||||
if _has_end_alias(v):
|
||||
# Defer resolution until total page count is known
|
||||
return v
|
||||
|
||||
# Convert string ranges to set of page numbers
|
||||
return _pages_from_ranges(v)
|
||||
@@ -582,6 +624,13 @@ class OcrOptions(BaseModel):
|
||||
value = getattr(self, flat_name)
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
# Plugin-scoped fields that aren't in the central OcrOptions
|
||||
# registry: argparse stores them in extra_attrs under the
|
||||
# namespace_field name.
|
||||
elif flat_name in self.extra_attrs:
|
||||
value = self.extra_attrs[flat_name]
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
# Also check direct field name (for fields like jbig2_lossy)
|
||||
elif field_name in OcrOptions.model_fields:
|
||||
value = getattr(self, field_name)
|
||||
|
||||
@@ -0,0 +1,253 @@
|
||||
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
||||
# SPDX-FileCopyrightText: 2025 ajdlinux
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Validate and repair malformed page-boundary boxes.
|
||||
|
||||
A page's boundary boxes (``/MediaBox``, ``/CropBox``, ``/TrimBox``, ``/ArtBox``,
|
||||
``/BleedBox``) are sometimes malformed in ways that PDF readers tolerate but
|
||||
that crash or corrupt downstream processing. This module normalizes them in
|
||||
place following the PDF 2.0 specification (ISO 32000-2:2020):
|
||||
|
||||
- **Non-decimal coordinates** (§7.3.3): a coordinate written in exponential
|
||||
notation is invalid PDF number syntax and is stored by qpdf/pikepdf as a
|
||||
string. We coerce it back to a number (issue #1398).
|
||||
- **Reversed corners** (§7.9.5): a rectangle is "a pair of diagonally opposite
|
||||
corners"; ``[llx lly urx ury]`` is only the typical order. We normalize to
|
||||
``[min_x, min_y, max_x, max_y]`` (issue #1526).
|
||||
- **Sub-box outside the MediaBox** (§14.11.2): "If the bounds of the crop,
|
||||
trim, bleed or art box extends outside of the bounds of the media box, a
|
||||
processor shall treat the box as its intersection with the media box." We
|
||||
clamp to that intersection, or discard the sub-box (so it inherits the
|
||||
MediaBox) when the intersection is empty (issue #1400).
|
||||
|
||||
A rectangle is treated as empty when its width or height is ``<= 0``; PDF 2.0
|
||||
permits zero-dimension rectangles and defines no minimum page size, so no other
|
||||
size floor is imposed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
from collections.abc import Iterable, Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pikepdf
|
||||
from pikepdf import Name
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_SUBBOXES = ('CropBox', 'TrimBox', 'ArtBox', 'BleedBox')
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BoxRepair:
|
||||
"""A single change made to a page box.
|
||||
|
||||
Attributes:
|
||||
box: The box name, e.g. ``"CropBox"``.
|
||||
kind: One of ``"reordered"`` (reversed corners normalized; lossless),
|
||||
``"recoded"`` (non-numeric/exponential coordinate coerced),
|
||||
``"clamped"`` (sub-box clamped to the MediaBox), ``"discarded"``
|
||||
(sub-box removed because its MediaBox intersection was empty), or
|
||||
``"degenerate_mediabox"`` (MediaBox has zero width or height).
|
||||
"""
|
||||
|
||||
box: str
|
||||
kind: str
|
||||
|
||||
|
||||
def _read_box(values: Sequence) -> tuple[list[float], bool, bool] | None:
|
||||
"""Coerce a box array to floats and normalize corner order.
|
||||
|
||||
Returns ``(normalized_values, recoded, reordered)`` where ``recoded`` is
|
||||
True if any element needed string/exponential coercion and ``reordered`` is
|
||||
True if the corners were given in non-standard order. Returns None if the
|
||||
array is not four finite numbers.
|
||||
"""
|
||||
if len(values) != 4:
|
||||
return None
|
||||
nums: list[float] = []
|
||||
recoded = False
|
||||
for v in values:
|
||||
try:
|
||||
n = float(v)
|
||||
except (TypeError, ValueError):
|
||||
try:
|
||||
n = float(str(v))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
recoded = True
|
||||
if not math.isfinite(n):
|
||||
return None
|
||||
nums.append(n)
|
||||
x0, y0, x1, y1 = nums
|
||||
normalized = [min(x0, x1), min(y0, y1), max(x0, x1), max(y0, y1)]
|
||||
reordered = normalized != nums
|
||||
return normalized, recoded, reordered
|
||||
|
||||
|
||||
def coerce_box(values: Iterable) -> list[float]:
|
||||
"""Return box values coerced to floats with corner order normalized.
|
||||
|
||||
Robust against exponential/string coordinates and reversed corners, so
|
||||
callers that only need to read a box (e.g. dimension calculations) do not
|
||||
crash on malformed input. Falls back to best-effort per-element coercion if
|
||||
the array is not four numbers.
|
||||
"""
|
||||
values = list(values)
|
||||
result = _read_box(values)
|
||||
if result is not None:
|
||||
return result[0]
|
||||
coerced = []
|
||||
for v in values:
|
||||
try:
|
||||
coerced.append(float(v))
|
||||
except (TypeError, ValueError):
|
||||
coerced.append(float(str(v)))
|
||||
return coerced
|
||||
|
||||
|
||||
def _is_empty(box: Sequence[float]) -> bool:
|
||||
"""A rectangle is empty when its width or height is non-positive."""
|
||||
return (box[2] - box[0]) <= 0 or (box[3] - box[1]) <= 0
|
||||
|
||||
|
||||
def repair_page_boxes(page: pikepdf.Page) -> list[BoxRepair]:
|
||||
"""Validate and repair the boundary boxes of a single page, in place.
|
||||
|
||||
Returns the list of changes made (empty if the page was already valid).
|
||||
Only boxes that actually change are written back, so valid pages are left
|
||||
untouched. Performs no logging or I/O.
|
||||
"""
|
||||
repairs: list[BoxRepair] = []
|
||||
|
||||
# MediaBox is the reference rectangle; read it inheritance-aware.
|
||||
mediabox: list[float] | None = None
|
||||
try:
|
||||
mb_result = _read_box(list(page.mediabox.as_list()))
|
||||
except (AttributeError, KeyError, RuntimeError):
|
||||
mb_result = None
|
||||
if mb_result is not None:
|
||||
mediabox, recoded, reordered = mb_result
|
||||
if reordered:
|
||||
repairs.append(BoxRepair('MediaBox', 'reordered'))
|
||||
if recoded:
|
||||
repairs.append(BoxRepair('MediaBox', 'recoded'))
|
||||
if recoded or reordered:
|
||||
page.obj.MediaBox = pikepdf.Array(mediabox)
|
||||
if _is_empty(mediabox):
|
||||
repairs.append(BoxRepair('MediaBox', 'degenerate_mediabox'))
|
||||
mediabox = None # don't clamp against a degenerate reference
|
||||
|
||||
for box in _SUBBOXES:
|
||||
name = Name('/' + box)
|
||||
if name not in page.obj:
|
||||
continue
|
||||
try:
|
||||
sub_result = _read_box(list(page.obj[name]))
|
||||
except (TypeError, RuntimeError):
|
||||
continue
|
||||
if sub_result is None:
|
||||
continue
|
||||
values, recoded, reordered = sub_result
|
||||
if reordered:
|
||||
repairs.append(BoxRepair(box, 'reordered'))
|
||||
if recoded:
|
||||
repairs.append(BoxRepair(box, 'recoded'))
|
||||
if recoded or reordered:
|
||||
page.obj[name] = pikepdf.Array(values)
|
||||
|
||||
if mediabox is None:
|
||||
continue
|
||||
intersection = [
|
||||
max(values[0], mediabox[0]),
|
||||
max(values[1], mediabox[1]),
|
||||
min(values[2], mediabox[2]),
|
||||
min(values[3], mediabox[3]),
|
||||
]
|
||||
if _is_empty(intersection):
|
||||
del page.obj[name]
|
||||
repairs.append(BoxRepair(box, 'discarded'))
|
||||
elif intersection != values:
|
||||
page.obj[name] = pikepdf.Array(intersection)
|
||||
repairs.append(BoxRepair(box, 'clamped'))
|
||||
|
||||
return repairs
|
||||
|
||||
|
||||
# Per-kind log severity and message template ({box} is substituted).
|
||||
_KIND_MESSAGES: dict[str, tuple[int, str]] = {
|
||||
'discarded': (
|
||||
logging.WARNING,
|
||||
'{box} lies outside the MediaBox and was discarded; '
|
||||
'the full page will be shown',
|
||||
),
|
||||
'clamped': (
|
||||
logging.WARNING,
|
||||
'{box} extended beyond the MediaBox and was clamped to it',
|
||||
),
|
||||
'recoded': (
|
||||
logging.WARNING,
|
||||
'{box} used invalid (e.g. exponential) coordinates, which were reinterpreted',
|
||||
),
|
||||
'degenerate_mediabox': (
|
||||
logging.WARNING,
|
||||
'MediaBox has zero width or height and could not be repaired; '
|
||||
'output may be invalid',
|
||||
),
|
||||
'reordered': (
|
||||
logging.DEBUG,
|
||||
'{box} corners were reversed and have been normalized',
|
||||
),
|
||||
}
|
||||
|
||||
# Kinds that change page appearance and warrant manual review of the output.
|
||||
_INSPECT_KINDS = frozenset({'discarded', 'clamped', 'recoded'})
|
||||
_INSPECT = ' Please visually inspect the output PDF.'
|
||||
|
||||
|
||||
def _format_pages(pagenos: Iterable[int]) -> str:
|
||||
"""Format 0-based page numbers as a compact 1-based range string."""
|
||||
nums = sorted(p + 1 for p in pagenos)
|
||||
ranges: list[tuple[int, int]] = []
|
||||
start = prev = nums[0]
|
||||
for n in nums[1:]:
|
||||
if n == prev + 1:
|
||||
prev = n
|
||||
continue
|
||||
ranges.append((start, prev))
|
||||
start = prev = n
|
||||
ranges.append((start, prev))
|
||||
return ', '.join(f'{a}' if a == b else f'{a}-{b}' for a, b in ranges)
|
||||
|
||||
|
||||
def summarize_box_repairs(
|
||||
repairs_by_page: Mapping[int, Sequence[BoxRepair]],
|
||||
) -> list[tuple[int, str]]:
|
||||
"""Aggregate per-page repairs into ``(log_level, message)`` pairs.
|
||||
|
||||
Repairs are grouped by ``(kind, box)`` so a defect shared across many pages
|
||||
yields a single message listing the affected pages, rather than one message
|
||||
per page.
|
||||
"""
|
||||
groups: dict[tuple[str, str], set[int]] = {}
|
||||
for pageno, repairs in repairs_by_page.items():
|
||||
for repair in repairs:
|
||||
groups.setdefault((repair.kind, repair.box), set()).add(pageno)
|
||||
|
||||
messages: list[tuple[int, str]] = []
|
||||
for (kind, box), pages in sorted(groups.items()):
|
||||
level, template = _KIND_MESSAGES[kind]
|
||||
text = f'Page(s) {_format_pages(pages)}: {template.format(box=box)}.'
|
||||
if kind in _INSPECT_KINDS:
|
||||
text += _INSPECT
|
||||
messages.append((level, text))
|
||||
return messages
|
||||
|
||||
|
||||
def log_box_repairs(repairs_by_page: Mapping[int, Sequence[BoxRepair]]) -> None:
|
||||
"""Emit aggregated log messages for the repairs made across all pages."""
|
||||
for level, message in summarize_box_repairs(repairs_by_page):
|
||||
log.log(level, message)
|
||||
+155
-54
@@ -29,22 +29,28 @@ from ocrmypdf._exec import unpaper
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._metadata import repair_docinfo_nuls
|
||||
from ocrmypdf._options import OcrOptions, ProcessingMode, TaggedPdfMode
|
||||
from ocrmypdf._pageboxes import log_box_repairs, repair_page_boxes
|
||||
from ocrmypdf._stdoutprotect import get_protected_stdout_fd
|
||||
from ocrmypdf.exceptions import (
|
||||
ColorConversionNeededError,
|
||||
DigitalSignatureError,
|
||||
DpiError,
|
||||
EncryptedPdfError,
|
||||
InputFileError,
|
||||
NonEmbeddedFontsError,
|
||||
PriorOcrFoundError,
|
||||
SubprocessOutputError,
|
||||
TaggedPDFError,
|
||||
UnsupportedImageFormatError,
|
||||
)
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
|
||||
from ocrmypdf.pdfa import (
|
||||
file_claims_pdfa,
|
||||
find_nonembedded_cid_fonts,
|
||||
generate_pdfa_ps,
|
||||
speculative_pdfa_conversion,
|
||||
)
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, FloatRect, PageInfo, PdfInfo
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, FloatRect, Ink, PageInfo, PdfInfo
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice, OrientationConfidence
|
||||
|
||||
try:
|
||||
@@ -116,8 +122,7 @@ def triage_image_file(input_file: Path, output_file: Path, options: OcrOptions)
|
||||
|
||||
if im.mode in ('RGBA', 'LA'):
|
||||
raise UnsupportedImageFormatError(
|
||||
"The input image has an alpha channel. Remove the alpha "
|
||||
"channel first."
|
||||
"The input image has an alpha channel. Remove the alpha channel first."
|
||||
)
|
||||
|
||||
if 'iccprofile' not in im.info:
|
||||
@@ -175,6 +180,12 @@ def triage(
|
||||
)
|
||||
try:
|
||||
with pikepdf.open(input_file) as pdf:
|
||||
repairs_by_page = {
|
||||
n: repairs
|
||||
for n, page in enumerate(pdf.pages)
|
||||
if (repairs := repair_page_boxes(page))
|
||||
}
|
||||
log_box_repairs(repairs_by_page)
|
||||
pdf.save(output_file)
|
||||
except pikepdf.PdfError as e:
|
||||
raise InputFileError() from e
|
||||
@@ -250,12 +261,15 @@ def validate_pdfinfo_options(context: PdfContext) -> None:
|
||||
"image of the form and all filled form fields. The output PDF "
|
||||
"will be 'flattened' and will no longer be fillable."
|
||||
)
|
||||
if pdfinfo.is_tagged:
|
||||
if pdfinfo.is_tagged or pdfinfo.has_structure_tree:
|
||||
log.warning(
|
||||
"This PDF is marked as a Tagged PDF. This often indicates "
|
||||
"that the PDF was generated from an office document and does "
|
||||
"not need OCR. PDF pages processed by OCRmyPDF may not be "
|
||||
"tagged correctly."
|
||||
"This PDF contains structural markup (it is a Tagged PDF or "
|
||||
"carries a logical structure tree). This often indicates that the "
|
||||
"PDF was generated from an office document or is otherwise born "
|
||||
"digital, and does not need OCR. OCRmyPDF cannot rebuild this "
|
||||
"structure to match new text, so any page it re-OCRs with "
|
||||
"--force-ocr or --redo-ocr will have its structural markup "
|
||||
"discarded."
|
||||
)
|
||||
if (
|
||||
options.tagged_pdf_mode == TaggedPdfMode.default
|
||||
@@ -325,6 +339,11 @@ def is_ocr_required(page_context: PageContext) -> bool:
|
||||
pageinfo = page_context.pageinfo
|
||||
options = page_context.options
|
||||
|
||||
if options.mode == ProcessingMode.strip_text:
|
||||
# Strip mode removes the OCR text layer in place; it never rasterizes
|
||||
# or runs OCR. The stripping happens in OcrGrafter.graft_page.
|
||||
return False
|
||||
|
||||
ocr_required = True
|
||||
|
||||
if options.pages and pageinfo.pageno not in options.pages:
|
||||
@@ -508,6 +527,49 @@ def calculate_raster_dpi(page_context: PageContext):
|
||||
return canvas_dpi, page_dpi
|
||||
|
||||
|
||||
def _select_raster_device(pageinfo: PageInfo) -> GhostscriptRasterDevice:
|
||||
"""Choose the minimum raster device that preserves the page's color depth.
|
||||
|
||||
The device escalates from 1-bit mono through grayscale, indexed, and full
|
||||
color as required by the page's images, image masks, and vector content.
|
||||
Image masks are painted with the current fill color, so a mask painted in
|
||||
gray or color escalates the device even though the mask itself is 1-bit.
|
||||
"""
|
||||
colorspaces = [
|
||||
GhostscriptRasterDevice.PNGMONOD,
|
||||
GhostscriptRasterDevice.PNGGRAY,
|
||||
GhostscriptRasterDevice.PNG256,
|
||||
GhostscriptRasterDevice.PNG16M,
|
||||
]
|
||||
device_idx = 0
|
||||
|
||||
def at_least(colorspace):
|
||||
return max(device_idx, colorspaces.index(colorspace))
|
||||
|
||||
for image in pageinfo.images:
|
||||
if image.type_ == 'stencil':
|
||||
# The fill color used to paint the mask, not the 1-bit mask data,
|
||||
# determines the color depth OCR needs.
|
||||
if image.ink == Ink.color:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
elif image.ink == Ink.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
continue
|
||||
if image.bpc > 1:
|
||||
if image.color == Colorspace.index:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG256)
|
||||
elif image.color == Colorspace.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
else:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
if pageinfo.has_vector:
|
||||
log.debug(f"Page has vector content, using {GhostscriptRasterDevice.PNG16M}")
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
return colorspaces[device_idx]
|
||||
|
||||
|
||||
def rasterize(
|
||||
input_file: Path,
|
||||
page_context: PageContext,
|
||||
@@ -529,39 +591,13 @@ def rasterize(
|
||||
Returns:
|
||||
Path: The output PNG file path.
|
||||
"""
|
||||
colorspaces = [
|
||||
GhostscriptRasterDevice.PNGMONO,
|
||||
GhostscriptRasterDevice.PNGGRAY,
|
||||
GhostscriptRasterDevice.PNG256,
|
||||
GhostscriptRasterDevice.PNG16M,
|
||||
]
|
||||
device_idx = 0
|
||||
|
||||
if remove_vectors is None:
|
||||
remove_vectors = page_context.options.remove_vectors
|
||||
|
||||
output_file = page_context.get_path(f'rasterize{output_tag}.png')
|
||||
pageinfo = page_context.pageinfo
|
||||
|
||||
def at_least(colorspace):
|
||||
return max(device_idx, colorspaces.index(colorspace))
|
||||
|
||||
for image in pageinfo.images:
|
||||
if image.type_ != 'image':
|
||||
continue # ignore masks
|
||||
if image.bpc > 1:
|
||||
if image.color == Colorspace.index:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG256)
|
||||
elif image.color == Colorspace.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
else:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
if pageinfo.has_vector:
|
||||
log.debug(f"Page has vector content, using {GhostscriptRasterDevice.PNG16M}")
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
device = colorspaces[device_idx]
|
||||
device = _select_raster_device(pageinfo)
|
||||
|
||||
log.debug(
|
||||
f"Rasterize with {device}, rotation {correction}, mediabox {pageinfo.mediabox}"
|
||||
@@ -945,6 +981,12 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext) -
|
||||
# pikepdf can deal with this, but we make the world a better place by
|
||||
# stamping them out as soon as possible.
|
||||
with pikepdf.open(input_pdf) as pdf_file:
|
||||
# Ghostscript would substitute and re-embed any non-embedded CID font to
|
||||
# satisfy PDF/A, corrupting CJK text (e.g. an Acrobat OCR layer) in the
|
||||
# process. Refuse rather than silently damage the user's text layer.
|
||||
nonembedded = find_nonembedded_cid_fonts(pdf_file)
|
||||
if nonembedded:
|
||||
raise NonEmbeddedFontsError(nonembedded)
|
||||
if repair_docinfo_nuls(pdf_file):
|
||||
pdf_file.save(fix_docinfo_file)
|
||||
else:
|
||||
@@ -1038,14 +1080,46 @@ def try_speculative_pdfa(input_pdf: Path, context: PdfContext) -> Path | None:
|
||||
return None
|
||||
|
||||
|
||||
def _ghostscript_pdfa_fallback(input_pdf: Path, context: PdfContext) -> Path | None:
|
||||
"""Best-effort PDF/A conversion via Ghostscript for 'auto' output type.
|
||||
|
||||
Returns the converted PDF/A path, or None if Ghostscript is unavailable,
|
||||
fails, or cannot produce valid PDF/A. Never raises: 'auto' mode degrades to
|
||||
a regular PDF instead of erroring or emitting corrupted output.
|
||||
|
||||
Args:
|
||||
input_pdf: Path to the PDF to convert.
|
||||
context: The PDF context.
|
||||
"""
|
||||
from ocrmypdf._exec import ghostscript
|
||||
|
||||
if not ghostscript.available():
|
||||
return None
|
||||
try:
|
||||
ps_stub = generate_postscript_stub(context)
|
||||
gs_out = convert_to_pdfa(input_pdf, ps_stub, context)
|
||||
except (
|
||||
SubprocessOutputError,
|
||||
ColorConversionNeededError,
|
||||
NonEmbeddedFontsError,
|
||||
) as e:
|
||||
log.info('Auto mode: Ghostscript could not produce PDF/A (%s)', e)
|
||||
return None
|
||||
if not file_claims_pdfa(gs_out)['pass']:
|
||||
log.info('Auto mode: Ghostscript output is not valid PDF/A')
|
||||
return None
|
||||
return gs_out
|
||||
|
||||
|
||||
def try_auto_pdfa(input_pdf: Path, context: PdfContext) -> tuple[Path, str]:
|
||||
"""Best-effort PDF/A for 'auto' output type.
|
||||
|
||||
This function attempts to produce PDF/A without requiring Ghostscript:
|
||||
1. If verapdf is available, tries speculative conversion with validation
|
||||
2. Without verapdf, passes through as PDF/A if safe (input already PDF/A
|
||||
or force-ocr was used)
|
||||
3. Falls back to regular PDF if neither condition is met
|
||||
Order of attempts, first success wins:
|
||||
1. Non-embedded CID fonts -> regular PDF (Ghostscript would corrupt them).
|
||||
2. Speculative conversion validated by verapdf (no Ghostscript).
|
||||
3. Without verapdf, pass through if already PDF/A or rebuilt with force-ocr.
|
||||
4. Ghostscript conversion (best-effort; failures fall through).
|
||||
5. Regular PDF if none of the above produced PDF/A.
|
||||
|
||||
Args:
|
||||
input_pdf: Path to the PDF to convert
|
||||
@@ -1057,25 +1131,42 @@ def try_auto_pdfa(input_pdf: Path, context: PdfContext) -> tuple[Path, str]:
|
||||
"""
|
||||
from ocrmypdf._exec import verapdf
|
||||
|
||||
# If verapdf available, try speculative conversion with validation
|
||||
# Non-embedded CID fonts cannot be made PDF/A without Ghostscript font
|
||||
# substitution that corrupts CID/CJK text. Rather than risk an existing
|
||||
# text layer, downgrade to a regular PDF (the same outcome as any other
|
||||
# case where best-effort PDF/A is not achievable).
|
||||
with pikepdf.open(input_pdf) as pdf_file:
|
||||
nonembedded = find_nonembedded_cid_fonts(pdf_file)
|
||||
if nonembedded:
|
||||
log.info(
|
||||
"Auto mode: input has non-embedded CID fonts (%s) that cannot be "
|
||||
"converted to PDF/A without corrupting the text; outputting a "
|
||||
"regular PDF. Use --output-type pdf to select this explicitly.",
|
||||
', '.join(sorted(nonembedded)),
|
||||
)
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
# Cheap path: speculative conversion validated by verapdf (no Ghostscript).
|
||||
if verapdf.available():
|
||||
result = try_speculative_pdfa(input_pdf, context)
|
||||
if result is not None:
|
||||
return (result, 'pdfa')
|
||||
# verapdf validation failed - fall through to regular PDF
|
||||
log.info(
|
||||
'Auto mode: speculative PDF/A validation failed, outputting regular PDF'
|
||||
)
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
# Without verapdf, check if we can pass through as PDF/A
|
||||
if _is_safe_pdfa(input_pdf, context.options):
|
||||
# Pass through as-is (no modifications needed)
|
||||
log.info('Auto mode: speculative PDF/A validation failed')
|
||||
elif _is_safe_pdfa(input_pdf, context.options):
|
||||
# No verapdf, but the input is already PDF/A or was rebuilt with
|
||||
# --force-ocr, so we can pass it through without Ghostscript.
|
||||
log.info('Auto mode: passing through as PDF/A (input already compliant)')
|
||||
return (input_pdf, 'pdfa')
|
||||
|
||||
# Fall through to regular PDF
|
||||
log.info('Auto mode: no verapdf available and input is not PDF/A, outputting PDF')
|
||||
# Fall back to Ghostscript to produce real PDF/A (v16 behavior). Best-effort:
|
||||
# if Ghostscript is unavailable or cannot safely produce PDF/A, keep a
|
||||
# regular PDF rather than error.
|
||||
gs_out = _ghostscript_pdfa_fallback(input_pdf, context)
|
||||
if gs_out is not None:
|
||||
log.info('Auto mode: produced PDF/A via Ghostscript')
|
||||
return (gs_out, 'pdfa')
|
||||
|
||||
log.info('Auto mode: could not produce PDF/A, outputting regular PDF')
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
|
||||
@@ -1247,8 +1338,18 @@ def copy_final(
|
||||
log.debug('%s -> %s', input_file, output_file)
|
||||
with input_file.open('rb') as input_stream:
|
||||
if output_file == '-':
|
||||
copyfileobj(input_stream, sys.stdout.buffer) # type: ignore[misc]
|
||||
sys.stdout.flush()
|
||||
fd = get_protected_stdout_fd()
|
||||
if fd is not None:
|
||||
# Stdout protection is active: write to the preserved real
|
||||
# stdout. dup the saved fd so the with-block's close() does not
|
||||
# close our long-lived descriptor.
|
||||
with os.fdopen(os.dup(fd), 'wb') as stdout_stream:
|
||||
copyfileobj(input_stream, stdout_stream)
|
||||
stdout_stream.flush()
|
||||
else:
|
||||
# No protection installed (e.g. plain API use): legacy behavior.
|
||||
copyfileobj(input_stream, sys.stdout.buffer) # type: ignore[misc]
|
||||
sys.stdout.flush()
|
||||
elif hasattr(output_file, 'writable'):
|
||||
output_stream = cast(BinaryIO, output_file)
|
||||
copyfileobj(input_stream, output_stream) # type: ignore[misc]
|
||||
|
||||
@@ -343,12 +343,17 @@ def setup_pipeline(
|
||||
|
||||
|
||||
def do_get_pdfinfo(pdf_path: Path, executor: Executor, options) -> PdfInfo:
|
||||
# Handle pages field - it might be a string that needs conversion
|
||||
# Handle pages field - it might be a string that needs conversion.
|
||||
# A string indicates the ``end`` alias was used and resolution was
|
||||
# deferred; we resolve it now using the document's actual page count.
|
||||
check_pages = options.pages
|
||||
if isinstance(check_pages, str):
|
||||
from ocrmypdf._options import _pages_from_ranges
|
||||
|
||||
check_pages = _pages_from_ranges(check_pages)
|
||||
with Pdf.open(pdf_path) as pdf:
|
||||
total_pages = len(pdf.pages)
|
||||
check_pages = _pages_from_ranges(check_pages, total_pages=total_pages)
|
||||
options.pages = check_pages
|
||||
|
||||
return get_pdfinfo(
|
||||
pdf_path,
|
||||
@@ -484,7 +489,7 @@ def postprocess(
|
||||
else:
|
||||
pdf_out = pdf_file
|
||||
if context.options.output_type == 'auto':
|
||||
# Best effort PDF/A - never uses Ghostscript
|
||||
# Best effort PDF/A - may use Ghostscript as a last resort
|
||||
pdf_out, actual_type = try_auto_pdfa(pdf_out, context)
|
||||
# Store actual output type for reporting
|
||||
context.options.extra_attrs['_actual_output_type'] = actual_type
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Protect the real standard output from corruption by stray writes.
|
||||
|
||||
When OCRmyPDF writes its final PDF to standard output (``ocrmypdf in.pdf -``),
|
||||
the bytes on stdout must be exactly the PDF and nothing else. Any accidental
|
||||
write to file descriptor 1 anywhere in the process -- from a third-party
|
||||
library, a plugin, or a stray ``print()`` -- would silently corrupt the output.
|
||||
|
||||
This module enforces that guarantee at the operating system level. It saves a
|
||||
private duplicate of the real stdout and points file descriptor 1 at standard
|
||||
error, so that anything that writes to stdout lands harmlessly on stderr. Only
|
||||
OCRmyPDF's final "produce the PDF" step writes to the preserved real stdout, via
|
||||
:func:`get_protected_stdout_fd`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
|
||||
_lock = threading.Lock()
|
||||
_saved_fd: int | None = None
|
||||
_active = False
|
||||
|
||||
|
||||
def protect_stdout() -> bool:
|
||||
"""Redirect file descriptor 1 to stderr and preserve the real stdout.
|
||||
|
||||
After this call, any write to file descriptor 1 -- including ``print()`` and
|
||||
writes from third-party C libraries -- is redirected to standard error and
|
||||
cannot corrupt the real standard output. The real stdout is preserved on a
|
||||
private file descriptor available from :func:`get_protected_stdout_fd`.
|
||||
|
||||
This mutates process-global state and affects the whole process. It must be
|
||||
called once, early, before any plugins are loaded or any worker
|
||||
process/thread is started, so that all of them inherit the redirected
|
||||
descriptor.
|
||||
|
||||
Returns:
|
||||
True if protection was installed (or was already active). False if
|
||||
stdout is not backed by a real OS file descriptor -- for example under
|
||||
a test harness that captures stdout -- in which case nothing is changed.
|
||||
"""
|
||||
global _saved_fd, _active
|
||||
with _lock:
|
||||
if _active:
|
||||
return True
|
||||
try:
|
||||
fd1 = sys.stdout.fileno()
|
||||
except (AttributeError, OSError, ValueError):
|
||||
# stdout is not backed by a real file descriptor (e.g. captured by
|
||||
# a test harness or replaced with an in-memory stream).
|
||||
return False
|
||||
try:
|
||||
sys.stdout.flush()
|
||||
saved = os.dup(fd1)
|
||||
os.dup2(2, fd1) # point stdout at stderr
|
||||
except OSError:
|
||||
return False
|
||||
_saved_fd = saved
|
||||
_active = True
|
||||
return True
|
||||
|
||||
|
||||
def get_protected_stdout_fd() -> int | None:
|
||||
"""Return the preserved real stdout file descriptor, or None if inactive."""
|
||||
return _saved_fd if _active else None
|
||||
|
||||
|
||||
def protected_stdout_isatty() -> bool | None:
|
||||
"""Whether the preserved real stdout is a terminal.
|
||||
|
||||
Returns None if protection is not active, in which case the caller should
|
||||
fall back to ``sys.stdout.isatty()``. When protection is active,
|
||||
``sys.stdout`` reports the terminal status of stderr (its descriptor was
|
||||
redirected), so this consults the saved real-stdout descriptor instead.
|
||||
"""
|
||||
if not _active or _saved_fd is None:
|
||||
return None
|
||||
return os.isatty(_saved_fd)
|
||||
@@ -17,8 +17,9 @@ import pikepdf
|
||||
|
||||
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
from ocrmypdf._exec import unpaper
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._options import OcrOptions, ProcessingMode
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager
|
||||
from ocrmypdf._stdoutprotect import protected_stdout_isatty
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
InputFileError,
|
||||
@@ -118,8 +119,36 @@ def check_options_preprocessing(options: OcrOptions) -> None:
|
||||
)
|
||||
|
||||
|
||||
def check_options_strip(options: OcrOptions) -> None:
|
||||
"""Reject options that cannot apply in strip mode.
|
||||
|
||||
``--mode strip`` removes the OCR text layer in place without rasterizing or
|
||||
running OCR, so image-processing and OCR-output options have no effect.
|
||||
"""
|
||||
if options.mode != ProcessingMode.strip_text:
|
||||
return
|
||||
incompatible = {
|
||||
'--deskew': options.deskew,
|
||||
'--clean': options.clean,
|
||||
'--clean-final': options.clean_final,
|
||||
'--remove-background': options.remove_background,
|
||||
'--rotate-pages': options.rotate_pages,
|
||||
'--oversample': options.oversample,
|
||||
'--remove-vectors': options.remove_vectors,
|
||||
'--sidecar': options.sidecar,
|
||||
}
|
||||
used = sorted(name for name, value in incompatible.items() if value)
|
||||
if used:
|
||||
raise BadArgsError(
|
||||
"--mode strip removes the OCR text layer without rasterizing or "
|
||||
"running OCR, so these options have no effect and are not allowed: "
|
||||
f"{', '.join(used)}"
|
||||
)
|
||||
|
||||
|
||||
def _check_plugin_invariant_options(options: OcrOptions) -> None:
|
||||
check_platform()
|
||||
check_options_strip(options)
|
||||
check_options_sidecar(options)
|
||||
check_options_preprocessing(options)
|
||||
|
||||
@@ -182,7 +211,7 @@ def create_input_file(options: OcrOptions, work_folder: Path) -> tuple[Path, str
|
||||
if running_in_docker(): # pragma: no cover
|
||||
msg += (
|
||||
"\nDocker cannot access your working directory unless you "
|
||||
"explicitly share it with the Docker container and set up"
|
||||
"explicitly share it with the Docker container and set up "
|
||||
"permissions correctly.\n"
|
||||
"You may find it easier to use stdin/stdout:"
|
||||
"\n"
|
||||
@@ -203,7 +232,13 @@ def create_input_file(options: OcrOptions, work_folder: Path) -> tuple[Path, str
|
||||
|
||||
def check_requested_output_file(options: OcrOptions) -> None:
|
||||
if options.output_file == '-':
|
||||
if sys.stdout.isatty():
|
||||
# When stdout protection is active, fd 1 has been redirected to stderr,
|
||||
# so sys.stdout.isatty() would report stderr's status. Consult the
|
||||
# preserved real stdout instead, falling back when protection is off.
|
||||
is_tty = protected_stdout_isatty()
|
||||
if is_tty is None:
|
||||
is_tty = sys.stdout.isatty()
|
||||
if is_tty:
|
||||
raise BadArgsError(
|
||||
"Output was set to stdout '-' but it looks like stdout "
|
||||
"is connected to a terminal. Please redirect stdout to a "
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
__version__ = "17.4.2"
|
||||
__version__ = "17.8.0"
|
||||
|
||||
@@ -56,6 +56,7 @@ from ocrmypdf._pipelines.hocr_to_ocr_pdf import run_hocr_to_ocr_pdf_pipeline
|
||||
from ocrmypdf._pipelines.ocr import run_pipeline, run_pipeline_cli
|
||||
from ocrmypdf._pipelines.pdf_to_hocr import run_hocr_pipeline
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager, get_plugin_manager
|
||||
from ocrmypdf._stdoutprotect import protect_stdout
|
||||
from ocrmypdf._validation import check_options
|
||||
from ocrmypdf.cli import ArgumentParser, get_parser
|
||||
from ocrmypdf.exceptions import ExitCode
|
||||
@@ -233,6 +234,37 @@ def configure_logging(
|
||||
return log
|
||||
|
||||
|
||||
def configure_stdout_protection() -> bool:
|
||||
"""Protect the process's real standard output from corruption.
|
||||
|
||||
When OCRmyPDF writes its final PDF to standard output (``output_file='-'``),
|
||||
the bytes on stdout must be exactly the PDF and nothing else. By default
|
||||
OCRmyPDF relies on no in-process code -- third party libraries, plugins, or
|
||||
stray ``print()`` calls -- ever writing to stdout. This function makes that
|
||||
guarantee real: it redirects file descriptor 1 to standard error and
|
||||
preserves a private copy of the real stdout, so that any accidental write to
|
||||
stdout lands harmlessly on stderr while OCRmyPDF still emits its final PDF to
|
||||
the preserved descriptor.
|
||||
|
||||
This is the same protection the ``ocrmypdf`` command line program installs.
|
||||
It is optional for API users and works like :func:`configure_logging`: call
|
||||
it before :func:`ocr` if you want command-line-like behavior. It must be
|
||||
called once, early -- before any plugins are loaded or any worker
|
||||
process/thread is started -- so that they inherit the redirected descriptor.
|
||||
|
||||
Because it mutates process-global file descriptors and affects the entire
|
||||
process, applications that manage their own standard output (for example,
|
||||
a long-lived service that calls :func:`ocr` in-process) should **not** call
|
||||
this function.
|
||||
|
||||
Returns:
|
||||
True if protection was installed (or was already active). False if
|
||||
stdout is not backed by a real operating system file descriptor, in
|
||||
which case nothing is changed.
|
||||
"""
|
||||
return protect_stdout()
|
||||
|
||||
|
||||
def _check_no_conflicting_ocr_params(
|
||||
locals_dict: dict,
|
||||
kwargs: dict,
|
||||
@@ -965,6 +997,7 @@ __all__ = [
|
||||
'Verbosity',
|
||||
'check_options',
|
||||
'configure_logging',
|
||||
'configure_stdout_protection',
|
||||
'create_options',
|
||||
'get_parser',
|
||||
'get_plugin_manager',
|
||||
|
||||
@@ -44,6 +44,30 @@ class PdfaImageCompression(StrEnum):
|
||||
LOSSLESS = 'lossless'
|
||||
|
||||
|
||||
def _resolve_auto_compression(
|
||||
compression: PdfaImageCompression, optimize_level: int
|
||||
) -> PdfaImageCompression:
|
||||
"""Resolve 'auto' image compression based on the optimization level.
|
||||
|
||||
At ``-O0`` (no optimization) ``auto`` maps to ``lossless`` so Ghostscript
|
||||
will not transcode lossless images to JPEG during PDF/A generation. At all
|
||||
other levels ``auto`` defers to Ghostscript's heuristic, which may
|
||||
recompress images lossily.
|
||||
|
||||
``-O1`` is a historical exception: although it is otherwise a
|
||||
lossless-only optimization level, coercing ``auto`` to ``lossless`` there
|
||||
can bloat output substantially (Ghostscript's heuristic often picks JPEG
|
||||
for photographic content), so the default is left alone for backwards
|
||||
compatibility. Users who want guaranteed lossless image handling at any
|
||||
level can pass ``--pdfa-image-compression=lossless`` explicitly.
|
||||
|
||||
Explicit ``jpeg`` and ``lossless`` choices are always respected.
|
||||
"""
|
||||
if compression == PdfaImageCompression.AUTO and optimize_level == 0:
|
||||
return PdfaImageCompression.LOSSLESS
|
||||
return compression
|
||||
|
||||
|
||||
class GhostscriptOptions(BaseModel):
|
||||
"""Options specific to Ghostscript operations."""
|
||||
|
||||
@@ -54,6 +78,27 @@ class GhostscriptOptions(BaseModel):
|
||||
pdfa_image_compression: Annotated[
|
||||
PdfaImageCompression, Field(description="PDF/A image compression method")
|
||||
] = PdfaImageCompression.AUTO
|
||||
jpeg_quality: Annotated[
|
||||
int | None,
|
||||
Field(
|
||||
ge=0,
|
||||
le=100,
|
||||
description=(
|
||||
"JPEG quality (0-100) for Ghostscript image recompression during "
|
||||
"PDF/A generation; None uses Ghostscript's default."
|
||||
),
|
||||
),
|
||||
] = None
|
||||
jpeg_maxdpi: Annotated[
|
||||
int | None,
|
||||
Field(
|
||||
ge=1,
|
||||
description=(
|
||||
"Maximum DPI for Ghostscript image downsampling during PDF/A "
|
||||
"generation."
|
||||
),
|
||||
),
|
||||
] = None
|
||||
|
||||
@classmethod
|
||||
def add_arguments_to_parser(cls, parser, namespace: str = 'ghostscript'):
|
||||
@@ -78,14 +123,48 @@ class GhostscriptOptions(BaseModel):
|
||||
choices=[pc.value for pc in PdfaImageCompression],
|
||||
default=PdfaImageCompression.AUTO.value,
|
||||
help="Specify how to compress images in the output PDF/A. 'auto' lets "
|
||||
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
|
||||
"OCRmyPDF decide: at -O0 it uses lossless image compression so "
|
||||
"Ghostscript does not transcode lossless images to JPEG; at -O1 and "
|
||||
"above it defers to Ghostscript's heuristic, which may recompress "
|
||||
"images lossily. 'jpeg' changes all grayscale and color images to "
|
||||
"JPEG compression. 'lossless' uses PNG-style lossless compression "
|
||||
"for all images. Monochrome images are always compressed using a "
|
||||
"for non-JPEG images and passes existing JPEGs through unchanged "
|
||||
"(re-encoding them losslessly would only inflate them). Monochrome "
|
||||
"images are always compressed using a "
|
||||
"lossless codec. Compression settings "
|
||||
"are applied to all pages, including those for which OCR was "
|
||||
"skipped. Not supported for --output-type=pdf ; that setting "
|
||||
"preserves the original compression of all images.",
|
||||
)
|
||||
gs.add_argument(
|
||||
'--ghostscript-jpeg-quality',
|
||||
type=int,
|
||||
metavar='Q',
|
||||
default=None,
|
||||
dest=f'{namespace}_jpeg_quality',
|
||||
help=(
|
||||
"Advanced: Set Ghostscript's -dJPEGQ for images that Ghostscript "
|
||||
"transcodes to JPEG during PDF/A generation. 0 is maximum "
|
||||
"compression; 100 is best quality. If omitted, Ghostscript's "
|
||||
"default is used. This only affects images Ghostscript chooses "
|
||||
"to recompress; for general JPEG quality tuning prefer "
|
||||
"--jpeg-quality, which is applied by the OCRmyPDF optimizer."
|
||||
),
|
||||
)
|
||||
gs.add_argument(
|
||||
'--ghostscript-jpeg-maxdpi',
|
||||
type=int,
|
||||
metavar='DPI',
|
||||
default=None,
|
||||
dest=f'{namespace}_jpeg_maxdpi',
|
||||
help=(
|
||||
"Advanced: Force Ghostscript to downsample color, grayscale, "
|
||||
"and monochrome images in PDF/A output to the given maximum DPI. "
|
||||
"Reducing JPEG quality usually gives better results than "
|
||||
"downsampling at the same file size, and can degrade quality "
|
||||
"of high-resolution monochrome masks."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@hookimpl
|
||||
@@ -177,6 +256,8 @@ def rasterize_pdf_page(
|
||||
# Let pypdfium handle it (it will error in check_options if unavailable)
|
||||
return None
|
||||
|
||||
log.debug("Rasterizing page %d with the Ghostscript rasterizer", pageno)
|
||||
|
||||
ghostscript.rasterize_pdf(
|
||||
input_file,
|
||||
output_file,
|
||||
@@ -347,11 +428,18 @@ def generate_pdfa(
|
||||
if output_type == 'pdfa':
|
||||
output_type = 'pdfa-2'
|
||||
|
||||
compression = _resolve_auto_compression(
|
||||
context.options.ghostscript.pdfa_image_compression,
|
||||
context.options.optimize,
|
||||
)
|
||||
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[pdfmark, *pdf_pages],
|
||||
output_file=output_file,
|
||||
compression=context.options.ghostscript.pdfa_image_compression,
|
||||
compression=compression,
|
||||
color_conversion_strategy=context.options.ghostscript.color_conversion_strategy,
|
||||
jpeg_quality=context.options.ghostscript.jpeg_quality,
|
||||
jpeg_maxdpi=context.options.ghostscript.jpeg_maxdpi,
|
||||
pdf_version=pdf_version,
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=progressbar_class,
|
||||
|
||||
@@ -48,27 +48,18 @@ def _open_pdf_document(input_file: Path):
|
||||
return pdfium.PdfDocument(input_file)
|
||||
|
||||
|
||||
def _calculate_mediabox_crop(page) -> tuple[float, float, float, float]:
|
||||
"""Calculate crop values to expand rendering from CropBox to MediaBox.
|
||||
def _expand_cropbox_to_mediabox(page) -> None:
|
||||
"""Set the page's CropBox to its MediaBox so PDFium renders the full page.
|
||||
|
||||
By default pypdfium2 renders to the CropBox. To render the full MediaBox,
|
||||
we need negative crop values to expand the rendering area.
|
||||
|
||||
Returns:
|
||||
Tuple of (left, bottom, right, top) crop values. Negative values
|
||||
expand the rendering area beyond the CropBox to the MediaBox.
|
||||
PDFium renders to the CropBox by default. Negative ``crop`` values to
|
||||
``render()`` are not supported and only pad the output canvas without
|
||||
expanding the rendered area — content outside the CropBox is clipped.
|
||||
The supported approach is to widen the CropBox in memory before rendering.
|
||||
The document is never saved back to disk, so this mutation is local.
|
||||
See https://github.com/ocrmypdf/OCRmyPDF/issues/1685.
|
||||
"""
|
||||
mediabox = page.get_mediabox() # (left, bottom, right, top)
|
||||
cropbox = page.get_cropbox() # (left, bottom, right, top), defaults to mediabox
|
||||
|
||||
# Calculate how much to expand from cropbox to mediabox
|
||||
# Negative values = expand, positive = shrink
|
||||
return (
|
||||
mediabox[0] - cropbox[0], # Expand left
|
||||
mediabox[1] - cropbox[1], # Expand bottom
|
||||
cropbox[2] - mediabox[2], # Expand right
|
||||
cropbox[3] - mediabox[3], # Expand top
|
||||
)
|
||||
page.set_cropbox(*mediabox)
|
||||
|
||||
|
||||
def _render_page_to_bitmap(
|
||||
@@ -105,16 +96,20 @@ def _render_page_to_bitmap(
|
||||
# Render the page to a bitmap
|
||||
# The scale parameter controls the resolution
|
||||
# Render in grayscale for mono and gray devices (better input for 1-bit conversion)
|
||||
grayscale = raster_device.lower() in ('pngmono', 'pnggray', 'jpeggray')
|
||||
grayscale = raster_device.lower() in (
|
||||
'pngmono',
|
||||
'pngmonod',
|
||||
'pnggray',
|
||||
'jpeggray',
|
||||
)
|
||||
|
||||
# Calculate crop to render the appropriate box
|
||||
# Default (use_cropbox=False) renders MediaBox for consistency with Ghostscript
|
||||
crop = (0, 0, 0, 0) if use_cropbox else _calculate_mediabox_crop(page)
|
||||
if not use_cropbox:
|
||||
_expand_cropbox_to_mediabox(page)
|
||||
|
||||
bitmap = page.render(
|
||||
scale=scale,
|
||||
rotation=0, # We already set rotation on the page
|
||||
crop=crop,
|
||||
may_draw_forms=True,
|
||||
draw_annots=True,
|
||||
grayscale=grayscale,
|
||||
@@ -167,8 +162,8 @@ def _process_image_for_output(
|
||||
# This ensures pypdfium output matches Ghostscript's native device output
|
||||
raster_device_lower = raster_device.lower()
|
||||
|
||||
if raster_device_lower == 'pngmono':
|
||||
# Convert to 1-bit black and white (matches Ghostscript pngmono device)
|
||||
if raster_device_lower in ('pngmono', 'pngmonod'):
|
||||
# Convert to 1-bit black and white (matches Ghostscript pngmono/pngmonod)
|
||||
if pil_image.mode != '1':
|
||||
if pil_image.mode not in ('L', '1'):
|
||||
pil_image = pil_image.convert('L')
|
||||
@@ -194,7 +189,15 @@ def _process_image_for_output(
|
||||
# pngalpha: keep RGBA as-is
|
||||
|
||||
# Determine output format based on raster_device
|
||||
png_devices = ('png', 'pngmono', 'pnggray', 'png256', 'png16m', 'pngalpha')
|
||||
png_devices = (
|
||||
'png',
|
||||
'pngmono',
|
||||
'pngmonod',
|
||||
'pnggray',
|
||||
'png256',
|
||||
'png16m',
|
||||
'pngalpha',
|
||||
)
|
||||
if raster_device_lower in png_devices:
|
||||
format_name = 'PNG'
|
||||
elif raster_device_lower in ('jpeg', 'jpeggray', 'jpg'):
|
||||
@@ -252,6 +255,8 @@ def rasterize_pdf_page(
|
||||
if pdfium is None:
|
||||
return None # Fall back to Ghostscript
|
||||
|
||||
log.debug("Rasterizing page %d with the pypdfium2 rasterizer", pageno)
|
||||
|
||||
# Acquire lock to ensure thread-safe access to pypdfium2
|
||||
with (
|
||||
_pdfium_lock,
|
||||
|
||||
+14
-5
@@ -327,7 +327,11 @@ Online documentation is located at:
|
||||
"'default' errors if text is found. "
|
||||
"'force' rasterizes all content and runs OCR (same as --force-ocr). "
|
||||
"'skip' skips pages with existing text (same as --skip-text). "
|
||||
"'redo' re-OCRs pages, replacing old invisible text (same as --redo-ocr).",
|
||||
"'redo' re-OCRs pages, replacing old invisible text (same as --redo-ocr). "
|
||||
"'strip' removes the invisible OCR text layer without rasterizing or "
|
||||
"running OCR, producing a smaller file; only text drawn as invisible "
|
||||
"(render mode 3) is removed, so text from some OCR engines cannot be "
|
||||
"removed this way.",
|
||||
)
|
||||
# Legacy flags for backward compatibility - these set the mode internally
|
||||
ocrsettings.add_argument(
|
||||
@@ -387,7 +391,8 @@ Online documentation is located at:
|
||||
type=str,
|
||||
help=(
|
||||
"Limit OCR to the specified pages (ranges or comma separated), "
|
||||
"skipping others"
|
||||
"skipping others. The token 'end' is an alias for the last page, "
|
||||
"so e.g. '3-end' OCRs from page 3 to the last page."
|
||||
),
|
||||
)
|
||||
advanced.add_argument(
|
||||
@@ -422,9 +427,13 @@ Online documentation is located at:
|
||||
'--rasterizer',
|
||||
choices=['auto', 'ghostscript', 'pypdfium'],
|
||||
default='auto',
|
||||
help="Choose PDF page rasterizer. 'auto' prefers pypdfium when available, "
|
||||
"falling back to Ghostscript. 'pypdfium' is faster but requires the "
|
||||
"pypdfium2 package. 'ghostscript' uses the traditional Ghostscript rasterizer.",
|
||||
help="Choose PDF page rasterizer. 'auto' (the default) prefers pypdfium2 "
|
||||
"when the pypdfium2 package is installed, falling back to Ghostscript "
|
||||
"otherwise. pypdfium2 anti-aliases page content and generally produces "
|
||||
"better input for OCR than Ghostscript 10.x, which can render aliased "
|
||||
"glyphs that OCR misreads as extra word breaks. 'pypdfium' forces the "
|
||||
"pypdfium2 rasterizer (requires the pypdfium2 package); 'ghostscript' "
|
||||
"forces the traditional Ghostscript rasterizer.",
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--rotate-pages-threshold',
|
||||
|
||||
+70
-10
@@ -139,14 +139,74 @@ class TaggedPDFError(InputFileError):
|
||||
)
|
||||
|
||||
|
||||
class ColorConversionNeededError(BadArgsError):
|
||||
"""PDF needs color conversion."""
|
||||
class NonEmbeddedFontsError(InputFileError):
|
||||
"""Input has non-embedded CID fonts that PDF/A conversion would corrupt.
|
||||
|
||||
message = dedent(
|
||||
"""\
|
||||
The input PDF has an unusual color space. Use
|
||||
--color-conversion-strategy to convert to a common color space
|
||||
such as RGB, or use --output-type pdf to skip PDF/A conversion
|
||||
and retain the original color space.
|
||||
"""
|
||||
)
|
||||
PDF/A requires all fonts to be embedded. Ghostscript substitutes and embeds
|
||||
a replacement for non-embedded CID (CJK) fonts, which corrupts the
|
||||
character-to-Unicode mapping and silently destroys an existing text layer
|
||||
(commonly an Adobe Acrobat CJK OCR layer). OCRmyPDF refuses to produce such
|
||||
output rather than damage the user's data
|
||||
(see https://github.com/ocrmypdf/OCRmyPDF/issues/1561).
|
||||
"""
|
||||
|
||||
def __init__(self, fonts: set[str]):
|
||||
"""Build guidance naming the offending fonts."""
|
||||
super().__init__()
|
||||
font_list = ', '.join(sorted(fonts))
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
The input PDF contains non-embedded CID (character ID) fonts: {font_list}.
|
||||
|
||||
PDF/A requires all fonts to be embedded. Converting to PDF/A would
|
||||
make Ghostscript substitute and embed replacement fonts, which
|
||||
corrupts CID (e.g. CJK/Chinese-Japanese-Korean) text and silently
|
||||
destroys an existing text layer such as one produced by Adobe Acrobat.
|
||||
|
||||
Use --output-type pdf to keep the existing text layer intact without
|
||||
PDF/A conversion, or --force-ocr to discard the existing layer and
|
||||
rebuild it with embedded fonts.
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
class ColorConversionNeededError(BadArgsError):
|
||||
"""PDF needs color conversion to a standard color space.
|
||||
|
||||
Ghostscript reported a DeviceN colorspace with an inappropriate alternate.
|
||||
The resulting PDF/A is liable to render incorrectly (often blank) in some
|
||||
viewers such as Adobe Reader, so the colorspace must be normalized to a
|
||||
common one. RGB, CMYK, and Gray are known to work; LeaveColorUnchanged
|
||||
performs no conversion and UseDeviceIndependentColor does not resolve the
|
||||
problem (see https://github.com/ocrmypdf/OCRmyPDF/issues/1187).
|
||||
"""
|
||||
|
||||
# Strategies that can normalize an unusual DeviceN colorspace into one that
|
||||
# PDF/A viewers render correctly.
|
||||
_effective_strategies = "RGB, CMYK, or Gray"
|
||||
|
||||
def __init__(self, color_conversion_strategy: str = "LeaveColorUnchanged"):
|
||||
"""Build guidance tailored to the conversion strategy that was used."""
|
||||
super().__init__()
|
||||
if color_conversion_strategy == "LeaveColorUnchanged":
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
The input PDF has an unusual DeviceN color space that cannot be
|
||||
represented in PDF/A; the output may appear blank in some viewers
|
||||
such as Adobe Reader. Convert it to a common color space with
|
||||
--color-conversion-strategy ({self._effective_strategies}), or use
|
||||
--output-type pdf to skip PDF/A conversion and retain the original
|
||||
color space.
|
||||
"""
|
||||
)
|
||||
else:
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
Color conversion with --color-conversion-strategy
|
||||
{color_conversion_strategy} did not resolve the input PDF's unusual
|
||||
DeviceN color space; the output may appear blank in some viewers
|
||||
such as Adobe Reader. Try a different --color-conversion-strategy
|
||||
({self._effective_strategies}), or use --output-type pdf to skip
|
||||
PDF/A conversion and retain the original color space.
|
||||
"""
|
||||
)
|
||||
|
||||
@@ -54,13 +54,15 @@ class MultiFontManager:
|
||||
'kok': 'NotoSansDevanagari-Regular', # Konkani
|
||||
'bho': 'NotoSansDevanagari-Regular', # Bhojpuri
|
||||
'mai': 'NotoSansDevanagari-Regular', # Maithili
|
||||
# CJK
|
||||
'chi': 'NotoSansCJK-Regular', # Chinese (generic)
|
||||
'zho': 'NotoSansCJK-Regular', # Chinese (ISO 639-3)
|
||||
'chi_sim': 'NotoSansCJK-Regular', # Chinese Simplified (Tesseract)
|
||||
'chi_tra': 'NotoSansCJK-Regular', # Chinese Traditional (Tesseract)
|
||||
'jpn': 'NotoSansCJK-Regular', # Japanese
|
||||
'kor': 'NotoSansCJK-Regular', # Korean
|
||||
# CJK — prefer the family matching the document language, because the
|
||||
# modern per-language Noto fonts are region subsets (e.g. NotoSansSC
|
||||
# lacks Japanese kana). The pan-CJK super font is a shared fallback.
|
||||
'chi': 'NotoSansSC-Regular', # Chinese (generic → Simplified)
|
||||
'zho': 'NotoSansSC-Regular', # Chinese (ISO 639-3)
|
||||
'chi_sim': 'NotoSansSC-Regular', # Chinese Simplified (Tesseract)
|
||||
'chi_tra': 'NotoSansTC-Regular', # Chinese Traditional (Tesseract)
|
||||
'jpn': 'NotoSansJP-Regular', # Japanese
|
||||
'kor': 'NotoSansKR-Regular', # Korean
|
||||
# Thai
|
||||
'tha': 'NotoSansThai-Regular', # Thai
|
||||
# Hebrew
|
||||
@@ -113,7 +115,14 @@ class MultiFontManager:
|
||||
'NotoSans-Regular', # Latin, Greek, Cyrillic
|
||||
'NotoSansArabic-Regular',
|
||||
'NotoSansDevanagari-Regular',
|
||||
# Pan-CJK super font first (full coverage), then the per-language
|
||||
# subsets so a glyph missing from one CJK family is found in another.
|
||||
'NotoSansCJK-Regular',
|
||||
'NotoSansSC-Regular',
|
||||
'NotoSansTC-Regular',
|
||||
'NotoSansHK-Regular',
|
||||
'NotoSansJP-Regular',
|
||||
'NotoSansKR-Regular',
|
||||
'NotoSansThai-Regular',
|
||||
'NotoSansHebrew-Regular',
|
||||
'NotoSansBengali-Regular',
|
||||
@@ -256,19 +265,23 @@ class MultiFontManager:
|
||||
self._warned_scripts.add(warn_key)
|
||||
|
||||
if line_language and line_language in self.LANGUAGE_FONT_MAP:
|
||||
font_name = self.LANGUAGE_FONT_MAP[line_language]
|
||||
font_family = self.LANGUAGE_FONT_MAP[line_language].removesuffix('-Regular')
|
||||
log.warning(
|
||||
"No font found with glyphs for '%s' text. "
|
||||
"Install %s for better rendering. "
|
||||
"See https://fonts.google.com/noto",
|
||||
"No installed font has glyphs for the detected '%s' text, so "
|
||||
"it was added as an invisible text layer: it stays searchable "
|
||||
"and copyable, but appears blank when highlighted in a PDF "
|
||||
"viewer. Install the %s font family (via your OS package "
|
||||
"manager or https://fonts.google.com/noto) for full rendering.",
|
||||
line_language,
|
||||
font_name,
|
||||
font_family,
|
||||
)
|
||||
else:
|
||||
log.warning(
|
||||
"No font found with glyphs for some text. "
|
||||
"Install Noto fonts for better rendering. "
|
||||
"See https://fonts.google.com/noto"
|
||||
"No installed font has glyphs for some of the detected text, "
|
||||
"so it was added as an invisible text layer: it stays "
|
||||
"searchable and copyable, but appears blank when highlighted "
|
||||
"in a PDF viewer. Install the matching Noto fonts "
|
||||
"(https://fonts.google.com/noto) for full rendering."
|
||||
)
|
||||
|
||||
def _has_all_glyphs(self, font: FontManager, text: str) -> bool:
|
||||
|
||||
@@ -9,6 +9,7 @@ Linux, macOS, and Windows platforms.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import glob
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
@@ -75,6 +76,35 @@ class SystemFontProvider:
|
||||
# Variable fonts
|
||||
'NotoSansCJKsc-VF.otf',
|
||||
],
|
||||
# Per-language CJK families. Modern Google Fonts / Homebrew ship these
|
||||
# as region subset variable fonts ('NotoSansJP[wght].ttf'), matched by
|
||||
# the flexible base search; the legacy per-region super OTFs (full
|
||||
# coverage) are listed here so they also satisfy the logical name.
|
||||
'NotoSansSC-Regular': [
|
||||
'NotoSansSC-Regular.otf',
|
||||
'NotoSansSC-Regular.ttf',
|
||||
'NotoSansCJKsc-Regular.otf',
|
||||
],
|
||||
'NotoSansTC-Regular': [
|
||||
'NotoSansTC-Regular.otf',
|
||||
'NotoSansTC-Regular.ttf',
|
||||
'NotoSansCJKtc-Regular.otf',
|
||||
],
|
||||
'NotoSansHK-Regular': [
|
||||
'NotoSansHK-Regular.otf',
|
||||
'NotoSansHK-Regular.ttf',
|
||||
'NotoSansCJKhk-Regular.otf',
|
||||
],
|
||||
'NotoSansJP-Regular': [
|
||||
'NotoSansJP-Regular.otf',
|
||||
'NotoSansJP-Regular.ttf',
|
||||
'NotoSansCJKjp-Regular.otf',
|
||||
],
|
||||
'NotoSansKR-Regular': [
|
||||
'NotoSansKR-Regular.otf',
|
||||
'NotoSansKR-Regular.ttf',
|
||||
'NotoSansCJKkr-Regular.otf',
|
||||
],
|
||||
'NotoSansThai-Regular': [
|
||||
'NotoSansThai-Regular.ttf',
|
||||
'NotoSansThai-Regular.otf',
|
||||
@@ -149,6 +179,28 @@ class SystemFontProvider:
|
||||
],
|
||||
}
|
||||
|
||||
# Font file extensions we know how to load.
|
||||
_FONT_EXTENSIONS = ('.ttf', '.otf', '.ttc')
|
||||
|
||||
# Acceptable filename variants for a font family, ranked best-first.
|
||||
# Lower rank wins when multiple variants of the same family are present.
|
||||
_VARIANT_RANK = {'regular': 0, 'variable': 1, 'vf': 2, 'plain': 3}
|
||||
|
||||
# Extra family bases that can satisfy a logical font, tried after its own
|
||||
# base (so the listed order is the preference). CJK is the case that needs
|
||||
# this: the legacy Adobe-style 'NotoSansCJKsc-Regular.otf' is handled by
|
||||
# NOTO_FONT_PATTERNS, but Homebrew casks and current Google Fonts ship the
|
||||
# per-language families as variable fonts (e.g. 'NotoSansSC[wght].ttf').
|
||||
_ALTERNATE_BASES: dict[str, list[str]] = {
|
||||
'NotoSansCJK-Regular': [
|
||||
'NotoSansSC', # Simplified Chinese
|
||||
'NotoSansTC', # Traditional Chinese
|
||||
'NotoSansHK', # Hong Kong
|
||||
'NotoSansJP', # Japanese
|
||||
'NotoSansKR', # Korean
|
||||
],
|
||||
}
|
||||
|
||||
def __init__(self) -> None:
|
||||
"""Initialize system font provider with empty caches."""
|
||||
# Cache: font_name -> FontManager (successfully loaded fonts)
|
||||
@@ -230,6 +282,76 @@ class SystemFontProvider:
|
||||
# Skip directories we can't read
|
||||
continue
|
||||
|
||||
# No exact static '-Regular' file. Many distributors (Homebrew casks,
|
||||
# current Google Fonts releases) ship Noto fonts as variable fonts with
|
||||
# bracketed axis filenames such as 'NotoSansArabic[wdth,wght].ttf'.
|
||||
# Fall back to a flexible search that also accepts those. See #1652.
|
||||
return self._find_variant_font_file(font_name)
|
||||
|
||||
@staticmethod
|
||||
def _classify_variant(stem: str, base: str) -> str | None:
|
||||
"""Classify a font filename stem as a usable variant of ``base``.
|
||||
|
||||
Args:
|
||||
stem: Filename without extension (e.g. 'NotoSansArabic[wdth,wght]')
|
||||
base: Family base name (e.g. 'NotoSansArabic')
|
||||
|
||||
Returns:
|
||||
The variant kind ('regular', 'variable', 'vf', 'plain') or None if
|
||||
the stem is not an acceptable representative of the family. The
|
||||
boundary after ``base`` is required so that 'NotoSans' does not
|
||||
match 'NotoSansArabic', and 'NotoSansArabicUI'/'NotoSansArabic-Bold'
|
||||
do not match a request for 'NotoSansArabic'.
|
||||
"""
|
||||
if stem == f'{base}-Regular':
|
||||
return 'regular'
|
||||
if stem.startswith(f'{base}['): # variable font, e.g. Base[wdth,wght]
|
||||
return 'variable'
|
||||
if stem == f'{base}-VF': # alternate variable-font naming
|
||||
return 'vf'
|
||||
if stem == base: # bare family name
|
||||
return 'plain'
|
||||
return None
|
||||
|
||||
def _find_variant_font_file(self, font_name: str) -> Path | None:
|
||||
"""Search for a variable font or other acceptable filename variant.
|
||||
|
||||
Tries the font's own family base first, then any alternate bases (used
|
||||
for the modern per-language CJK families). Within that, a static Regular
|
||||
is preferred over a variable font. See issue #1652.
|
||||
|
||||
Args:
|
||||
font_name: Logical font name (e.g. 'NotoSansArabic-Regular')
|
||||
|
||||
Returns:
|
||||
Path to the best-ranked matching font file, or None.
|
||||
"""
|
||||
bases = [font_name.removesuffix('-Regular')]
|
||||
bases.extend(self._ALTERNATE_BASES.get(font_name, []))
|
||||
|
||||
# Selection key (base_index, variant_rank): earlier base wins, then the
|
||||
# better variant. Path is carried along but not part of the comparison.
|
||||
best: tuple[tuple[int, int], Path] | None = None
|
||||
for base_index, base in enumerate(bases):
|
||||
for font_dir in self._get_font_dirs():
|
||||
if not font_dir.exists():
|
||||
continue
|
||||
try:
|
||||
for path in font_dir.rglob(glob.escape(base) + '*'):
|
||||
if path.suffix.lower() not in self._FONT_EXTENSIONS:
|
||||
continue
|
||||
kind = self._classify_variant(path.stem, base)
|
||||
if kind is None:
|
||||
continue
|
||||
key = (base_index, self._VARIANT_RANK[kind])
|
||||
if best is None or key < best[0]:
|
||||
best = (key, path)
|
||||
except PermissionError:
|
||||
# Skip directories we can't read
|
||||
continue
|
||||
if best is not None:
|
||||
log.debug("Found system font %s at %s (variant match)", font_name, best[1])
|
||||
return best[1]
|
||||
return None
|
||||
|
||||
def get_font(self, font_name: str) -> FontManager | None:
|
||||
|
||||
@@ -260,10 +260,19 @@ def _find_image_xrefs_container(
|
||||
exclude_xrefs: MutableSet[Xref],
|
||||
pageno_for_xref: dict[Xref, int],
|
||||
depth: int = 0,
|
||||
visited_forms: MutableSet[Xref] | None = None,
|
||||
):
|
||||
"""Find all image XRefs or Form XObject and add to the include/exclude sets."""
|
||||
# Form XObjects are not added to include/exclude_xrefs, so the dedup
|
||||
# check below doesn't catch Form-XObject cycles or DAGs. Track them in
|
||||
# a shared set so each Form is only descended into once per document
|
||||
# (issue #1321).
|
||||
if visited_forms is None:
|
||||
visited_forms = set()
|
||||
if depth > 10:
|
||||
log.warning("Recursion depth exceeded in _find_image_xrefs_page")
|
||||
# With visited_forms memoization, this is a soft DAG-height guard
|
||||
# rather than a cycle defense, so a debug log is sufficient.
|
||||
log.debug("Recursion depth exceeded in _find_image_xrefs_page")
|
||||
return
|
||||
try:
|
||||
xobjs = container.Resources.XObject
|
||||
@@ -276,7 +285,9 @@ def _find_image_xrefs_container(
|
||||
if xref in include_xrefs or xref in exclude_xrefs:
|
||||
continue # Already processed
|
||||
if Name.Subtype in image and image.Subtype == Name.Form:
|
||||
# Recurse into Form XObjects
|
||||
if xref in visited_forms:
|
||||
continue
|
||||
visited_forms.add(xref)
|
||||
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
||||
_find_image_xrefs_container(
|
||||
pdf,
|
||||
@@ -286,6 +297,7 @@ def _find_image_xrefs_container(
|
||||
exclude_xrefs,
|
||||
pageno_for_xref,
|
||||
depth + 1,
|
||||
visited_forms,
|
||||
)
|
||||
continue
|
||||
if Name.SMask in image:
|
||||
@@ -342,9 +354,16 @@ def extract_images(
|
||||
pdf=pdf, root=root, image=image, xref=xref, options=options
|
||||
)
|
||||
except Exception: # pylint: disable=broad-except
|
||||
log.exception(
|
||||
f"xref {xref}: While extracting this image, an error occurred"
|
||||
# Optimization is best-effort: an image we cannot process is simply
|
||||
# left unchanged in the output, which remains valid. Report this as
|
||||
# a concise warning rather than an alarming traceback (issue #846);
|
||||
# the full detail is still available at debug verbosity.
|
||||
log.warning(
|
||||
f"xref {xref}: this image could not be processed by the "
|
||||
"optimizer and was left unchanged. The output file is still "
|
||||
"valid."
|
||||
)
|
||||
log.debug(f"xref {xref}: image optimization error detail", exc_info=True)
|
||||
errors += 1
|
||||
else:
|
||||
if result:
|
||||
|
||||
+67
-6
@@ -137,6 +137,65 @@ def file_claims_pdfa(filename: Path):
|
||||
return pdfa_dict
|
||||
|
||||
|
||||
def _cid_font_is_embedded(type0_font: Dictionary) -> bool:
|
||||
"""Return True if a Type0 font's CID descendant carries embedded glyphs."""
|
||||
for descendant in type0_font.get(Name.DescendantFonts, []):
|
||||
descriptor = descendant.get(Name.FontDescriptor, None)
|
||||
if descriptor is not None and any(
|
||||
key in descriptor for key in (Name.FontFile, Name.FontFile2, Name.FontFile3)
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def find_nonembedded_cid_fonts(pdf: Pdf) -> set[str]:
|
||||
"""Find CID-keyed (Type0) fonts that lack embedded glyph data.
|
||||
|
||||
PDF/A requires every font to be embedded. When Ghostscript converts a PDF
|
||||
to PDF/A it must substitute and embed a replacement for any non-embedded
|
||||
font. For CID-keyed fonts -- which is how CJK text is encoded, including the
|
||||
OCR text layers produced by Adobe Acrobat -- this substitution routinely
|
||||
corrupts the character-to-Unicode mapping, silently destroying the
|
||||
searchable text. Detecting these fonts lets the caller refuse PDF/A
|
||||
conversion rather than emit corrupted output.
|
||||
|
||||
Simple (non-CID) non-embedded fonts are not reported: Ghostscript
|
||||
substitutes standard encodings for them without corrupting the text, and
|
||||
they are far too common to treat as conversion blockers.
|
||||
|
||||
Args:
|
||||
pdf: An open ``pikepdf.Pdf`` to scan.
|
||||
|
||||
Returns:
|
||||
The set of ``BaseFont`` names of non-embedded CID fonts found.
|
||||
"""
|
||||
found: set[str] = set()
|
||||
|
||||
def scan_resources(resources, depth: int = 0) -> None:
|
||||
if resources is None or depth > 10:
|
||||
return
|
||||
fonts = resources.get(Name.Font, None)
|
||||
if fonts is not None:
|
||||
for font in fonts.values():
|
||||
try:
|
||||
if font.get(Name.Subtype) != Name.Type0:
|
||||
continue
|
||||
if not _cid_font_is_embedded(font):
|
||||
basefont = str(font.get(Name.BaseFont, '/(unnamed)'))
|
||||
found.add(basefont.lstrip('/'))
|
||||
except (AttributeError, TypeError, KeyError):
|
||||
continue
|
||||
xobjects = resources.get(Name.XObject, None)
|
||||
if xobjects is not None:
|
||||
for xobj in xobjects.values():
|
||||
if xobj.get(Name.Subtype) == Name.Form and Name.Resources in xobj:
|
||||
scan_resources(xobj[Name.Resources], depth + 1)
|
||||
|
||||
for page in pdf.pages:
|
||||
scan_resources(page.get(Name.Resources, None))
|
||||
return found
|
||||
|
||||
|
||||
def _load_srgb_icc_profile() -> bytes:
|
||||
"""Load the sRGB ICC profile from package data."""
|
||||
return (package_files('ocrmypdf.data') / SRGB_ICC_PROFILE_NAME).read_bytes()
|
||||
@@ -191,12 +250,14 @@ def add_srgb_output_intent(pdf: Pdf) -> None:
|
||||
icc_stream[Name.N] = 3 # RGB has 3 components
|
||||
|
||||
# Create OutputIntent dictionary
|
||||
output_intent = Dictionary({
|
||||
'/Type': Name.OutputIntent,
|
||||
'/S': Name('/GTS_PDFA1'),
|
||||
'/OutputConditionIdentifier': 'sRGB',
|
||||
'/DestOutputProfile': icc_stream,
|
||||
})
|
||||
output_intent = Dictionary(
|
||||
{
|
||||
'/Type': Name.OutputIntent,
|
||||
'/S': Name('/GTS_PDFA1'),
|
||||
'/OutputConditionIdentifier': 'sRGB',
|
||||
'/DestOutputProfile': icc_stream,
|
||||
}
|
||||
)
|
||||
|
||||
# Add to catalog's OutputIntents array
|
||||
if Name.OutputIntents not in pdf.Root:
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ocrmypdf.pdfinfo._types import Colorspace, Encoding, FloatRect
|
||||
from ocrmypdf.pdfinfo._types import Colorspace, Encoding, FloatRect, Ink
|
||||
from ocrmypdf.pdfinfo.info import PageInfo, PdfInfo
|
||||
|
||||
__all__ = ["Colorspace", "Encoding", "FloatRect", "PageInfo", "PdfInfo"]
|
||||
__all__ = ["Colorspace", "Encoding", "FloatRect", "Ink", "PageInfo", "PdfInfo"]
|
||||
|
||||
@@ -11,11 +11,11 @@ from math import hypot, inf, isclose
|
||||
from typing import NamedTuple
|
||||
from warnings import warn
|
||||
|
||||
from pikepdf import Matrix, Object, PdfInlineImage, parse_content_stream
|
||||
from pikepdf import Matrix, Name, Object, PdfInlineImage, parse_content_stream
|
||||
|
||||
from ocrmypdf.exceptions import InputFileError
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pdfinfo._types import UNIT_SQUARE
|
||||
from ocrmypdf.pdfinfo._types import UNIT_SQUARE, Ink
|
||||
|
||||
|
||||
class XobjectSettings(NamedTuple):
|
||||
@@ -24,6 +24,7 @@ class XobjectSettings(NamedTuple):
|
||||
name: str
|
||||
shorthand: tuple[float, float, float, float, float, float]
|
||||
stack_depth: int
|
||||
fill_ink: Ink
|
||||
|
||||
|
||||
class InlineSettings(NamedTuple):
|
||||
@@ -32,6 +33,7 @@ class InlineSettings(NamedTuple):
|
||||
iimage: PdfInlineImage
|
||||
shorthand: tuple[float, float, float, float, float, float]
|
||||
stack_depth: int
|
||||
fill_ink: Ink
|
||||
|
||||
|
||||
class ContentsInfo(NamedTuple):
|
||||
@@ -67,6 +69,60 @@ def _is_unit_square(shorthand):
|
||||
return all(isclose(a, b, rel_tol=1e-3) for a, b in pairwise)
|
||||
|
||||
|
||||
_INK_EPSILON = 1e-3
|
||||
|
||||
# Maps a fill-colorspace name (set by the `cs` operator) to a device color
|
||||
# family we can classify. Names not present here (Separation, ICCBased,
|
||||
# Indexed, DeviceN, Pattern, resource names like /CS0) are treated as color.
|
||||
_DEVICE_FILL_SPACE = {
|
||||
'/DeviceGray': 'gray',
|
||||
'/CalGray': 'gray',
|
||||
'/G': 'gray',
|
||||
'/DeviceRGB': 'rgb',
|
||||
'/CalRGB': 'rgb',
|
||||
'/RGB': 'rgb',
|
||||
'/DeviceCMYK': 'cmyk',
|
||||
'/CMYK': 'cmyk',
|
||||
}
|
||||
|
||||
|
||||
def _ink_from_components(space: str, comps: list[float]) -> Ink:
|
||||
"""Classify a device-color fill into mono/gray/color.
|
||||
|
||||
``space`` is one of 'gray', 'rgb', 'cmyk'. Any other value is treated
|
||||
conservatively as color, since we cannot prove it is achromatic.
|
||||
"""
|
||||
eps = _INK_EPSILON
|
||||
if space == 'gray' and len(comps) == 1:
|
||||
return Ink.mono if comps[0] <= eps else Ink.gray
|
||||
if space == 'rgb' and len(comps) == 3:
|
||||
r, g, b = comps
|
||||
if max(r, g, b) <= eps:
|
||||
return Ink.mono
|
||||
if abs(r - g) <= eps and abs(g - b) <= eps:
|
||||
return Ink.gray
|
||||
return Ink.color
|
||||
if space == 'cmyk' and len(comps) == 4:
|
||||
c, m, y, k = comps
|
||||
if c <= eps and m <= eps and y <= eps:
|
||||
return Ink.mono if k <= eps else Ink.gray
|
||||
return Ink.color
|
||||
return Ink.color # conservative-to-color
|
||||
|
||||
|
||||
def _operand_floats(operands) -> list[float] | None:
|
||||
"""Convert color operands to floats, or None if any is non-numeric.
|
||||
|
||||
Color operators in a malformed content stream may carry the wrong number
|
||||
of operands or a non-numeric operand (e.g. a Name). Returning None lets
|
||||
the caller keep the prior fill state instead of raising.
|
||||
"""
|
||||
try:
|
||||
return [float(o) for o in operands]
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _normalize_stack(graphobjs):
|
||||
"""Convert runs of qQ's in the stack into single graphobjs."""
|
||||
for operands, operator in graphobjs:
|
||||
@@ -78,12 +134,15 @@ def _normalize_stack(graphobjs):
|
||||
yield (operands, operator)
|
||||
|
||||
|
||||
def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
def _interpret_contents(
|
||||
contentstream: Object, initial_shorthand=UNIT_SQUARE, initial_fill_ink=Ink.mono
|
||||
):
|
||||
"""Interpret the PDF content stream.
|
||||
|
||||
The stack represents the state of the PDF graphics stack. We are only
|
||||
interested in the current transformation matrix (CTM) so we only track
|
||||
this object; a full implementation would need to track many other items.
|
||||
The stack represents the state of the PDF graphics stack. We track the
|
||||
current transformation matrix (CTM) and the current fill color (so that
|
||||
image masks, which are painted with the fill color, can be classified);
|
||||
a full implementation would need to track many other items.
|
||||
|
||||
The CTM is initialized to the mapping from user space to device space.
|
||||
PDF units are 1/72". In a PDF viewer or printer this matrix is initialized
|
||||
@@ -102,10 +161,12 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
stack depth exceeds the spec limit and set a hard limit beyond this to
|
||||
bound our memory requirements. If the stack underflows behavior is
|
||||
undefined in the spec, but we just pretend nothing happened and leave the
|
||||
CTM unchanged.
|
||||
graphics state unchanged.
|
||||
"""
|
||||
stack = []
|
||||
ctm = Matrix(initial_shorthand)
|
||||
fill_ink = initial_fill_ink # PDF default fill color is black
|
||||
fill_space = '/DeviceGray' # current fill colorspace name (for sc/scn)
|
||||
xobject_settings: list[XobjectSettings] = []
|
||||
inline_images: list[InlineSettings] = []
|
||||
name_index = defaultdict(lambda: [])
|
||||
@@ -114,14 +175,15 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
vector_ops = set('S s f F f* B B* b b*'.split())
|
||||
text_showing_ops = set("""TJ Tj " '""".split())
|
||||
image_ops = set('BI ID EI q Q Do cm'.split())
|
||||
operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops)
|
||||
color_ops = set('g rg k cs sc scn'.split())
|
||||
operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops | color_ops)
|
||||
|
||||
for n, graphobj in enumerate(
|
||||
_normalize_stack(parse_content_stream(contentstream, operator_whitelist))
|
||||
):
|
||||
operands, operator = graphobj
|
||||
if operator == 'q':
|
||||
stack.append(ctm)
|
||||
stack.append((ctm, fill_ink, fill_space))
|
||||
if len(stack) > 32: # See docstring
|
||||
if len(stack) > 128:
|
||||
raise RuntimeError(
|
||||
@@ -130,9 +192,9 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
warn("PDF graphics stack overflowed spec limit")
|
||||
elif operator == 'Q':
|
||||
try:
|
||||
ctm = stack.pop()
|
||||
ctm, fill_ink, fill_space = stack.pop()
|
||||
except IndexError:
|
||||
# Keeping the ctm the same seems to be the only sensible thing
|
||||
# Keeping the state the same seems to be the only sensible thing
|
||||
# to do. Just pretend nothing happened, keep calm and carry on.
|
||||
warn("PDF graphics stack underflowed - PDF may be malformed")
|
||||
elif operator == 'cm':
|
||||
@@ -143,17 +205,51 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
"PDF content stream is corrupt - this PDF is malformed. "
|
||||
"Use a PDF editor that is capable of visually inspecting the PDF."
|
||||
) from e
|
||||
elif operator == 'g':
|
||||
if vals := _operand_floats(operands):
|
||||
fill_ink = _ink_from_components('gray', vals)
|
||||
fill_space = '/DeviceGray'
|
||||
elif operator == 'rg':
|
||||
if vals := _operand_floats(operands):
|
||||
fill_ink = _ink_from_components('rgb', vals)
|
||||
fill_space = '/DeviceRGB'
|
||||
elif operator == 'k':
|
||||
if vals := _operand_floats(operands):
|
||||
fill_ink = _ink_from_components('cmyk', vals)
|
||||
fill_space = '/DeviceCMYK'
|
||||
elif operator == 'cs':
|
||||
# Selecting a colorspace resets the fill color to that space's
|
||||
# initial value, which is black for all device colorspaces.
|
||||
fill_ink = Ink.mono
|
||||
if operands:
|
||||
fill_space = str(operands[0])
|
||||
elif operator in ('sc', 'scn'):
|
||||
if any(isinstance(o, Name) for o in operands):
|
||||
fill_ink = Ink.color # pattern fill
|
||||
else:
|
||||
space = _DEVICE_FILL_SPACE.get(fill_space)
|
||||
vals = _operand_floats(operands)
|
||||
if space is None or vals is None:
|
||||
fill_ink = Ink.color # conservative for non-device space
|
||||
else:
|
||||
fill_ink = _ink_from_components(space, vals)
|
||||
elif operator == 'Do':
|
||||
image_name = operands[0]
|
||||
settings = XobjectSettings(
|
||||
name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack)
|
||||
name=image_name,
|
||||
shorthand=ctm.shorthand,
|
||||
stack_depth=len(stack),
|
||||
fill_ink=fill_ink,
|
||||
)
|
||||
xobject_settings.append(settings)
|
||||
name_index[str(image_name)].append(settings)
|
||||
elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this
|
||||
iimage = operands[0]
|
||||
inline = InlineSettings(
|
||||
iimage=iimage, shorthand=ctm.shorthand, stack_depth=len(stack)
|
||||
iimage=iimage,
|
||||
shorthand=ctm.shorthand,
|
||||
stack_depth=len(stack),
|
||||
fill_ink=fill_ink,
|
||||
)
|
||||
inline_images.append(inline)
|
||||
elif operator in vector_ops:
|
||||
|
||||
@@ -36,6 +36,7 @@ from ocrmypdf.pdfinfo._types import (
|
||||
UNIT_SQUARE,
|
||||
Colorspace,
|
||||
Encoding,
|
||||
Ink,
|
||||
)
|
||||
|
||||
logger = logging.getLogger()
|
||||
@@ -61,10 +62,12 @@ class ImageInfo:
|
||||
pdfimage: Object | None = None,
|
||||
inline: PdfInlineImage | None = None,
|
||||
shorthand=None,
|
||||
fill_ink: Ink | None = None,
|
||||
):
|
||||
"""Initialize an ImageInfo."""
|
||||
self._name = str(name)
|
||||
self._shorthand = shorthand
|
||||
self._fill_ink = fill_ink
|
||||
|
||||
pim: PdfInlineImage | PdfImage
|
||||
|
||||
@@ -175,6 +178,17 @@ class ImageInfo:
|
||||
"""Type of image, either 'image' or 'stencil'."""
|
||||
return self._type
|
||||
|
||||
@property
|
||||
def ink(self) -> Ink | None:
|
||||
"""Fill-color classification for stencil masks, else None.
|
||||
|
||||
A stencil (image mask) is painted with the current fill color; this
|
||||
reports whether that color is mono/gray/color so the rasterizer can
|
||||
choose a device that does not discard the distinction. Non-stencil
|
||||
images return None.
|
||||
"""
|
||||
return self._fill_ink if self._type == 'stencil' else None
|
||||
|
||||
@property
|
||||
def width(self) -> int:
|
||||
"""Width of the image in pixels."""
|
||||
@@ -249,7 +263,10 @@ def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
||||
"""Find inline images in the contentstream."""
|
||||
for n, inline in enumerate(contentsinfo.inline_images):
|
||||
yield ImageInfo(
|
||||
name=f'inline-{n:02d}', shorthand=inline.shorthand, inline=inline.iimage
|
||||
name=f'inline-{n:02d}',
|
||||
shorthand=inline.shorthand,
|
||||
inline=inline.iimage,
|
||||
fill_ink=inline.fill_ink,
|
||||
)
|
||||
|
||||
|
||||
@@ -300,7 +317,12 @@ def _find_regular_images(
|
||||
# these from our DPI calculation for the page.
|
||||
continue
|
||||
|
||||
yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand)
|
||||
yield ImageInfo(
|
||||
name=draw.name,
|
||||
pdfimage=pdfimage,
|
||||
shorthand=draw.shorthand,
|
||||
fill_ink=draw.fill_ink,
|
||||
)
|
||||
|
||||
|
||||
def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo):
|
||||
@@ -330,13 +352,19 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
||||
# but in practice both Form XObjects and multiple drawing of the
|
||||
# same object are both very rare.
|
||||
ctm_shorthand = settings.shorthand
|
||||
# A Form XObject inherits the graphics state (including fill color)
|
||||
# in effect at the Do that draws it, so a mask painted with an
|
||||
# inherited gray/color fill must carry that classification inward.
|
||||
yield from _process_content_streams(
|
||||
pdf=pdf, container=form_xobject, shorthand=ctm_shorthand
|
||||
pdf=pdf,
|
||||
container=form_xobject,
|
||||
shorthand=ctm_shorthand,
|
||||
initial_fill_ink=settings.fill_ink,
|
||||
)
|
||||
|
||||
|
||||
def _process_content_streams(
|
||||
*, pdf: Pdf, container: Object, shorthand=None
|
||||
*, pdf: Pdf, container: Object, shorthand=None, initial_fill_ink=Ink.mono
|
||||
) -> Iterator[VectorMarker | TextMarker | ImageInfo]:
|
||||
"""Find all individual instances of images drawn in the container.
|
||||
|
||||
@@ -377,7 +405,7 @@ def _process_content_streams(
|
||||
else:
|
||||
return
|
||||
|
||||
contentsinfo = _interpret_contents(container, initial_shorthand)
|
||||
contentsinfo = _interpret_contents(container, initial_shorthand, initial_fill_ink)
|
||||
|
||||
if contentsinfo.found_vector:
|
||||
yield VectorMarker()
|
||||
|
||||
@@ -39,6 +39,20 @@ class Encoding(Enum):
|
||||
flate_jpeg = auto()
|
||||
|
||||
|
||||
class Ink(Enum):
|
||||
"""Classification of the fill color used to paint a stencil image mask.
|
||||
|
||||
A stencil (image mask) is painted with the current fill color, so the
|
||||
color depth needed to rasterize it for OCR depends on that fill color,
|
||||
not on the mask's 1-bit data.
|
||||
"""
|
||||
|
||||
# pylint: disable=invalid-name
|
||||
mono = auto() # black (or no color information to preserve)
|
||||
gray = auto() # achromatic but not pure black
|
||||
color = auto() # chromatic, or a fill we cannot prove is achromatic
|
||||
|
||||
|
||||
FloatRect = tuple[float, float, float, float]
|
||||
|
||||
FRIENDLY_COLORSPACE: dict[str, Colorspace] = {
|
||||
|
||||
@@ -19,6 +19,7 @@ from pdfminer.layout import LTPage, LTTextBox
|
||||
from pikepdf import Name, Page, Pdf
|
||||
|
||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||
from ocrmypdf._pageboxes import coerce_box
|
||||
from ocrmypdf.exceptions import EncryptedPdfError
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pdfinfo._contentstream import TextboxInfo, TextMarker, VectorMarker
|
||||
@@ -34,6 +35,12 @@ from ocrmypdf.pdfinfo.layout import (
|
||||
logger = logging.getLogger()
|
||||
|
||||
|
||||
def _box_rect(values: Iterable) -> FloatRect:
|
||||
"""Coerce a page box to a normalized ``FloatRect`` (4-tuple)."""
|
||||
b = coerce_box(values)
|
||||
return (b[0], b[1], b[2], b[3])
|
||||
|
||||
|
||||
def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) -> bool:
|
||||
"""Smarter text detection that ignores text in margins."""
|
||||
pw, ph = float(page_width), float(page_height) # pylint: disable=invalid-name
|
||||
@@ -140,15 +147,15 @@ class PageInfo:
|
||||
miner_state: PdfMinerState | None,
|
||||
):
|
||||
page: Page = pdf.pages[pageno]
|
||||
mediabox = [Decimal(d) for d in page.mediabox.as_list()]
|
||||
mediabox = [Decimal(str(d)) for d in coerce_box(page.mediabox.as_list())]
|
||||
width_pt = mediabox[2] - mediabox[0]
|
||||
height_pt = mediabox[3] - mediabox[1]
|
||||
|
||||
self._artbox = [float(d) for d in page.artbox.as_list()]
|
||||
self._bleedbox = [float(d) for d in page.bleedbox.as_list()]
|
||||
self._cropbox = [float(d) for d in page.cropbox.as_list()]
|
||||
self._mediabox = [float(d) for d in page.mediabox.as_list()]
|
||||
self._trimbox = [float(d) for d in page.trimbox.as_list()]
|
||||
self._artbox = _box_rect(page.artbox.as_list())
|
||||
self._bleedbox = _box_rect(page.bleedbox.as_list())
|
||||
self._cropbox = _box_rect(page.cropbox.as_list())
|
||||
self._mediabox = _box_rect(page.mediabox.as_list())
|
||||
self._trimbox = _box_rect(page.trimbox.as_list())
|
||||
|
||||
check_this_page = pageno in check_pages
|
||||
|
||||
@@ -398,6 +405,7 @@ class PdfInfo:
|
||||
_has_acroform: bool = False
|
||||
_has_signature: bool = False
|
||||
_needs_rendering: bool = False
|
||||
_has_structure_tree: bool = False
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
@@ -449,6 +457,7 @@ class PdfInfo:
|
||||
self._is_tagged = bool(
|
||||
pdf.Root.get(Name.MarkInfo, {}).get(Name.Marked, False)
|
||||
)
|
||||
self._has_structure_tree = Name.StructTreeRoot in pdf.Root
|
||||
|
||||
@property
|
||||
def pages(self) -> list[PageInfo | None]:
|
||||
@@ -481,6 +490,11 @@ class PdfInfo:
|
||||
"""Return True if the document catalog indicates this is a Tagged PDF."""
|
||||
return self._is_tagged
|
||||
|
||||
@property
|
||||
def has_structure_tree(self) -> bool:
|
||||
"""Return True if the document catalog has a logical structure tree."""
|
||||
return self._has_structure_tree
|
||||
|
||||
@property
|
||||
def filename(self) -> str | Path:
|
||||
"""Return filename of PDF."""
|
||||
|
||||
@@ -17,7 +17,6 @@ import pdfminer
|
||||
import pdfminer.encodingdb
|
||||
import pdfminer.pdfdevice
|
||||
import pdfminer.pdfinterp
|
||||
import pdfminer.psparser
|
||||
from deprecation import deprecated
|
||||
from pdfminer.converter import PDFLayoutAnalyzer
|
||||
from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox
|
||||
@@ -60,12 +59,6 @@ def pdfsimplefont__init__(
|
||||
|
||||
PDFSimpleFont.__init__ = pdfsimplefont__init__
|
||||
|
||||
# Patch pdfminer.six buffer size
|
||||
# The parser doesn't properly handle keyword tokens are split across the end of the
|
||||
# buffer, so increase the buffer size something far larger than will ever be seen.
|
||||
pdfminer.psparser.PSBaseParser.BUFSIZ = 256 * 1024 * 1024
|
||||
|
||||
|
||||
def pdftype3font__pscript5_get_height(self):
|
||||
"""Monkeypatch for PScript5.dll PDFs.
|
||||
|
||||
|
||||
@@ -38,6 +38,7 @@ class GhostscriptRasterDevice(StrEnum):
|
||||
JPEGGRAY = 'jpeggray'
|
||||
JPEGCOLOR = 'jpeg'
|
||||
PNGMONO = 'pngmono'
|
||||
PNGMONOD = 'pngmonod'
|
||||
PNGGRAY = 'pnggray'
|
||||
PNG256 = 'png256'
|
||||
PNG16M = 'png16m'
|
||||
|
||||
@@ -1,345 +1,31 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Wrappers to manage subprocess calls."""
|
||||
"""Wrappers to manage subprocess calls.
|
||||
|
||||
This package is split into three private submodules by concern:
|
||||
|
||||
- :mod:`ocrmypdf.subprocess._run` - low-level execution wrappers (``run``,
|
||||
``run_polling_stderr``) that add OCRmyPDF-aware logging and Windows PATH
|
||||
resolution. Useful as drop-in replacements for :func:`subprocess.run`.
|
||||
- :mod:`ocrmypdf.subprocess._version` - version probing (``get_version``).
|
||||
- :mod:`ocrmypdf.subprocess._check` - startup validation
|
||||
(``check_external_program``) with platform-aware error messages.
|
||||
|
||||
The names below are the stable public API. Importing from the private
|
||||
submodules directly is not supported for external code.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
from contextlib import suppress
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||
from subprocess import run as subprocess_run
|
||||
from ocrmypdf.subprocess._check import check_external_program
|
||||
from ocrmypdf.subprocess._run import Args, Environ, run, run_polling_stderr
|
||||
from ocrmypdf.subprocess._version import get_version
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
Args = Sequence[Path | str]
|
||||
Environ = Mapping[str, str] | os._Environ # pylint: disable=protected-access
|
||||
|
||||
|
||||
def run(
|
||||
args: Args,
|
||||
*,
|
||||
env: Environ | None = None,
|
||||
logs_errors_to_stdout: bool = False,
|
||||
check: bool = False,
|
||||
**kwargs,
|
||||
) -> CompletedProcess:
|
||||
"""Wrapper around :py:func:`subprocess.run`.
|
||||
|
||||
The main purpose of this wrapper is to log subprocess output in an orderly
|
||||
fashion that identifies the responsible subprocess. An additional
|
||||
task is that this function goes to greater lengths to find possible Windows
|
||||
locations of our dependencies when they are not on the system PATH.
|
||||
|
||||
Arguments should be identical to ``subprocess.run``, except for following:
|
||||
|
||||
Args:
|
||||
args: Positional arguments to pass to ``subprocess.run``.
|
||||
env: A set of environment variables. If None, the OS environment is used.
|
||||
logs_errors_to_stdout: If True, indicates that the process writes its error
|
||||
messages to stdout rather than stderr, so stdout should be logged
|
||||
if there is an error. If False, stderr is logged. Could be used with
|
||||
stderr=STDOUT, stdout=PIPE for example.
|
||||
check: If True, raise an exception if the process exits with a non-zero
|
||||
status code. If False, the return value will indicate success or failure.
|
||||
kwargs: Additional arguments to pass to ``subprocess.run``.
|
||||
"""
|
||||
args, env, process_log, _text = _fix_process_args(args, env, kwargs)
|
||||
|
||||
stderr = None
|
||||
stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout'
|
||||
try:
|
||||
proc = subprocess_run(args, env=env, check=check, **kwargs)
|
||||
except CalledProcessError as e:
|
||||
stderr = getattr(e, stderr_name, None)
|
||||
raise
|
||||
else:
|
||||
stderr = getattr(proc, stderr_name, None)
|
||||
finally:
|
||||
if process_log.isEnabledFor(logging.DEBUG) and stderr:
|
||||
with suppress(AttributeError, UnicodeDecodeError):
|
||||
stderr = stderr.decode('utf-8', 'replace')
|
||||
if logs_errors_to_stdout:
|
||||
process_log.debug("stdout/stderr = %s", stderr)
|
||||
else:
|
||||
process_log.debug("stderr = %s", stderr)
|
||||
return proc
|
||||
|
||||
|
||||
def run_polling_stderr(
|
||||
args: Args,
|
||||
*,
|
||||
callback: Callable[[str], None],
|
||||
check: bool = False,
|
||||
env: Environ | None = None,
|
||||
**kwargs,
|
||||
) -> CompletedProcess:
|
||||
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
||||
|
||||
Every line of produced by stderr will be forwarded to the callback function.
|
||||
The intended use is monitoring progress of subprocesses that output their
|
||||
own progress indicators. In addition, each line will be logged if debug
|
||||
logging is enabled.
|
||||
|
||||
Requires stderr to be opened in text mode for ease of handling errors. In
|
||||
addition the expected encoding= and errors= arguments should be set. Note
|
||||
that if stdout is already set up, it need not be binary.
|
||||
"""
|
||||
args, env, process_log, text = _fix_process_args(args, env, kwargs)
|
||||
assert text, "Must use text=True"
|
||||
|
||||
with Popen(args, env=env, **kwargs) as proc:
|
||||
lines = []
|
||||
while proc.poll() is None:
|
||||
if proc.stderr is None:
|
||||
continue
|
||||
for msg in iter(proc.stderr.readline, ''):
|
||||
if process_log.isEnabledFor(logging.DEBUG):
|
||||
process_log.debug(msg.strip())
|
||||
callback(msg)
|
||||
lines.append(msg)
|
||||
stderr = ''.join(lines)
|
||||
|
||||
if check and proc.returncode != 0:
|
||||
raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr)
|
||||
return CompletedProcess(args, proc.returncode, None, stderr=stderr)
|
||||
|
||||
|
||||
def _fix_process_args(
|
||||
args: Args, env: Environ | None, kwargs
|
||||
) -> tuple[Args, Environ, logging.Logger, bool]:
|
||||
if not env:
|
||||
env = os.environ
|
||||
|
||||
# Search in spoof path if necessary
|
||||
program = str(args[0])
|
||||
|
||||
if sys.platform == 'win32':
|
||||
# pylint: disable=import-outside-toplevel
|
||||
from ocrmypdf.subprocess._windows import fix_windows_args
|
||||
|
||||
args = fix_windows_args(program, args, env)
|
||||
|
||||
log.debug("Running: %s", args)
|
||||
process_log = log.getChild(os.path.basename(program))
|
||||
text = bool(kwargs.get('text', False))
|
||||
|
||||
return args, env, process_log, text
|
||||
|
||||
|
||||
def get_version(
|
||||
program: str,
|
||||
*,
|
||||
version_arg: str = '--version',
|
||||
regex=r'(\d+(\.\d+)*)',
|
||||
env: Environ | None = None,
|
||||
) -> str:
|
||||
"""Get the version of the specified program.
|
||||
|
||||
Arguments:
|
||||
program: The program to version check.
|
||||
version_arg: The argument needed to ask for its version, e.g. ``--version``.
|
||||
regex: A regular expression to parse the program's output and obtain the
|
||||
version.
|
||||
env: Custom ``os.environ`` in which to run program.
|
||||
"""
|
||||
args_prog = [program, version_arg]
|
||||
try:
|
||||
proc = run(
|
||||
args_prog,
|
||||
close_fds=True,
|
||||
text=True,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
check=True,
|
||||
env=env,
|
||||
)
|
||||
output: str = proc.stdout
|
||||
except FileNotFoundError as e:
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
except CalledProcessError as e:
|
||||
if e.returncode != 0:
|
||||
log.exception(e)
|
||||
raise MissingDependencyError(
|
||||
f"Ran program '{program}' but it exited with an error:\n{e.output}"
|
||||
) from e
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
|
||||
match = re.match(regex, output.strip())
|
||||
if not match:
|
||||
raise MissingDependencyError(
|
||||
f"The program '{program}' did not report its version. "
|
||||
f"Message was:\n{output}"
|
||||
)
|
||||
version = match.group(1)
|
||||
|
||||
return version
|
||||
|
||||
|
||||
MISSING_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH.
|
||||
'''
|
||||
|
||||
MISSING_OPTIONAL_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH. This program is required when you use the
|
||||
{required_for} arguments. You could try omitting these arguments, or install
|
||||
the package.
|
||||
'''
|
||||
|
||||
MISSING_RECOMMEND_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH. This program is recommended when using the {required_for} arguments,
|
||||
but not required, so we will proceed. For best results, install the program.
|
||||
'''
|
||||
|
||||
OLD_VERSION = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher. Your system appears
|
||||
to have {found_version}. Please update this program.
|
||||
'''
|
||||
|
||||
OLD_VERSION_REQUIRED_FOR = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
||||
{required_for} arguments. {program} {found_version} is installed.
|
||||
|
||||
If you omit these arguments, OCRmyPDF may be able to
|
||||
proceed. For best results, update the program.
|
||||
'''
|
||||
|
||||
OSX_INSTALL_ADVICE = '''
|
||||
If you have homebrew installed, try these command to install the missing
|
||||
package:
|
||||
brew install {package}
|
||||
'''
|
||||
|
||||
LINUX_INSTALL_ADVICE = '''
|
||||
On systems with the aptitude package manager (Debian, Ubuntu), try these
|
||||
commands:
|
||||
sudo apt update
|
||||
sudo apt install {package}
|
||||
|
||||
On RPM-based systems (Red Hat, Fedora), try this command:
|
||||
sudo dnf install {package}
|
||||
'''
|
||||
|
||||
WINDOWS_INSTALL_ADVICE = '''
|
||||
If not already installed, install the Chocolatey package manager. Then use
|
||||
a command prompt to install the missing package:
|
||||
choco install {package}
|
||||
'''
|
||||
|
||||
|
||||
def _get_platform() -> str:
|
||||
if sys.platform.startswith('freebsd'):
|
||||
return 'freebsd'
|
||||
elif sys.platform.startswith('linux'):
|
||||
return 'linux'
|
||||
elif sys.platform.startswith('win'):
|
||||
return 'windows'
|
||||
return sys.platform
|
||||
|
||||
|
||||
def _error_trailer(program: str, package: str | Mapping[str, str], **kwargs) -> None:
|
||||
del kwargs
|
||||
if isinstance(package, Mapping):
|
||||
package = package.get(_get_platform(), program)
|
||||
|
||||
if _get_platform() == 'darwin':
|
||||
log.info(OSX_INSTALL_ADVICE.format(**locals()))
|
||||
elif _get_platform() == 'linux':
|
||||
log.info(LINUX_INSTALL_ADVICE.format(**locals()))
|
||||
elif _get_platform() == 'windows':
|
||||
log.info(WINDOWS_INSTALL_ADVICE.format(**locals()))
|
||||
|
||||
|
||||
def _error_missing_program(
|
||||
program: str, package: str, required_for: str | None, recommended: bool
|
||||
) -> None:
|
||||
# pylint: disable=unused-argument
|
||||
if recommended:
|
||||
log.warning(MISSING_RECOMMEND_PROGRAM.format(**locals()))
|
||||
elif required_for:
|
||||
log.error(MISSING_OPTIONAL_PROGRAM.format(**locals()))
|
||||
else:
|
||||
log.error(MISSING_PROGRAM.format(**locals()))
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def _error_old_version(
|
||||
program: str,
|
||||
package: str,
|
||||
need_version: str,
|
||||
found_version: str,
|
||||
required_for: str | None,
|
||||
) -> None:
|
||||
# pylint: disable=unused-argument
|
||||
if required_for:
|
||||
log.error(OLD_VERSION_REQUIRED_FOR.format(**locals()))
|
||||
else:
|
||||
log.error(OLD_VERSION.format(**locals()))
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def check_external_program(
|
||||
*,
|
||||
program: str,
|
||||
package: str,
|
||||
version_checker: Callable[[], Version],
|
||||
need_version: str | Version,
|
||||
required_for: str | None = None,
|
||||
recommended: bool = False,
|
||||
version_parser: type[Version] = Version,
|
||||
) -> None:
|
||||
"""Check for required version of external program and raise exception if not.
|
||||
|
||||
Args:
|
||||
program: The name of the program to test.
|
||||
package: The name of a software package that typically supplies this program.
|
||||
Usually the same as program.
|
||||
version_checker: A callable without arguments that retrieves the installed
|
||||
version of program.
|
||||
need_version: The minimum required version.
|
||||
required_for: The name of an argument of feature that requires this program.
|
||||
recommended: If this external program is recommended, instead of raising
|
||||
an exception, log a warning and allow execution to continue.
|
||||
version_parser: A class that should be used to parse and compare version
|
||||
numbers. Used when version numbers do not follow standard conventions.
|
||||
"""
|
||||
if not isinstance(need_version, Version):
|
||||
need_version = version_parser(need_version)
|
||||
try:
|
||||
found_version = version_checker()
|
||||
except (CalledProcessError, FileNotFoundError) as e:
|
||||
_error_missing_program(program, package, required_for, recommended)
|
||||
if not recommended:
|
||||
raise MissingDependencyError(program) from e
|
||||
return
|
||||
except MissingDependencyError:
|
||||
_error_missing_program(program, package, required_for, recommended)
|
||||
if not recommended:
|
||||
raise
|
||||
return
|
||||
|
||||
if found_version and found_version < need_version:
|
||||
_error_old_version(
|
||||
program, package, str(need_version), str(found_version), required_for
|
||||
)
|
||||
if not recommended:
|
||||
raise MissingDependencyError(program)
|
||||
|
||||
log.debug('Found %s %s', program, found_version)
|
||||
__all__ = [
|
||||
'Args',
|
||||
'Environ',
|
||||
'check_external_program',
|
||||
'get_version',
|
||||
'run',
|
||||
'run_polling_stderr',
|
||||
]
|
||||
|
||||
@@ -0,0 +1,171 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Validate that required external programs are installed and new enough."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import sys
|
||||
from collections.abc import Callable, Mapping
|
||||
from subprocess import CalledProcessError
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
|
||||
log = logging.getLogger('ocrmypdf.subprocess')
|
||||
|
||||
|
||||
MISSING_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH.
|
||||
'''
|
||||
|
||||
MISSING_OPTIONAL_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH. This program is required when you use the
|
||||
{required_for} arguments. You could try omitting these arguments, or install
|
||||
the package.
|
||||
'''
|
||||
|
||||
MISSING_RECOMMEND_PROGRAM = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH. This program is recommended when using the {required_for} arguments,
|
||||
but not required, so we will proceed. For best results, install the program.
|
||||
'''
|
||||
|
||||
OLD_VERSION = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher. Your system appears
|
||||
to have {found_version}. Please update this program.
|
||||
'''
|
||||
|
||||
OLD_VERSION_REQUIRED_FOR = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
||||
{required_for} arguments. {program} {found_version} is installed.
|
||||
|
||||
If you omit these arguments, OCRmyPDF may be able to
|
||||
proceed. For best results, update the program.
|
||||
'''
|
||||
|
||||
OSX_INSTALL_ADVICE = '''
|
||||
If you have homebrew installed, try these command to install the missing
|
||||
package:
|
||||
brew install {package}
|
||||
'''
|
||||
|
||||
LINUX_INSTALL_ADVICE = '''
|
||||
On systems with the aptitude package manager (Debian, Ubuntu), try these
|
||||
commands:
|
||||
sudo apt update
|
||||
sudo apt install {package}
|
||||
|
||||
On RPM-based systems (Red Hat, Fedora), try this command:
|
||||
sudo dnf install {package}
|
||||
'''
|
||||
|
||||
WINDOWS_INSTALL_ADVICE = '''
|
||||
If not already installed, install the Chocolatey package manager. Then use
|
||||
a command prompt to install the missing package:
|
||||
choco install {package}
|
||||
'''
|
||||
|
||||
|
||||
def _get_platform() -> str:
|
||||
if sys.platform.startswith('freebsd'):
|
||||
return 'freebsd'
|
||||
elif sys.platform.startswith('linux'):
|
||||
return 'linux'
|
||||
elif sys.platform.startswith('win'):
|
||||
return 'windows'
|
||||
return sys.platform
|
||||
|
||||
|
||||
def _error_trailer(program: str, package: str | Mapping[str, str], **kwargs) -> None:
|
||||
del kwargs
|
||||
if isinstance(package, Mapping):
|
||||
package = package.get(_get_platform(), program)
|
||||
|
||||
if _get_platform() == 'darwin':
|
||||
log.info(OSX_INSTALL_ADVICE.format(**locals()))
|
||||
elif _get_platform() == 'linux':
|
||||
log.info(LINUX_INSTALL_ADVICE.format(**locals()))
|
||||
elif _get_platform() == 'windows':
|
||||
log.info(WINDOWS_INSTALL_ADVICE.format(**locals()))
|
||||
|
||||
|
||||
def _error_missing_program(
|
||||
program: str, package: str, required_for: str | None, recommended: bool
|
||||
) -> None:
|
||||
# pylint: disable=unused-argument
|
||||
if recommended:
|
||||
log.warning(MISSING_RECOMMEND_PROGRAM.format(**locals()))
|
||||
elif required_for:
|
||||
log.error(MISSING_OPTIONAL_PROGRAM.format(**locals()))
|
||||
else:
|
||||
log.error(MISSING_PROGRAM.format(**locals()))
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def _error_old_version(
|
||||
program: str,
|
||||
package: str,
|
||||
need_version: str,
|
||||
found_version: str,
|
||||
required_for: str | None,
|
||||
) -> None:
|
||||
# pylint: disable=unused-argument
|
||||
if required_for:
|
||||
log.error(OLD_VERSION_REQUIRED_FOR.format(**locals()))
|
||||
else:
|
||||
log.error(OLD_VERSION.format(**locals()))
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def check_external_program(
|
||||
*,
|
||||
program: str,
|
||||
package: str,
|
||||
version_checker: Callable[[], Version],
|
||||
need_version: str | Version,
|
||||
required_for: str | None = None,
|
||||
recommended: bool = False,
|
||||
version_parser: type[Version] = Version,
|
||||
) -> None:
|
||||
"""Check for required version of external program and raise exception if not.
|
||||
|
||||
Args:
|
||||
program: The name of the program to test.
|
||||
package: The name of a software package that typically supplies this program.
|
||||
Usually the same as program.
|
||||
version_checker: A callable without arguments that retrieves the installed
|
||||
version of program.
|
||||
need_version: The minimum required version.
|
||||
required_for: The name of an argument of feature that requires this program.
|
||||
recommended: If this external program is recommended, instead of raising
|
||||
an exception, log a warning and allow execution to continue.
|
||||
version_parser: A class that should be used to parse and compare version
|
||||
numbers. Used when version numbers do not follow standard conventions.
|
||||
"""
|
||||
if not isinstance(need_version, Version):
|
||||
need_version = version_parser(need_version)
|
||||
try:
|
||||
found_version = version_checker()
|
||||
except (CalledProcessError, FileNotFoundError) as e:
|
||||
_error_missing_program(program, package, required_for, recommended)
|
||||
if not recommended:
|
||||
raise MissingDependencyError(program) from e
|
||||
return
|
||||
except MissingDependencyError:
|
||||
_error_missing_program(program, package, required_for, recommended)
|
||||
if not recommended:
|
||||
raise
|
||||
return
|
||||
|
||||
if found_version and found_version < need_version:
|
||||
_error_old_version(
|
||||
program, package, str(need_version), str(found_version), required_for
|
||||
)
|
||||
if not recommended:
|
||||
raise MissingDependencyError(program)
|
||||
|
||||
log.debug('Found %s %s', program, found_version)
|
||||
@@ -0,0 +1,137 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Low-level wrappers around :py:mod:`subprocess`.
|
||||
|
||||
These functions exist to give OCRmyPDF child processes uniform logging
|
||||
behavior and to route through any platform-specific PATH fix-ups before
|
||||
invocation. They are intended as drop-in replacements for
|
||||
:py:func:`subprocess.run` in contexts where that routing is desirable
|
||||
(for example, plugin-provided tools).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from collections.abc import Callable, Mapping, Sequence
|
||||
from contextlib import suppress
|
||||
from pathlib import Path
|
||||
from subprocess import CalledProcessError, CompletedProcess, Popen
|
||||
from subprocess import run as subprocess_run
|
||||
|
||||
log = logging.getLogger('ocrmypdf.subprocess')
|
||||
|
||||
Args = Sequence[Path | str]
|
||||
Environ = Mapping[str, str] | os._Environ # pylint: disable=protected-access
|
||||
|
||||
|
||||
def run(
|
||||
args: Args,
|
||||
*,
|
||||
env: Environ | None = None,
|
||||
logs_errors_to_stdout: bool = False,
|
||||
check: bool = False,
|
||||
**kwargs,
|
||||
) -> CompletedProcess:
|
||||
"""Wrapper around :py:func:`subprocess.run`.
|
||||
|
||||
The main purpose of this wrapper is to log subprocess output in an orderly
|
||||
fashion that identifies the responsible subprocess. An additional
|
||||
task is that this function goes to greater lengths to find possible Windows
|
||||
locations of our dependencies when they are not on the system PATH.
|
||||
|
||||
Arguments should be identical to ``subprocess.run``, except for following:
|
||||
|
||||
Args:
|
||||
args: Positional arguments to pass to ``subprocess.run``.
|
||||
env: A set of environment variables. If None, the OS environment is used.
|
||||
logs_errors_to_stdout: If True, indicates that the process writes its error
|
||||
messages to stdout rather than stderr, so stdout should be logged
|
||||
if there is an error. If False, stderr is logged. Could be used with
|
||||
stderr=STDOUT, stdout=PIPE for example.
|
||||
check: If True, raise an exception if the process exits with a non-zero
|
||||
status code. If False, the return value will indicate success or failure.
|
||||
kwargs: Additional arguments to pass to ``subprocess.run``.
|
||||
"""
|
||||
args, env, process_log, _text = _fix_process_args(args, env, kwargs)
|
||||
|
||||
stderr = None
|
||||
stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout'
|
||||
try:
|
||||
proc = subprocess_run(args, env=env, check=check, **kwargs)
|
||||
except CalledProcessError as e:
|
||||
stderr = getattr(e, stderr_name, None)
|
||||
raise
|
||||
else:
|
||||
stderr = getattr(proc, stderr_name, None)
|
||||
finally:
|
||||
if process_log.isEnabledFor(logging.DEBUG) and stderr:
|
||||
with suppress(AttributeError, UnicodeDecodeError):
|
||||
stderr = stderr.decode('utf-8', 'replace')
|
||||
if logs_errors_to_stdout:
|
||||
process_log.debug("stdout/stderr = %s", stderr)
|
||||
else:
|
||||
process_log.debug("stderr = %s", stderr)
|
||||
return proc
|
||||
|
||||
|
||||
def run_polling_stderr(
|
||||
args: Args,
|
||||
*,
|
||||
callback: Callable[[str], None],
|
||||
check: bool = False,
|
||||
env: Environ | None = None,
|
||||
**kwargs,
|
||||
) -> CompletedProcess:
|
||||
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
||||
|
||||
Every line of produced by stderr will be forwarded to the callback function.
|
||||
The intended use is monitoring progress of subprocesses that output their
|
||||
own progress indicators. In addition, each line will be logged if debug
|
||||
logging is enabled.
|
||||
|
||||
Requires stderr to be opened in text mode for ease of handling errors. In
|
||||
addition the expected encoding= and errors= arguments should be set. Note
|
||||
that if stdout is already set up, it need not be binary.
|
||||
"""
|
||||
args, env, process_log, text = _fix_process_args(args, env, kwargs)
|
||||
assert text, "Must use text=True"
|
||||
|
||||
with Popen(args, env=env, **kwargs) as proc:
|
||||
lines = []
|
||||
while proc.poll() is None:
|
||||
if proc.stderr is None:
|
||||
continue
|
||||
for msg in iter(proc.stderr.readline, ''):
|
||||
if process_log.isEnabledFor(logging.DEBUG):
|
||||
process_log.debug(msg.strip())
|
||||
callback(msg)
|
||||
lines.append(msg)
|
||||
stderr = ''.join(lines)
|
||||
|
||||
if check and proc.returncode != 0:
|
||||
raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr)
|
||||
return CompletedProcess(args, proc.returncode, None, stderr=stderr)
|
||||
|
||||
|
||||
def _fix_process_args(
|
||||
args: Args, env: Environ | None, kwargs
|
||||
) -> tuple[Args, Environ, logging.Logger, bool]:
|
||||
if not env:
|
||||
env = os.environ
|
||||
|
||||
# Search in spoof path if necessary
|
||||
program = str(args[0])
|
||||
|
||||
if sys.platform == 'win32':
|
||||
# pylint: disable=import-outside-toplevel
|
||||
from ocrmypdf.subprocess._windows import fix_windows_args
|
||||
|
||||
args = fix_windows_args(program, args, env)
|
||||
|
||||
log.debug("Running: %s", args)
|
||||
process_log = log.getChild(os.path.basename(program))
|
||||
text = bool(kwargs.get('text', False))
|
||||
|
||||
return args, env, process_log, text
|
||||
@@ -0,0 +1,79 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Extract version strings from external programs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import re
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess._run import Environ
|
||||
|
||||
log = logging.getLogger('ocrmypdf.subprocess')
|
||||
|
||||
|
||||
def get_version(
|
||||
program: str,
|
||||
*,
|
||||
version_arg: str = '--version',
|
||||
regex=r'(\d+(\.\d+)*)',
|
||||
env: Environ | None = None,
|
||||
) -> str:
|
||||
"""Get the version of the specified program.
|
||||
|
||||
Arguments:
|
||||
program: The program to version check.
|
||||
version_arg: The argument needed to ask for its version, e.g. ``--version``.
|
||||
regex: A regular expression to parse the program's output and obtain the
|
||||
version.
|
||||
env: Custom ``os.environ`` in which to run program.
|
||||
"""
|
||||
# Late import of the public ``run`` so that tests patching
|
||||
# ``ocrmypdf.subprocess.run`` affect this function. Binding ``run`` at
|
||||
# module load time would capture the real implementation and bypass the
|
||||
# patch.
|
||||
from ocrmypdf import subprocess as _sp
|
||||
|
||||
args_prog = [program, version_arg]
|
||||
try:
|
||||
proc = _sp.run(
|
||||
args_prog,
|
||||
close_fds=True,
|
||||
text=True,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
check=True,
|
||||
env=env,
|
||||
)
|
||||
output: str = proc.stdout
|
||||
except FileNotFoundError as e:
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
except CalledProcessError as e:
|
||||
if e.returncode != 0:
|
||||
log.exception(e)
|
||||
raise MissingDependencyError(
|
||||
f"Ran program '{program}' but it exited with an error:\n{e.output}"
|
||||
) from e
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
|
||||
# Some tools (e.g. veraPDF launched on a recent JDK) print warnings before
|
||||
# the version line, so scan each line rather than only the start of output.
|
||||
version = None
|
||||
for line in output.splitlines():
|
||||
match = re.match(regex, line.strip())
|
||||
if match:
|
||||
version = match.group(1)
|
||||
break
|
||||
if version is None:
|
||||
raise MissingDependencyError(
|
||||
f"The program '{program}' did not report its version. "
|
||||
f"Message was:\n{output}"
|
||||
)
|
||||
|
||||
return version
|
||||
@@ -0,0 +1,42 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MIT
|
||||
"""Test plugin that deliberately writes garbage to stdout.
|
||||
|
||||
Used to verify that OCRmyPDF's stdout protection diverts stray writes (from
|
||||
plugins or libraries) to stderr, so that a PDF written to stdout is never
|
||||
corrupted. Pollutes at three points: plugin import (main process), the
|
||||
``validate`` hook (main process), and the ``filter_ocr_image`` hook (worker
|
||||
process/thread).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
|
||||
from ocrmypdf import hookimpl
|
||||
|
||||
POLLUTION = b'POLLUTION'
|
||||
|
||||
|
||||
def _pollute(where: bytes) -> None:
|
||||
# Write to file descriptor 1 directly (as a careless C library might) and
|
||||
# via Python's sys.stdout (as a stray print() might).
|
||||
os.write(1, POLLUTION + b'-fd1-' + where + b'\n')
|
||||
print(POLLUTION.decode() + '-stdout-' + where.decode())
|
||||
sys.stdout.flush()
|
||||
|
||||
|
||||
# Pollute at import time, which happens while plugins are being loaded.
|
||||
_pollute(b'import')
|
||||
|
||||
|
||||
@hookimpl
|
||||
def validate(pdfinfo, options):
|
||||
_pollute(b'validate')
|
||||
|
||||
|
||||
@hookimpl
|
||||
def filter_ocr_image(page, image):
|
||||
_pollute(b'filter_ocr_image')
|
||||
return image
|
||||
@@ -76,6 +76,9 @@ the copyright holder(s) and license(s) applicable to these resources.
|
||||
* - missing_docinfo.pdf
|
||||
- synthetic
|
||||
- PDF file with no /DocumentInfo section
|
||||
* - docinfo_latin1_key.pdf
|
||||
- synthetic
|
||||
- PDF whose /DocumentInfo dictionary has a /Name key with Latin-1 bytes (/Saks#e5r) that is not valid UTF-8
|
||||
* - overlay.pdf
|
||||
- synthetic
|
||||
- PDF file generated by PDFPen pro that triggered content stream parse errors
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
%PDF-1.3
|
||||
%¿÷¢þ
|
||||
1 0 obj
|
||||
<< /Pages 3 0 R /Type /Catalog >>
|
||||
endobj
|
||||
2 0 obj
|
||||
<< /Author (Geomatikk AS) /Beskrivelse () /Creator (OCRmyPDF 16.10.0 / EasyOCR-PDF 1.7.2) /CreatorVersion (6.36.0.918) /Dokumentidplanreg () /Enhetsnavn () /Hyperlink (1) /Opprinnelse () /Producer (pikepdf 9.5.2) /RegistrationDate (N/A) /Saksansvarlig#20enhet () /Saksbehandler () /Saksnr () /Saks#e5r () /Status () >>
|
||||
endobj
|
||||
3 0 obj
|
||||
<< /Count 1 /Kids [ 4 0 R ] /Type /Pages >>
|
||||
endobj
|
||||
4 0 obj
|
||||
<< /Contents 5 0 R /MediaBox [ 0 0 612 792 ] /Parent 3 0 R /Resources << >> /Type /Page >>
|
||||
endobj
|
||||
5 0 obj
|
||||
<< /Length 0 /Filter /FlateDecode >>
|
||||
stream
|
||||
|
||||
endstream
|
||||
endobj
|
||||
xref
|
||||
0 6
|
||||
0000000000 65535 f
|
||||
0000000015 00000 n
|
||||
0000000064 00000 n
|
||||
0000000398 00000 n
|
||||
0000000457 00000 n
|
||||
0000000563 00000 n
|
||||
trailer << /Info 2 0 R /Root 1 0 R /Size 6 /ID [<c5231b8cfab9c82526c0da7475add5da><c5231b8cfab9c82526c0da7475add5da>] >>
|
||||
startxref
|
||||
633
|
||||
%%EOF
|
||||
+327
-5
@@ -17,7 +17,11 @@ from PIL import Image, UnidentifiedImageError
|
||||
|
||||
from ocrmypdf._exec import ghostscript
|
||||
from ocrmypdf._exec.ghostscript import DuplicateFilter, rasterize_pdf
|
||||
from ocrmypdf.builtin_plugins.ghostscript import _repair_gs106_jpeg_corruption
|
||||
from ocrmypdf.builtin_plugins.ghostscript import (
|
||||
PdfaImageCompression,
|
||||
_repair_gs106_jpeg_corruption,
|
||||
_resolve_auto_compression,
|
||||
)
|
||||
from ocrmypdf.exceptions import ColorConversionNeededError, ExitCode, InputFileError
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
@@ -137,6 +141,185 @@ def test_rasterize_low_dpi_one_axis(francais, outdir):
|
||||
assert im.info['dpi'] == forced_dpi
|
||||
|
||||
|
||||
def _capture_rasterize_args(resources, outdir, raster_device):
|
||||
"""Run rasterize_pdf with the gs subprocess mocked; return the gs argv."""
|
||||
out = outdir / 'out.png'
|
||||
captured = {}
|
||||
|
||||
def fake_run(args, **kwargs):
|
||||
captured['args'] = list(args)
|
||||
# Produce a valid PNG so rasterize_pdf's post-processing succeeds.
|
||||
Image.new('RGB', (2, 2)).save(out)
|
||||
return subprocess.CompletedProcess(args, returncode=0, stdout=b'', stderr=b'')
|
||||
|
||||
with patch('ocrmypdf._exec.ghostscript.run', side_effect=fake_run):
|
||||
rasterize_pdf(
|
||||
resources / 'francais.pdf',
|
||||
out,
|
||||
raster_device=raster_device,
|
||||
raster_dpi=Resolution(150.0, 150.0),
|
||||
)
|
||||
return captured['args']
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'raster_device',
|
||||
[
|
||||
GhostscriptRasterDevice.PNGGRAY,
|
||||
GhostscriptRasterDevice.PNG256,
|
||||
GhostscriptRasterDevice.PNG16M,
|
||||
],
|
||||
)
|
||||
def test_rasterize_antialiases_contone_devices(resources, outdir, raster_device):
|
||||
"""Contone raster devices receive anti-aliasing flags to aid OCR.
|
||||
|
||||
Ghostscript 10.x renders aliased glyphs that OCR misreads as extra word
|
||||
breaks; -dTextAlphaBits/-dGraphicsAlphaBits markedly improve accuracy,
|
||||
especially for small fonts at moderate DPI (see issue #1439).
|
||||
"""
|
||||
args = _capture_rasterize_args(resources, outdir, raster_device)
|
||||
assert '-dTextAlphaBits=4' in args
|
||||
assert '-dGraphicsAlphaBits=4' in args
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'raster_device',
|
||||
[GhostscriptRasterDevice.PNGMONO, GhostscriptRasterDevice.PNGMONOD],
|
||||
)
|
||||
def test_rasterize_no_antialias_on_mono_devices(resources, outdir, raster_device):
|
||||
"""1-bit mono devices must not receive alpha-bit flags.
|
||||
|
||||
Older Ghostscript versions reject -dTextAlphaBits on 1-bit devices, and
|
||||
pngmonod performs its own anti-aliased downscaling.
|
||||
"""
|
||||
args = _capture_rasterize_args(resources, outdir, raster_device)
|
||||
assert not any(a.startswith('-dTextAlphaBits') for a in args)
|
||||
assert not any(a.startswith('-dGraphicsAlphaBits') for a in args)
|
||||
|
||||
|
||||
def test_generate_pdfa_default_jpeg_quality(outdir):
|
||||
"""When jpeg_quality is None, Ghostscript receives -dJPEGQ=95 (default)."""
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='auto',
|
||||
color_conversion_strategy='LeaveColorUnchanged',
|
||||
)
|
||||
|
||||
args = run_mock.call_args.args[0]
|
||||
assert '-dJPEGQ=95' in args
|
||||
# No downsample switches when jpeg_maxdpi is not set
|
||||
assert not any(a.startswith('-dDownsampleColorImages') for a in args)
|
||||
assert not any(a.startswith('-dColorImageResolution') for a in args)
|
||||
|
||||
|
||||
def test_generate_pdfa_uses_user_jpeg_quality(outdir):
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='jpeg',
|
||||
color_conversion_strategy='RGB',
|
||||
jpeg_quality=72,
|
||||
)
|
||||
|
||||
args = run_mock.call_args.args[0]
|
||||
assert '-dJPEGQ=72' in args
|
||||
assert '-dJPEGQ=95' not in args
|
||||
|
||||
|
||||
def test_generate_pdfa_jpeg_quality_zero_is_max_compression(outdir):
|
||||
"""Explicit jpeg_quality=0 must reach Ghostscript as -dJPEGQ=0.
|
||||
|
||||
Ghostscript accepts 0 as a valid quality value (maximum compression);
|
||||
it must not be silently replaced by the default 95.
|
||||
"""
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='jpeg',
|
||||
color_conversion_strategy='RGB',
|
||||
jpeg_quality=0,
|
||||
)
|
||||
|
||||
args = run_mock.call_args.args[0]
|
||||
assert '-dJPEGQ=0' in args
|
||||
assert '-dJPEGQ=95' not in args
|
||||
|
||||
|
||||
def test_generate_pdfa_honors_jpeg_maxdpi(outdir):
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='auto',
|
||||
color_conversion_strategy='LeaveColorUnchanged',
|
||||
jpeg_maxdpi=300,
|
||||
)
|
||||
|
||||
args = run_mock.call_args.args[0]
|
||||
assert '-dJPEGQ=95' in args
|
||||
assert '-dDownsampleColorImages=true' in args
|
||||
assert '-dColorImageDownsampleThreshold=1.0' in args
|
||||
assert '-dDownsampleGrayImages=true' in args
|
||||
assert '-dGrayImageDownsampleThreshold=1.0' in args
|
||||
assert '-dDownsampleMonoImages=true' in args
|
||||
assert '-dMonoImageDownsampleThreshold=1.0' in args
|
||||
assert '-dColorImageResolution=300' in args
|
||||
assert '-dGrayImageResolution=300' in args
|
||||
assert '-dMonoImageResolution=300' in args
|
||||
|
||||
|
||||
def test_ghostscript_jpeg_options_via_cli(resources, outpdf):
|
||||
"""End-to-end: CLI flags reach the ghostscript plugin namespace."""
|
||||
with patch(
|
||||
'ocrmypdf._exec.ghostscript.generate_pdfa',
|
||||
wraps=ghostscript.generate_pdfa,
|
||||
) as gen_mock:
|
||||
run_ocrmypdf_api(
|
||||
resources / 'francais.pdf',
|
||||
outpdf,
|
||||
'--output-type',
|
||||
'pdfa',
|
||||
'--ghostscript-jpeg-quality',
|
||||
'60',
|
||||
'--ghostscript-jpeg-maxdpi',
|
||||
'150',
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
assert gen_mock.called
|
||||
call_kwargs = gen_mock.call_args.kwargs
|
||||
assert call_kwargs['jpeg_quality'] == 60
|
||||
assert call_kwargs['jpeg_maxdpi'] == 150
|
||||
|
||||
|
||||
def test_gs_render_failure(resources, outpdf, caplog):
|
||||
exitcode = run_ocrmypdf_api(
|
||||
resources / 'blank.pdf',
|
||||
@@ -176,9 +359,9 @@ def test_ghostscript_pdfa_failure(resources, outpdf, caplog):
|
||||
'--plugin',
|
||||
'tests/plugins/gs_pdfa_failure.py',
|
||||
)
|
||||
assert (
|
||||
exitcode == ExitCode.pdfa_conversion_failed
|
||||
), "Unexpected return when PDF/A fails"
|
||||
assert exitcode == ExitCode.pdfa_conversion_failed, (
|
||||
"Unexpected return when PDF/A fails"
|
||||
)
|
||||
|
||||
|
||||
def test_ghostscript_feature_elision(resources, outpdf):
|
||||
@@ -204,6 +387,88 @@ def test_ghostscript_mandatory_color_conversion(resources, outpdf):
|
||||
)
|
||||
|
||||
|
||||
def _run_generate_pdfa_with_devicen_warning(outdir, color_conversion_strategy):
|
||||
"""Invoke generate_pdfa with Ghostscript mocked to emit the DeviceN warning.
|
||||
|
||||
Ghostscript emits this warning when it writes a DeviceN colorspace with an
|
||||
inappropriate alternate, i.e. when it could not normalize the colorspace for
|
||||
PDF/A. The output is then liable to render blank in viewers such as Adobe
|
||||
Reader (see issue #1187), regardless of which conversion strategy was
|
||||
requested.
|
||||
"""
|
||||
(outdir / 'input.pdf').write_bytes(b'%PDF-1.5\n%fake\n')
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'],
|
||||
returncode=0,
|
||||
stdout='',
|
||||
stderr='Attempting to write a DeviceN space with an inappropriate '
|
||||
'alternate, reverting to the alternate color space.',
|
||||
)
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='auto',
|
||||
color_conversion_strategy=color_conversion_strategy,
|
||||
)
|
||||
|
||||
|
||||
def test_devicen_warning_default_strategy_raises_with_guidance(outdir):
|
||||
"""Default (no conversion): raise and tell the user to pick a strategy."""
|
||||
with pytest.raises(ColorConversionNeededError) as exc_info:
|
||||
_run_generate_pdfa_with_devicen_warning(outdir, 'LeaveColorUnchanged')
|
||||
message = str(exc_info.value)
|
||||
assert '--color-conversion-strategy' in message
|
||||
assert 'RGB' in message
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'strategy',
|
||||
[
|
||||
# A strategy that genuinely cannot fix the colorspace; confirmed in #1187.
|
||||
'UseDeviceIndependentColor',
|
||||
# A normally-effective strategy that nonetheless failed on this input:
|
||||
# if Ghostscript still warns, the output is still broken and we must not
|
||||
# silently pass it through (the behaviour PR #1692 would have introduced).
|
||||
'RGB',
|
||||
],
|
||||
)
|
||||
def test_devicen_warning_persists_despite_strategy_still_raises(outdir, strategy):
|
||||
"""If the warning survives the requested conversion, the output is broken.
|
||||
|
||||
We must still raise rather than silently emit a PDF/A that may render blank.
|
||||
The guidance should acknowledge that the chosen strategy did not work and
|
||||
point at strategies that do (or --output-type pdf).
|
||||
"""
|
||||
with pytest.raises(ColorConversionNeededError) as exc_info:
|
||||
_run_generate_pdfa_with_devicen_warning(outdir, strategy)
|
||||
message = str(exc_info.value)
|
||||
assert strategy in message
|
||||
assert '--output-type pdf' in message
|
||||
|
||||
|
||||
def test_no_devicen_warning_does_not_raise(outdir):
|
||||
"""When Ghostscript does not warn, conversion succeeded; never raise."""
|
||||
(outdir / 'input.pdf').write_bytes(b'%PDF-1.5\n%fake\n')
|
||||
with (
|
||||
patch('ocrmypdf._exec.ghostscript.version', return_value=Version('10.05.1')),
|
||||
patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as run_mock,
|
||||
):
|
||||
run_mock.return_value = subprocess.CompletedProcess(
|
||||
['gs'], returncode=0, stdout='', stderr=''
|
||||
)
|
||||
# Must not raise for any strategy when there is no DeviceN warning.
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[outdir / 'input.pdf'],
|
||||
output_file=outdir / 'out.pdf',
|
||||
compression='auto',
|
||||
color_conversion_strategy='RGB',
|
||||
)
|
||||
|
||||
|
||||
def test_rasterize_pdf_errors(resources, no_outpdf, caplog):
|
||||
with patch('ocrmypdf._exec.ghostscript.run') as mock:
|
||||
# ghostscript can produce empty files with return code 0
|
||||
@@ -439,7 +704,9 @@ class TestGs106JpegCorruptionRepair:
|
||||
repaired_bytes_list.append(obj.read_raw_bytes())
|
||||
|
||||
assert len(repaired_bytes_list) == len(original_bytes_list)
|
||||
for orig, repaired_bytes in zip(original_bytes_list, repaired_bytes_list, strict=False):
|
||||
for orig, repaired_bytes in zip(
|
||||
original_bytes_list, repaired_bytes_list, strict=False
|
||||
):
|
||||
assert orig == repaired_bytes, "Repaired bytes should match original"
|
||||
|
||||
# Check that error/warning was logged
|
||||
@@ -468,3 +735,58 @@ class TestGs106JpegCorruptionRepair:
|
||||
repaired = _repair_gs106_jpeg_corruption(source_path, damaged_path)
|
||||
assert repaired is False, "Should not repair truncation > 15 bytes"
|
||||
assert "JPEG corruption detected" not in caplog.text
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
('compression', 'optimize', 'expected'),
|
||||
[
|
||||
# auto coerces to lossless only at -O0; -O1 is a historical exception
|
||||
# that keeps Ghostscript's (possibly lossy) heuristic, as do -O2/-O3
|
||||
(PdfaImageCompression.AUTO, 0, PdfaImageCompression.LOSSLESS),
|
||||
(PdfaImageCompression.AUTO, 1, PdfaImageCompression.AUTO),
|
||||
(PdfaImageCompression.AUTO, 2, PdfaImageCompression.AUTO),
|
||||
(PdfaImageCompression.AUTO, 3, PdfaImageCompression.AUTO),
|
||||
# explicit choices are always respected, regardless of optimize level
|
||||
(PdfaImageCompression.JPEG, 0, PdfaImageCompression.JPEG),
|
||||
(PdfaImageCompression.JPEG, 1, PdfaImageCompression.JPEG),
|
||||
(PdfaImageCompression.LOSSLESS, 1, PdfaImageCompression.LOSSLESS),
|
||||
(PdfaImageCompression.LOSSLESS, 3, PdfaImageCompression.LOSSLESS),
|
||||
],
|
||||
)
|
||||
def test_resolve_auto_compression(compression, optimize, expected):
|
||||
assert _resolve_auto_compression(compression, optimize) == expected
|
||||
|
||||
|
||||
def _capture_generate_pdfa_args(tmp_path, compression):
|
||||
"""Run generate_pdfa with a mocked Ghostscript and return the argv it built."""
|
||||
from subprocess import CompletedProcess
|
||||
|
||||
captured = {}
|
||||
|
||||
def fake_run(args, **kwargs):
|
||||
captured['args'] = list(args)
|
||||
return CompletedProcess(args, 0, None, stderr='')
|
||||
|
||||
out = tmp_path / 'out.pdf'
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr', side_effect=fake_run):
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=['dummy.pdf'],
|
||||
output_file=out,
|
||||
compression=compression,
|
||||
color_conversion_strategy='RGB',
|
||||
)
|
||||
return captured['args']
|
||||
|
||||
|
||||
def test_lossless_compression_passes_through_jpegs(tmp_path):
|
||||
# Re-encoding an existing JPEG losslessly only bloats it (the lossy data is
|
||||
# already baked in), so lossless mode must let Ghostscript pass JPEGs through
|
||||
# untouched while still keeping lossless images lossless.
|
||||
args = _capture_generate_pdfa_args(tmp_path, 'lossless')
|
||||
assert '-dPassThroughJPEGImages=true' in args
|
||||
assert '-dColorImageFilter=/FlateEncode' in args
|
||||
|
||||
|
||||
def test_jpeg_compression_does_not_force_passthrough(tmp_path):
|
||||
args = _capture_generate_pdfa_args(tmp_path, 'jpeg')
|
||||
assert '-dPassThroughJPEGImages=true' not in args
|
||||
|
||||
+27
-6
@@ -100,9 +100,9 @@ def test_redo_ocr(resources, outpdf):
|
||||
out = check_ocrmypdf(in_, out, '--redo-ocr')
|
||||
after = PdfInfo(out, detailed_analysis=True)
|
||||
assert before[0].has_text and after[0].has_text
|
||||
assert (
|
||||
before[0].get_textareas() != after[0].get_textareas()
|
||||
), "Expected text to be different after re-OCR"
|
||||
assert before[0].get_textareas() != after[0].get_textareas(), (
|
||||
"Expected text to be different after re-OCR"
|
||||
)
|
||||
|
||||
|
||||
def test_argsfile(resources, outdir):
|
||||
@@ -768,9 +768,9 @@ def test_sidecar_pagecount(resources, outpdf):
|
||||
|
||||
# There should a formfeed between each pair of pages, so the count of
|
||||
# formfeeds is the page count less one
|
||||
assert (
|
||||
ocr_text.count('\f') == num_pages - 1
|
||||
), "Sidecar page count does not match PDF page count"
|
||||
assert ocr_text.count('\f') == num_pages - 1, (
|
||||
"Sidecar page count does not match PDF page count"
|
||||
)
|
||||
|
||||
|
||||
def test_sidecar_nonempty(resources, outpdf):
|
||||
@@ -889,6 +889,27 @@ def test_version_check():
|
||||
get_version('echo')
|
||||
|
||||
|
||||
def test_get_version_skips_leading_warning_lines(monkeypatch):
|
||||
"""VeraPDF 1.30.0 prints JVM warnings before its version line."""
|
||||
from subprocess import CompletedProcess
|
||||
|
||||
import ocrmypdf.subprocess as sp
|
||||
|
||||
output = (
|
||||
"WARNING: Final field flavour has been mutated reflectively\n"
|
||||
"WARNING: Use --enable-final-field-mutation=ALL-UNNAMED to avoid this\n"
|
||||
"veraPDF 1.30.0\n"
|
||||
"Built: Wed Jun 03 13:29:00 PDT 2026\n"
|
||||
)
|
||||
|
||||
def fake_run(args, **kwargs):
|
||||
return CompletedProcess(args, 0, stdout=output, stderr="")
|
||||
|
||||
monkeypatch.setattr(sp, 'run', fake_run)
|
||||
version = get_version('verapdf', regex=r'veraPDF (\d+(\.\d+)*)')
|
||||
assert version == '1.30.0'
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'threshold, optimize, output_type, expected',
|
||||
[
|
||||
|
||||
+28
-1
@@ -6,13 +6,14 @@ from __future__ import annotations
|
||||
import datetime as dt
|
||||
import warnings
|
||||
from shutil import copyfile
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from pikepdf.models.metadata import decode_pdf_date
|
||||
|
||||
from ocrmypdf._jobcontext import PdfContext
|
||||
from ocrmypdf._metadata import metadata_fixup
|
||||
from ocrmypdf._metadata import metadata_fixup, repair_docinfo_nuls
|
||||
from ocrmypdf._pipeline import convert_to_pdfa
|
||||
from ocrmypdf.api import setup_plugin_infrastructure
|
||||
from ocrmypdf.cli import get_options_and_plugins
|
||||
@@ -43,6 +44,32 @@ def test_preserve_docinfo(output_type, resources, outpdf):
|
||||
assert pdfa_info['output'] == output_type
|
||||
|
||||
|
||||
def test_repair_docinfo_nuls_undecodable_key(caplog):
|
||||
"""A DocumentInfo key with bytes that don't decode must not crash.
|
||||
|
||||
Some PDFs use a /Name dictionary key in DocumentInfo whose bytes are not
|
||||
valid PDFDocEncoding/UTF-8 (e.g. Latin-1 ``/Saks#e5r``). Older pikepdf
|
||||
raised UnicodeDecodeError while iterating such a dictionary. The repair
|
||||
must log and continue rather than propagate the exception. See #1540.
|
||||
"""
|
||||
pdf = MagicMock()
|
||||
pdf.docinfo.items.side_effect = UnicodeDecodeError(
|
||||
'utf-8', b'Saks\xe5r', 4, 5, 'invalid continuation byte'
|
||||
)
|
||||
# Make isinstance(pdf.docinfo, Dictionary) succeed so we reach the loop.
|
||||
with patch('ocrmypdf._metadata.Dictionary', MagicMock):
|
||||
result = repair_docinfo_nuls(pdf)
|
||||
assert result is False
|
||||
assert 'malformed DocumentInfo' in caplog.text
|
||||
|
||||
|
||||
def test_repair_docinfo_nuls_undecodable_key_real_file(resources):
|
||||
"""Opening a real file with a Latin-1 DocumentInfo key must not crash."""
|
||||
with pikepdf.open(resources / 'docinfo_latin1_key.pdf') as pdf:
|
||||
# Should return without raising regardless of pikepdf's decode behavior.
|
||||
repair_docinfo_nuls(pdf)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("output_type", ['pdfa', 'pdf'])
|
||||
def test_override_metadata(output_type, resources, outpdf, caplog):
|
||||
input_file = resources / 'c02-22.pdf'
|
||||
|
||||
@@ -149,7 +149,9 @@ def test_select_font_for_chinese_language(multi_font_manager):
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("你好", "zho")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
# A real, glyph-covering CJK font is selected (which specific family
|
||||
# depends on what is installed: pan-CJK super font or a per-language subset).
|
||||
assert font_manager.font_path.name != 'Occulta.ttf'
|
||||
|
||||
|
||||
def test_select_font_for_chinese_generic(multi_font_manager):
|
||||
@@ -157,7 +159,9 @@ def test_select_font_for_chinese_generic(multi_font_manager):
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("中文", "chi")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
# A real, glyph-covering CJK font is selected (which specific family
|
||||
# depends on what is installed: pan-CJK super font or a per-language subset).
|
||||
assert font_manager.font_path.name != 'Occulta.ttf'
|
||||
|
||||
|
||||
def test_select_font_for_chinese_simplified(multi_font_manager):
|
||||
@@ -165,7 +169,9 @@ def test_select_font_for_chinese_simplified(multi_font_manager):
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("简体字", "chi_sim")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
# A real, glyph-covering CJK font is selected (which specific family
|
||||
# depends on what is installed: pan-CJK super font or a per-language subset).
|
||||
assert font_manager.font_path.name != 'Occulta.ttf'
|
||||
|
||||
|
||||
def test_select_font_for_chinese_traditional(multi_font_manager):
|
||||
@@ -173,7 +179,9 @@ def test_select_font_for_chinese_traditional(multi_font_manager):
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("漢字", "chi_tra")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
# A real, glyph-covering CJK font is selected (which specific family
|
||||
# depends on what is installed: pan-CJK super font or a per-language subset).
|
||||
assert font_manager.font_path.name != 'Occulta.ttf'
|
||||
|
||||
|
||||
def test_select_font_for_japanese_language(multi_font_manager):
|
||||
@@ -181,7 +189,9 @@ def test_select_font_for_japanese_language(multi_font_manager):
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("こんにちは", "jpn")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
# A real, glyph-covering CJK font is selected (which specific family
|
||||
# depends on what is installed: pan-CJK super font or a per-language subset).
|
||||
assert font_manager.font_path.name != 'Occulta.ttf'
|
||||
|
||||
|
||||
def test_select_font_for_korean_language(multi_font_manager):
|
||||
@@ -189,7 +199,9 @@ def test_select_font_for_korean_language(multi_font_manager):
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("안녕하세요", "kor")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
# A real, glyph-covering CJK font is selected (which specific family
|
||||
# depends on what is installed: pan-CJK super font or a per-language subset).
|
||||
assert font_manager.font_path.name != 'Occulta.ttf'
|
||||
|
||||
|
||||
# --- Latin/English Tests ---
|
||||
@@ -232,7 +244,9 @@ def test_cjk_text_without_language_hint(multi_font_manager):
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("你好", None)
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
# A real, glyph-covering CJK font is selected (which specific family
|
||||
# depends on what is installed: pan-CJK super font or a per-language subset).
|
||||
assert font_manager.font_path.name != 'Occulta.ttf'
|
||||
|
||||
|
||||
def test_fallback_to_occulta_font(multi_font_manager):
|
||||
@@ -444,3 +458,116 @@ def test_builtin_font_provider_missing_occulta_raises(tmp_path):
|
||||
"""Test that missing Occulta.ttf raises FileNotFoundError."""
|
||||
with pytest.raises(FileNotFoundError, match="Required fallback font"):
|
||||
BuiltinFontProvider(tmp_path)
|
||||
|
||||
|
||||
class _StubHbFont:
|
||||
"""Minimal uharfbuzz Font stand-in with controllable glyph coverage."""
|
||||
|
||||
def __init__(self, covered_codepoints: set[int]):
|
||||
self._covered = covered_codepoints
|
||||
|
||||
def get_nominal_glyph(self, codepoint: int) -> int:
|
||||
return 1 if codepoint in self._covered else 0
|
||||
|
||||
|
||||
class _FakeFontManager:
|
||||
"""FontManager stand-in whose glyph coverage is fixed per test."""
|
||||
|
||||
def __init__(self, name: str, covered_chars: str):
|
||||
self.font_path = Path(name)
|
||||
self._hb = _StubHbFont({ord(c) for c in covered_chars})
|
||||
|
||||
def get_hb_font(self) -> _StubHbFont:
|
||||
return self._hb
|
||||
|
||||
|
||||
class _FakeFontProvider:
|
||||
"""FontProvider returning controlled fonts by logical name."""
|
||||
|
||||
def __init__(self, fonts: dict[str, _FakeFontManager]):
|
||||
self._fonts = fonts
|
||||
self._fallback = _FakeFontManager('Occulta.ttf', '')
|
||||
|
||||
def get_font(self, name: str) -> _FakeFontManager | None:
|
||||
return self._fonts.get(name)
|
||||
|
||||
def get_available_fonts(self) -> list[str]:
|
||||
return list(self._fonts)
|
||||
|
||||
def get_fallback_font(self) -> _FakeFontManager:
|
||||
return self._fallback
|
||||
|
||||
|
||||
def test_japanese_prefers_jp_family_over_other_cjk():
|
||||
"""A Japanese language hint selects NotoSansJP, not another CJK family."""
|
||||
fonts = {
|
||||
'NotoSansSC-Regular': _FakeFontManager('NotoSansSC.ttf', '中'),
|
||||
'NotoSansJP-Regular': _FakeFontManager('NotoSansJP.ttf', '中こ'),
|
||||
}
|
||||
manager = MultiFontManager(font_provider=_FakeFontProvider(fonts))
|
||||
# 'こ' (kana) is only covered by JP; both cover the kanji '中'.
|
||||
font = manager.select_font_for_word('中こ', 'jpn')
|
||||
assert font.font_path.name == 'NotoSansJP.ttf'
|
||||
|
||||
|
||||
def test_chinese_simplified_prefers_sc_family():
|
||||
"""A Simplified Chinese hint selects NotoSansSC over the pan-CJK font."""
|
||||
fonts = {
|
||||
'NotoSansSC-Regular': _FakeFontManager('NotoSansSC.ttf', '简'),
|
||||
'NotoSansCJK-Regular': _FakeFontManager('NotoSansCJK.ttc', '简'),
|
||||
}
|
||||
manager = MultiFontManager(font_provider=_FakeFontProvider(fonts))
|
||||
font = manager.select_font_for_word('简', 'chi_sim')
|
||||
assert font.font_path.name == 'NotoSansSC.ttf'
|
||||
|
||||
|
||||
def test_cjk_fails_over_when_preferred_subset_lacks_glyph():
|
||||
"""If the language's subset font lacks a glyph, another CJK family is used."""
|
||||
fonts = {
|
||||
# Simplified Chinese subset cannot render Japanese kana.
|
||||
'NotoSansSC-Regular': _FakeFontManager('NotoSansSC.ttf', '中'),
|
||||
'NotoSansJP-Regular': _FakeFontManager('NotoSansJP.ttf', '中こ'),
|
||||
}
|
||||
manager = MultiFontManager(font_provider=_FakeFontProvider(fonts))
|
||||
# Tagged Simplified Chinese, but the text needs kana only JP covers.
|
||||
font = manager.select_font_for_word('こ', 'chi_sim')
|
||||
assert font.font_path.name == 'NotoSansJP.ttf'
|
||||
|
||||
|
||||
def test_cjk_falls_back_to_pan_cjk_super_font():
|
||||
"""When only the full-coverage pan-CJK font exists, it serves any CJK lang."""
|
||||
fonts = {
|
||||
'NotoSansCJK-Regular': _FakeFontManager('NotoSansCJK.ttc', '中こ안'),
|
||||
}
|
||||
manager = MultiFontManager(font_provider=_FakeFontProvider(fonts))
|
||||
assert manager.select_font_for_word('こ', 'jpn').font_path.name == 'NotoSansCJK.ttc'
|
||||
|
||||
|
||||
def test_missing_cjk_font_warning_names_language_font(font_dir, caplog):
|
||||
"""The missing-font warning names the language-specific CJK family (#1652)."""
|
||||
manager = MultiFontManager(font_provider=BuiltinFontProvider(font_dir))
|
||||
with caplog.at_level(logging.WARNING):
|
||||
manager.select_font_for_word("こんにちは", "jpn")
|
||||
assert 'NotoSansJP' in caplog.text
|
||||
|
||||
|
||||
def test_missing_font_warning_explains_consequences(font_dir, caplog):
|
||||
"""The glyphless-fallback warning should be actionable, not cryptic (#1652).
|
||||
|
||||
With only builtin fonts available, Arabic text cannot be covered, so the
|
||||
manager falls back to glyphless Occulta and must warn helpfully.
|
||||
"""
|
||||
# Builtin-only provider: NotoSansArabic is never available, forcing fallback.
|
||||
manager = MultiFontManager(font_provider=BuiltinFontProvider(font_dir))
|
||||
|
||||
with caplog.at_level(logging.WARNING):
|
||||
manager.select_font_for_word("سلام", "fas")
|
||||
|
||||
msg = caplog.text
|
||||
# Identifies the affected language and the font family to install.
|
||||
assert 'fas' in msg
|
||||
assert 'NotoSansArabic' in msg
|
||||
# Explains the user-visible consequence so the message is not cryptic:
|
||||
# the text stays searchable but renders blank when highlighted.
|
||||
assert 'searchable' in msg.lower()
|
||||
assert 'highlight' in msg.lower() or 'select' in msg.lower()
|
||||
|
||||
+72
-6
@@ -197,14 +197,14 @@ def test_optimize_off(resources, outpdf):
|
||||
def test_group3(resources):
|
||||
with pikepdf.open(resources / 'ccitt.pdf') as pdf:
|
||||
im = pdf.pages[0].Resources.XObject['/Im1']
|
||||
assert (
|
||||
opt.extract_image_filter(im, im.objgen[0]) is not None
|
||||
), "Group 4 should be allowed"
|
||||
assert opt.extract_image_filter(im, im.objgen[0]) is not None, (
|
||||
"Group 4 should be allowed"
|
||||
)
|
||||
|
||||
im.DecodeParms['/K'] = 0
|
||||
assert (
|
||||
opt.extract_image_filter(im, im.objgen[0]) is None
|
||||
), "Group 3 should be disallowed"
|
||||
assert opt.extract_image_filter(im, im.objgen[0]) is None, (
|
||||
"Group 3 should be disallowed"
|
||||
)
|
||||
|
||||
|
||||
def test_find_formx(resources):
|
||||
@@ -215,6 +215,72 @@ def test_find_formx(resources):
|
||||
assert pagenos[xref] == 0
|
||||
|
||||
|
||||
def test_find_formx_circular_reference(resources, tmp_path, caplog):
|
||||
"""Regression for issue #1321.
|
||||
|
||||
Some PDFs (notably PowerPoint exports) contain Form XObjects that
|
||||
reference themselves or each other in a cycle. The recursion guard in
|
||||
_find_image_xrefs_container only deduplicates *image* xrefs, so a Form
|
||||
XObject cycle would re-enter every branch until the depth limit fired,
|
||||
producing thousands of "Recursion depth exceeded" warnings (and minutes
|
||||
of wall-clock time on real-world inputs).
|
||||
"""
|
||||
import logging
|
||||
|
||||
src = resources / 'formxobject.pdf'
|
||||
out = tmp_path / 'circular_form.pdf'
|
||||
with pikepdf.open(src) as pdf:
|
||||
# /Form1 lives at xref 10. Replace its Resources.XObject with three
|
||||
# entries that all point back to /Form1 itself, creating a fan-out
|
||||
# cycle of branching factor 3.
|
||||
form = pdf.pages[0].obj.Resources.XObject.Form1
|
||||
form.Resources.XObject = Dictionary({'/Fm0': form, '/Fm1': form, '/Fm2': form})
|
||||
pdf.save(out)
|
||||
|
||||
caplog.set_level(logging.WARNING, logger='ocrmypdf.optimize')
|
||||
with pikepdf.open(out) as pdf:
|
||||
opt._find_image_xrefs(pdf)
|
||||
|
||||
n_warnings = sum(
|
||||
1 for r in caplog.records if 'Recursion depth exceeded' in r.getMessage()
|
||||
)
|
||||
# Without the fix this is in the tens of thousands.
|
||||
assert n_warnings == 0, (
|
||||
f"Form XObject cycle should be detected without depth-limit warnings; "
|
||||
f"got {n_warnings}"
|
||||
)
|
||||
|
||||
|
||||
def test_extract_images_traps_errors_as_warning(resources, tmp_path, caplog):
|
||||
"""Regression for issue #846.
|
||||
|
||||
The optimizer is best-effort: any image it cannot process can simply be
|
||||
passed through unchanged. When extraction of an image raises (e.g. an
|
||||
exotic colorspace pikepdf cannot transcode), the user should see a concise
|
||||
warning that the image was left unchanged, not an alarming traceback
|
||||
logged at ERROR level.
|
||||
"""
|
||||
import logging
|
||||
from unittest.mock import Mock
|
||||
|
||||
def boom(*, pdf, root, image, xref, options):
|
||||
raise NotImplementedError("synthetic extraction failure")
|
||||
|
||||
caplog.set_level(logging.DEBUG, logger='ocrmypdf.optimize')
|
||||
with pikepdf.open(resources / 'francais.pdf') as pdf:
|
||||
results = list(opt.extract_images(pdf, tmp_path, Mock(), boom))
|
||||
|
||||
# The error is trapped, not propagated, and nothing is extracted.
|
||||
assert results == []
|
||||
# A friendly warning is emitted...
|
||||
assert any(
|
||||
r.levelno == logging.WARNING and 'left unchanged' in r.getMessage()
|
||||
for r in caplog.records
|
||||
)
|
||||
# ...and no traceback is logged at ERROR level or above.
|
||||
assert not any(r.levelno >= logging.ERROR for r in caplog.records)
|
||||
|
||||
|
||||
def test_extract_image_filter_with_pdf_image():
|
||||
image = Dictionary()
|
||||
image.Subtype = Name.Image
|
||||
|
||||
@@ -7,6 +7,7 @@ import pikepdf
|
||||
import pytest
|
||||
|
||||
from ocrmypdf._exec import verapdf
|
||||
from ocrmypdf._pageboxes import repair_page_boxes
|
||||
|
||||
from .conftest import check_ocrmypdf
|
||||
|
||||
@@ -127,3 +128,147 @@ def test_crop_box(
|
||||
with pikepdf.open(outdir / 'processed.pdf') as pdf:
|
||||
page = pdf.pages[0]
|
||||
assert [float(x) for x in page.cropbox] == crop_expected
|
||||
|
||||
|
||||
# --- Unit tests for repair_page_boxes (issues #1398, #1526, #1400) ---
|
||||
|
||||
|
||||
def _is_numeric(obj) -> bool:
|
||||
try:
|
||||
float(obj)
|
||||
return True
|
||||
except (TypeError, ValueError):
|
||||
return False
|
||||
|
||||
|
||||
def _one_page_pdf(**boxes):
|
||||
"""Build a one-page PDF, setting the named boxes to the given arrays."""
|
||||
pdf = pikepdf.new()
|
||||
page = pdf.add_blank_page(page_size=(612, 792))
|
||||
for name, rect in boxes.items():
|
||||
setattr(page.obj, name, pikepdf.Array(rect))
|
||||
return pdf, page
|
||||
|
||||
|
||||
def test_repair_reversed_mediabox_is_normalized():
|
||||
# #1526: diagonally-opposite corners given in reversed order
|
||||
_pdf, page = _one_page_pdf(MediaBox=[0, 792, 612, 0])
|
||||
repairs = repair_page_boxes(page)
|
||||
assert [float(x) for x in page.obj.MediaBox] == [0, 0, 612, 792]
|
||||
assert any(r.box == 'MediaBox' and r.kind == 'reordered' for r in repairs)
|
||||
|
||||
|
||||
def test_repair_cropbox_entirely_outside_mediabox_is_discarded():
|
||||
# #1400: CropBox lies entirely outside the MediaBox -> empty intersection
|
||||
_pdf, page = _one_page_pdf(
|
||||
MediaBox=[0, 0, 612, 792], CropBox=[1000, 1000, 1500, 1500]
|
||||
)
|
||||
repairs = repair_page_boxes(page)
|
||||
assert '/CropBox' not in page.obj
|
||||
assert any(r.box == 'CropBox' and r.kind == 'discarded' for r in repairs)
|
||||
|
||||
|
||||
def test_repair_cropbox_partially_outside_mediabox_is_clamped():
|
||||
_pdf, page = _one_page_pdf(MediaBox=[0, 0, 612, 792], CropBox=[200, 200, 800, 900])
|
||||
repairs = repair_page_boxes(page)
|
||||
assert [float(x) for x in page.obj.CropBox] == [200, 200, 612, 792]
|
||||
assert any(r.box == 'CropBox' and r.kind == 'clamped' for r in repairs)
|
||||
|
||||
|
||||
def test_repair_exponential_coordinate_is_coerced():
|
||||
# #1398: a coordinate stored as a string in exponential notation
|
||||
_pdf, page = _one_page_pdf(
|
||||
MediaBox=[0, 0, 612, 792],
|
||||
TrimBox=[pikepdf.String('3.05175781e-005'), 0, 612, 792],
|
||||
)
|
||||
repairs = repair_page_boxes(page)
|
||||
trim = page.obj.TrimBox
|
||||
assert all(_is_numeric(x) for x in trim)
|
||||
assert float(trim[0]) == pytest.approx(3.05175781e-5, abs=1e-4)
|
||||
assert any(r.box == 'TrimBox' and r.kind == 'recoded' for r in repairs)
|
||||
|
||||
|
||||
def test_repair_degenerate_mediabox_is_reported():
|
||||
_pdf, page = _one_page_pdf(MediaBox=[0, 0, 0, 792]) # zero width
|
||||
repairs = repair_page_boxes(page)
|
||||
assert any(r.box == 'MediaBox' and r.kind == 'degenerate_mediabox' for r in repairs)
|
||||
|
||||
|
||||
def test_repair_valid_page_makes_no_changes():
|
||||
_pdf, page = _one_page_pdf(MediaBox=[0, 0, 612, 792], CropBox=[10, 10, 600, 780])
|
||||
repairs = repair_page_boxes(page)
|
||||
assert repairs == []
|
||||
assert [float(x) for x in page.obj.MediaBox] == [0, 0, 612, 792]
|
||||
assert [float(x) for x in page.obj.CropBox] == [10, 10, 600, 780]
|
||||
|
||||
|
||||
def test_summarize_box_repairs_aggregates_and_sets_severity():
|
||||
import logging
|
||||
|
||||
from ocrmypdf._pageboxes import BoxRepair, summarize_box_repairs
|
||||
|
||||
repairs_by_page = {
|
||||
0: [BoxRepair('CropBox', 'discarded')],
|
||||
2: [BoxRepair('CropBox', 'discarded')],
|
||||
3: [BoxRepair('CropBox', 'discarded')],
|
||||
1: [BoxRepair('MediaBox', 'reordered')],
|
||||
}
|
||||
messages = summarize_box_repairs(repairs_by_page)
|
||||
|
||||
discard = [(lvl, m) for lvl, m in messages if 'discarded' in m]
|
||||
assert len(discard) == 1
|
||||
level, text = discard[0]
|
||||
assert level == logging.WARNING
|
||||
assert 'Page(s) 1, 3-4' in text # 0-based keys shown 1-based, ranges compacted
|
||||
assert 'visually inspect' in text
|
||||
|
||||
reordered = [(lvl, m) for lvl, m in messages if 'reversed' in m]
|
||||
assert len(reordered) == 1
|
||||
assert reordered[0][0] == logging.DEBUG
|
||||
assert 'visually inspect' not in reordered[0][1]
|
||||
|
||||
|
||||
def test_cropbox_outside_mediabox_yields_valid_output(resources, outdir):
|
||||
# #1400: a CropBox entirely outside the MediaBox produces an effective
|
||||
# page of N x 0 pt; the pipeline must repair it to valid output.
|
||||
with pikepdf.open(resources / 'ccitt.pdf') as pdf:
|
||||
page = pdf.pages[0]
|
||||
mb = [float(x) for x in page.mediabox]
|
||||
page.CropBox = [mb[2] + 100, mb[3] + 100, mb[2] + 200, mb[3] + 200]
|
||||
pdf.save(outdir / 'badcrop.pdf')
|
||||
|
||||
check_ocrmypdf(
|
||||
outdir / 'badcrop.pdf',
|
||||
outdir / 'out.pdf',
|
||||
'--output-type',
|
||||
'pdf',
|
||||
'--optimize',
|
||||
'0',
|
||||
)
|
||||
|
||||
with pikepdf.open(outdir / 'out.pdf') as pdf:
|
||||
cb = [float(x) for x in pdf.pages[0].cropbox] # resolves to MediaBox
|
||||
assert (cb[2] - cb[0]) > 0 and (cb[3] - cb[1]) > 0
|
||||
|
||||
|
||||
def test_reversed_mediabox_does_not_crash(resources, outdir):
|
||||
# #1526: reversed MediaBox corners previously raised NegativeDimensionError.
|
||||
with pikepdf.open(resources / 'ccitt.pdf') as pdf:
|
||||
page = pdf.pages[0]
|
||||
mb = [float(x) for x in page.mediabox]
|
||||
page.MediaBox = [mb[0], mb[3], mb[2], mb[1]] # swap y corners
|
||||
pdf.save(outdir / 'reversed.pdf')
|
||||
|
||||
check_ocrmypdf(
|
||||
outdir / 'reversed.pdf',
|
||||
outdir / 'out.pdf',
|
||||
'--force-ocr',
|
||||
'--output-type',
|
||||
'pdf',
|
||||
'--optimize',
|
||||
'0',
|
||||
)
|
||||
|
||||
with pikepdf.open(outdir / 'out.pdf') as pdf:
|
||||
mb = [float(x) for x in pdf.pages[0].mediabox]
|
||||
assert (mb[2] - mb[0]) > 0 and (mb[3] - mb[1]) > 0
|
||||
|
||||
@@ -42,6 +42,35 @@ def test_pages(pages, result):
|
||||
assert _pages_from_ranges(pages) == result
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'pages, total_pages, result',
|
||||
[
|
||||
['end', 10, {9}],
|
||||
['END', 10, {9}],
|
||||
['1-end', 3, {0, 1, 2}],
|
||||
['3-end', 5, {2, 3, 4}],
|
||||
['end-end', 7, {6}],
|
||||
['1,end', 4, {0, 3}],
|
||||
['2-4,end', 10, {1, 2, 3, 9}],
|
||||
['end,end,end', 5, {4}],
|
||||
['end-1', 5, BadArgsError], # empty range when end > 1
|
||||
],
|
||||
)
|
||||
def test_pages_end_alias(pages, total_pages, result):
|
||||
if isinstance(result, type):
|
||||
with pytest.raises(result):
|
||||
_pages_from_ranges(pages, total_pages=total_pages)
|
||||
else:
|
||||
assert _pages_from_ranges(pages, total_pages=total_pages) == result
|
||||
|
||||
|
||||
def test_end_alias_requires_total_pages():
|
||||
with pytest.raises(BadArgsError, match="total page count"):
|
||||
_pages_from_ranges('1-end')
|
||||
with pytest.raises(BadArgsError, match="total page count"):
|
||||
_pages_from_ranges('end')
|
||||
|
||||
|
||||
def test_nonmonotonic_warning(caplog):
|
||||
pages = _pages_from_ranges('1, 3, 2')
|
||||
assert pages == {0, 1, 2}
|
||||
@@ -61,3 +90,33 @@ def test_limited_pages(multipage, outpdf):
|
||||
assert not pi.pages[0].has_text
|
||||
assert pi.pages[4].has_text
|
||||
assert pi.pages[5].has_text
|
||||
|
||||
|
||||
def test_limited_pages_end_alias(multipage, outpdf):
|
||||
# multipage has 6 pages; 5-end == pages 5..6
|
||||
ocrmypdf.ocr(
|
||||
multipage,
|
||||
outpdf,
|
||||
pages='5-end',
|
||||
optimize=0,
|
||||
output_type='pdf',
|
||||
plugins=['tests/plugins/tesseract_cache.py'],
|
||||
)
|
||||
pi = PdfInfo(outpdf)
|
||||
assert not pi.pages[0].has_text
|
||||
assert pi.pages[4].has_text
|
||||
assert pi.pages[5].has_text
|
||||
|
||||
|
||||
def test_pages_end_alone(multipage, outpdf):
|
||||
ocrmypdf.ocr(
|
||||
multipage,
|
||||
outpdf,
|
||||
pages='end',
|
||||
optimize=0,
|
||||
output_type='pdf',
|
||||
plugins=['tests/plugins/tesseract_cache.py'],
|
||||
)
|
||||
pi = PdfInfo(outpdf)
|
||||
assert not pi.pages[0].has_text
|
||||
assert pi.pages[5].has_text
|
||||
|
||||
+213
-2
@@ -7,10 +7,221 @@ import os
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from pikepdf import Name
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.exceptions import ExitCode, MissingDependencyError
|
||||
from ocrmypdf.pdfa import file_claims_pdfa, find_nonembedded_cid_fonts
|
||||
|
||||
from .conftest import check_ocrmypdf
|
||||
from .conftest import check_ocrmypdf, run_ocrmypdf_api
|
||||
|
||||
|
||||
def _make_cid_font(
|
||||
pdf: pikepdf.Pdf, *, embedded: bool, basefont: str
|
||||
) -> pikepdf.Object:
|
||||
"""Build a Type0/CID font object, optionally embedding glyph data."""
|
||||
descriptor = pikepdf.Dictionary(
|
||||
Type=Name.FontDescriptor, FontName=Name(basefont), Flags=4
|
||||
)
|
||||
if embedded:
|
||||
# The actual bytes do not matter; only the presence of FontFile2 marks
|
||||
# the CID font as embedded.
|
||||
descriptor.FontFile2 = pdf.make_stream(b'\x00\x01\x00\x00 fake font program')
|
||||
cidfont = pdf.make_indirect(
|
||||
pikepdf.Dictionary(
|
||||
Type=Name.Font,
|
||||
Subtype=Name.CIDFontType2,
|
||||
BaseFont=Name(basefont),
|
||||
FontDescriptor=descriptor,
|
||||
CIDSystemInfo=pikepdf.Dictionary(
|
||||
Registry='Adobe', Ordering='Identity', Supplement=0
|
||||
),
|
||||
)
|
||||
)
|
||||
return pdf.make_indirect(
|
||||
pikepdf.Dictionary(
|
||||
Type=Name.Font,
|
||||
Subtype=Name.Type0,
|
||||
BaseFont=Name(basefont),
|
||||
Encoding=Name.Identity_H,
|
||||
DescendantFonts=pikepdf.Array([cidfont]),
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def _write_cid_font_pdf(path, *, embedded: bool, basefont='/ABCDEF+TestCID'):
|
||||
with pikepdf.new() as pdf:
|
||||
page = pdf.add_blank_page()
|
||||
font = _make_cid_font(pdf, embedded=embedded, basefont=basefont)
|
||||
page.Resources = pikepdf.Dictionary(Font=pikepdf.Dictionary(F0=font))
|
||||
pdf.save(path)
|
||||
|
||||
|
||||
class TestFindNonembeddedCidFonts:
|
||||
def test_blank_page_reports_nothing(self, tmp_path):
|
||||
path = tmp_path / 'blank.pdf'
|
||||
with pikepdf.new() as pdf:
|
||||
pdf.add_blank_page()
|
||||
pdf.save(path)
|
||||
with pikepdf.open(path) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf) == set()
|
||||
|
||||
def test_detects_nonembedded_cid_font(self, tmp_path):
|
||||
path = tmp_path / 'nonembedded.pdf'
|
||||
_write_cid_font_pdf(path, embedded=False)
|
||||
with pikepdf.open(path) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf) == {'ABCDEF+TestCID'}
|
||||
|
||||
def test_ignores_embedded_cid_font(self, tmp_path):
|
||||
path = tmp_path / 'embedded.pdf'
|
||||
_write_cid_font_pdf(path, embedded=True)
|
||||
with pikepdf.open(path) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf) == set()
|
||||
|
||||
def test_detects_nonembedded_cid_font_in_form_xobject(self, tmp_path):
|
||||
path = tmp_path / 'xobject.pdf'
|
||||
with pikepdf.new() as pdf:
|
||||
page = pdf.add_blank_page()
|
||||
font = _make_cid_font(pdf, embedded=False, basefont='/ZZZ+Hidden')
|
||||
form = pdf.make_stream(
|
||||
b'',
|
||||
Type=Name.XObject,
|
||||
Subtype=Name.Form,
|
||||
BBox=pikepdf.Array([0, 0, 1, 1]),
|
||||
Resources=pikepdf.Dictionary(Font=pikepdf.Dictionary(F0=font)),
|
||||
)
|
||||
page.Resources = pikepdf.Dictionary(XObject=pikepdf.Dictionary(Fm0=form))
|
||||
pdf.save(path)
|
||||
with pikepdf.open(path) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf) == {'ZZZ+Hidden'}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def nonembedded_cid_pdf(tmp_path):
|
||||
"""A PDF with a real, non-embedded CID (CJK) text layer, as Acrobat produces."""
|
||||
reportlab = pytest.importorskip('reportlab')
|
||||
del reportlab
|
||||
from reportlab.lib.pagesizes import letter
|
||||
from reportlab.pdfbase import pdfmetrics
|
||||
from reportlab.pdfbase.cidfonts import UnicodeCIDFont
|
||||
from reportlab.pdfgen import canvas
|
||||
|
||||
path = tmp_path / 'cjk_nonembedded.pdf'
|
||||
pdfmetrics.registerFont(UnicodeCIDFont('STSong-Light')) # Adobe-GB1, not embedded
|
||||
c = canvas.Canvas(str(path), pagesize=letter)
|
||||
c.setFont('STSong-Light', 24)
|
||||
c.drawString(60, 650, '你好世界')
|
||||
c.showPage()
|
||||
c.save()
|
||||
# Sanity check that we built the structure under test.
|
||||
with pikepdf.open(path) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf)
|
||||
return path
|
||||
|
||||
|
||||
def test_pdfa_rejects_nonembedded_cid_font(nonembedded_cid_pdf, outpdf):
|
||||
"""Explicit PDF/A on a non-embedded CID layer must error, not corrupt it."""
|
||||
exitcode = run_ocrmypdf_api(
|
||||
nonembedded_cid_pdf,
|
||||
outpdf,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
'--skip-text',
|
||||
'--output-type',
|
||||
'pdfa',
|
||||
)
|
||||
assert exitcode == ExitCode.input_file
|
||||
assert not outpdf.exists() or outpdf.stat().st_size == 0
|
||||
|
||||
|
||||
def test_auto_downgrades_nonembedded_cid_font_to_pdf(nonembedded_cid_pdf, outpdf):
|
||||
"""Auto mode preserves the text layer by outputting a regular PDF."""
|
||||
check_ocrmypdf(
|
||||
nonembedded_cid_pdf,
|
||||
outpdf,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
'--skip-text',
|
||||
'--output-type',
|
||||
'auto',
|
||||
)
|
||||
# Not PDF/A, and the original non-embedded layer survived untouched.
|
||||
assert not file_claims_pdfa(outpdf)['pass']
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf)
|
||||
|
||||
|
||||
def test_auto_falls_back_to_ghostscript_for_pdfa(resources, outpdf, monkeypatch):
|
||||
"""Auto mode produces PDF/A via Ghostscript when the cheap path can't."""
|
||||
# Force the speculative (veraPDF) path off so the fallback is exercised.
|
||||
monkeypatch.setattr('ocrmypdf._exec.verapdf.available', lambda: False)
|
||||
check_ocrmypdf(
|
||||
resources / 'francais.pdf',
|
||||
outpdf,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
'--output-type',
|
||||
'auto',
|
||||
)
|
||||
assert file_claims_pdfa(outpdf)['pass']
|
||||
|
||||
|
||||
def test_auto_outputs_pdf_when_ghostscript_unavailable(resources, outpdf, monkeypatch):
|
||||
"""With neither veraPDF nor Ghostscript, auto outputs a plain PDF."""
|
||||
monkeypatch.setattr('ocrmypdf._exec.verapdf.available', lambda: False)
|
||||
monkeypatch.setattr('ocrmypdf._exec.ghostscript.available', lambda: False)
|
||||
check_ocrmypdf(
|
||||
resources / 'francais.pdf',
|
||||
outpdf,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
'--output-type',
|
||||
'auto',
|
||||
)
|
||||
assert not file_claims_pdfa(outpdf)['pass']
|
||||
|
||||
|
||||
def test_auto_degrades_when_ghostscript_cannot_make_pdfa(
|
||||
resources, outpdf, monkeypatch
|
||||
):
|
||||
"""If Ghostscript produces non-PDF/A output, auto keeps a plain PDF (no error)."""
|
||||
monkeypatch.setattr('ocrmypdf._exec.verapdf.available', lambda: False)
|
||||
exitcode = run_ocrmypdf_api(
|
||||
resources / 'francais.pdf',
|
||||
outpdf,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
'--plugin',
|
||||
'tests/plugins/gs_pdfa_failure.py',
|
||||
'--output-type',
|
||||
'auto',
|
||||
)
|
||||
assert exitcode == ExitCode.ok
|
||||
assert outpdf.exists()
|
||||
assert not file_claims_pdfa(outpdf)['pass']
|
||||
|
||||
|
||||
def test_auto_degrades_when_ghostscript_raises(resources, outpdf, monkeypatch):
|
||||
"""A Ghostscript conversion exception in auto mode degrades to plain PDF."""
|
||||
from ocrmypdf.exceptions import ColorConversionNeededError
|
||||
|
||||
monkeypatch.setattr('ocrmypdf._exec.verapdf.available', lambda: False)
|
||||
monkeypatch.setattr('ocrmypdf._exec.ghostscript.available', lambda: True)
|
||||
|
||||
def boom(*args, **kwargs):
|
||||
raise ColorConversionNeededError()
|
||||
|
||||
monkeypatch.setattr('ocrmypdf._pipeline.convert_to_pdfa', boom)
|
||||
exitcode = run_ocrmypdf_api(
|
||||
resources / 'francais.pdf',
|
||||
outpdf,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
'--output-type',
|
||||
'auto',
|
||||
)
|
||||
assert exitcode == ExitCode.ok
|
||||
assert outpdf.exists()
|
||||
assert not file_claims_pdfa(outpdf)['pass']
|
||||
|
||||
|
||||
@pytest.mark.parametrize('optimize', (0, 3))
|
||||
|
||||
+191
-2
@@ -18,8 +18,8 @@ from reportlab.pdfgen.canvas import Canvas
|
||||
from ocrmypdf import pdfinfo
|
||||
from ocrmypdf.exceptions import InputFileError
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding
|
||||
from ocrmypdf.pdfinfo._contentstream import _interpret_contents
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, Ink
|
||||
from ocrmypdf.pdfinfo._contentstream import _ink_from_components, _interpret_contents
|
||||
from ocrmypdf.pdfinfo.layout import PDFPage
|
||||
|
||||
warnings.filterwarnings(
|
||||
@@ -290,3 +290,192 @@ def test_image_scale0(image_scale0):
|
||||
)
|
||||
assert not pi.pages[0]._images[0].dpi.is_finite
|
||||
assert pi.pages[0].dpi == Resolution(0, 0)
|
||||
|
||||
|
||||
def test_ink_enum_is_picklable():
|
||||
# ImageInfo crosses the worker-process boundary, so Ink must pickle.
|
||||
for member in (Ink.mono, Ink.gray, Ink.color):
|
||||
assert pickle.loads(pickle.dumps(member)) is member
|
||||
|
||||
|
||||
def test_pngmonod_device_exists():
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
|
||||
assert GhostscriptRasterDevice.PNGMONOD == 'pngmonod'
|
||||
# PNGMONO retained for compatibility / explicit use
|
||||
assert GhostscriptRasterDevice.PNGMONO == 'pngmono'
|
||||
|
||||
|
||||
def _ink_of_first_xobject(body: bytes):
|
||||
from ocrmypdf.pdfinfo._contentstream import _interpret_contents
|
||||
|
||||
p = pikepdf.Pdf.new()
|
||||
stream = pikepdf.Stream(p, body)
|
||||
info = _interpret_contents(stream)
|
||||
return info.xobject_settings[0].fill_ink
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"body, expected",
|
||||
[
|
||||
(b"/Im0 Do", 'mono'), # default fill is black
|
||||
(b"0.263 0.263 0.263 rg /Im0 Do", 'gray'),
|
||||
(b"0.5 g /Im0 Do", 'gray'),
|
||||
(b"0 g /Im0 Do", 'mono'),
|
||||
(b"0.8 0.2 0.2 rg /Im0 Do", 'color'),
|
||||
(b"0 0 0 0.5 k /Im0 Do", 'gray'),
|
||||
(b"0.5 0.1 0 0 k /Im0 Do", 'color'),
|
||||
],
|
||||
)
|
||||
def test_fill_ink_tracked_per_draw(body, expected):
|
||||
assert _ink_of_first_xobject(body) is Ink[expected]
|
||||
|
||||
|
||||
def test_fill_ink_non_device_colorspace_is_color():
|
||||
# cs to a non-device colorspace then scn -> conservative color
|
||||
assert _ink_of_first_xobject(b"/CS0 cs 0.4 scn /Im0 Do") is Ink.color
|
||||
|
||||
|
||||
def test_fill_ink_pattern_scn_is_color():
|
||||
assert _ink_of_first_xobject(b"/Pattern cs /P0 scn /Im0 Do") is Ink.color
|
||||
|
||||
|
||||
def test_fill_ink_respects_graphics_stack():
|
||||
# Set red, save, set gray, restore -> red again at the Do
|
||||
assert _ink_of_first_xobject(b"0.8 0.1 0.1 rg q 0.5 g Q /Im0 Do") is Ink.color
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"body",
|
||||
[
|
||||
b"g /Im0 Do", # g with no operand
|
||||
b"/Foo g /Im0 Do", # g with a non-numeric operand
|
||||
b"cs /Im0 Do", # cs with no operand
|
||||
b"0.5 /Foo k /Im0 Do", # k with a non-numeric operand
|
||||
b"/DeviceRGB cs /Foo 0.5 scn /Im0 Do", # scn with mixed bad operands
|
||||
],
|
||||
)
|
||||
def test_fill_ink_tolerates_malformed_color_operands(body):
|
||||
# Malformed color operators must not crash the interpreter; they leave the
|
||||
# fill state at its prior value (default mono) or fall back conservatively.
|
||||
assert _ink_of_first_xobject(body) in (Ink.mono, Ink.color)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"space, comps, expected",
|
||||
[
|
||||
('gray', [0.0], 'mono'),
|
||||
('gray', [0.263], 'gray'),
|
||||
('gray', [1.0], 'gray'), # white -> gray (harmless)
|
||||
('rgb', [0.0, 0.0, 0.0], 'mono'),
|
||||
('rgb', [0.263, 0.263, 0.263], 'gray'),
|
||||
('rgb', [0.8, 0.2, 0.2], 'color'),
|
||||
('rgb', [1.0, 1.0, 1.0], 'gray'),
|
||||
('cmyk', [0.0, 0.0, 0.0, 0.0], 'mono'), # white
|
||||
('cmyk', [0.0, 0.0, 0.0, 0.5], 'gray'),
|
||||
('cmyk', [0.5, 0.1, 0.0, 0.0], 'color'),
|
||||
('unknown', [0.5], 'color'), # conservative fallback
|
||||
],
|
||||
)
|
||||
def test_ink_from_components(space, comps, expected):
|
||||
assert _ink_from_components(space, comps) is Ink[expected]
|
||||
|
||||
|
||||
def _make_image_mask_pdf(path, content_fill: bytes):
|
||||
"""Build a 1-page PDF with one 8x8 image mask painted with content_fill.
|
||||
|
||||
content_fill is the color operator sequence emitted before drawing the
|
||||
mask, e.g. b"0.263 0.263 0.263 rg".
|
||||
"""
|
||||
pdf = pikepdf.Pdf.new()
|
||||
pdf.add_blank_page(page_size=(72, 72))
|
||||
# 8x8 1-bpc mask, each row padded to a byte (1 byte per row).
|
||||
mask_bytes = bytes([0x7E] * 8)
|
||||
mask = pikepdf.Stream(pdf, mask_bytes)
|
||||
mask.Type = pikepdf.Name.XObject
|
||||
mask.Subtype = pikepdf.Name.Image
|
||||
mask.Width = 8
|
||||
mask.Height = 8
|
||||
mask.ImageMask = True
|
||||
mask.BitsPerComponent = 1
|
||||
name = pdf.pages[0].add_resource(mask, pikepdf.Name.XObject)
|
||||
pdf.pages[0].Contents = pikepdf.Stream(
|
||||
pdf, b"q 72 0 0 72 0 0 cm %s %s Do Q" % (content_fill, bytes(name))
|
||||
)
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mask_gray_pdf(outdir):
|
||||
return _make_image_mask_pdf(outdir / 'mask_gray.pdf', b"0.263 0.263 0.263 rg")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mask_rgb_pdf(outdir):
|
||||
return _make_image_mask_pdf(outdir / 'mask_rgb.pdf', b"0.8 0.2 0.2 rg")
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mask_black_pdf(outdir):
|
||||
return _make_image_mask_pdf(outdir / 'mask_black.pdf', b"0 g")
|
||||
|
||||
|
||||
def test_imageinfo_ink_gray(mask_gray_pdf):
|
||||
image = pdfinfo.PdfInfo(mask_gray_pdf)[0].images[0]
|
||||
assert image.type_ == 'stencil'
|
||||
assert image.ink is Ink.gray
|
||||
|
||||
|
||||
def test_imageinfo_ink_color(mask_rgb_pdf):
|
||||
image = pdfinfo.PdfInfo(mask_rgb_pdf)[0].images[0]
|
||||
assert image.ink is Ink.color
|
||||
|
||||
|
||||
def test_imageinfo_ink_black(mask_black_pdf):
|
||||
image = pdfinfo.PdfInfo(mask_black_pdf)[0].images[0]
|
||||
assert image.ink is Ink.mono
|
||||
|
||||
|
||||
def test_imageinfo_ink_none_for_regular_image(eight_by_eight_regular_image):
|
||||
image = pdfinfo.PdfInfo(eight_by_eight_regular_image)[0].images[0]
|
||||
assert image.ink is None
|
||||
|
||||
|
||||
def test_fill_ink_cs_resets_color_to_black():
|
||||
# `cs` resets the fill color to the colorspace's initial value (black),
|
||||
# so a stale color set before `cs` must not leak to the drawn mask.
|
||||
assert _ink_of_first_xobject(b"0.8 0.2 0.2 rg /DeviceGray cs /Im0 Do") is Ink.mono
|
||||
|
||||
|
||||
def test_imageinfo_ink_inherited_in_form_xobject(outdir):
|
||||
# A mask drawn inside a Form XObject inherits the fill color set before the
|
||||
# Do that paints the form; the gray classification must reach the mask.
|
||||
pdf = pikepdf.Pdf.new()
|
||||
pdf.add_blank_page(page_size=(72, 72))
|
||||
|
||||
mask = pikepdf.Stream(pdf, bytes([0x7E] * 8))
|
||||
mask.Type = pikepdf.Name.XObject
|
||||
mask.Subtype = pikepdf.Name.Image
|
||||
mask.Width = 8
|
||||
mask.Height = 8
|
||||
mask.ImageMask = True
|
||||
mask.BitsPerComponent = 1
|
||||
|
||||
# Form draws the mask with no color of its own, inheriting the caller's.
|
||||
form = pikepdf.Stream(pdf, b"q 72 0 0 72 0 0 cm /Im0 Do Q")
|
||||
form.Type = pikepdf.Name.XObject
|
||||
form.Subtype = pikepdf.Name.Form
|
||||
form.BBox = [0, 0, 72, 72]
|
||||
form.Resources = pikepdf.Dictionary(XObject=pikepdf.Dictionary(Im0=mask))
|
||||
|
||||
fname = pdf.pages[0].add_resource(form, pikepdf.Name.XObject)
|
||||
pdf.pages[0].Contents = pikepdf.Stream(
|
||||
pdf, b"0.263 0.263 0.263 rg %s Do" % bytes(fname)
|
||||
)
|
||||
out = outdir / 'form_mask.pdf'
|
||||
pdf.save(out)
|
||||
|
||||
image = pdfinfo.PdfInfo(out)[0].images[0]
|
||||
assert image.type_ == 'stencil'
|
||||
assert image.ink is Ink.gray
|
||||
|
||||
@@ -6,6 +6,7 @@ from __future__ import annotations
|
||||
import warnings
|
||||
from unittest.mock import Mock
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from PIL import Image
|
||||
from reportlab.lib.units import inch
|
||||
@@ -13,8 +14,10 @@ from reportlab.lib.utils import ImageReader
|
||||
from reportlab.pdfgen.canvas import Canvas
|
||||
|
||||
from ocrmypdf import _pipeline, pdfinfo
|
||||
from ocrmypdf._pipeline import _select_raster_device
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pdfinfo import Encoding
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
|
||||
warnings.filterwarnings(
|
||||
"ignore", category=DeprecationWarning, module="reportlab.lib.rl_safe_eval"
|
||||
@@ -176,3 +179,39 @@ def test_should_visible_page_image_use_jpg(encodings, expected):
|
||||
pageinfo = Mock()
|
||||
pageinfo.images = [Mock(enc=enc) for enc in encodings]
|
||||
assert _pipeline.should_visible_page_image_use_jpg(pageinfo) == expected
|
||||
|
||||
|
||||
def _make_image_mask_pdf(path, content_fill: bytes):
|
||||
pdf = pikepdf.Pdf.new()
|
||||
pdf.add_blank_page(page_size=(72, 72))
|
||||
mask = pikepdf.Stream(pdf, bytes([0x7E] * 8))
|
||||
mask.Type = pikepdf.Name.XObject
|
||||
mask.Subtype = pikepdf.Name.Image
|
||||
mask.Width = 8
|
||||
mask.Height = 8
|
||||
mask.ImageMask = True
|
||||
mask.BitsPerComponent = 1
|
||||
name = pdf.pages[0].add_resource(mask, pikepdf.Name.XObject)
|
||||
pdf.pages[0].Contents = pikepdf.Stream(
|
||||
pdf, b"q 72 0 0 72 0 0 cm %s %s Do Q" % (content_fill, bytes(name))
|
||||
)
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
|
||||
def test_select_device_gray_mask(tmp_path):
|
||||
p = _make_image_mask_pdf(tmp_path / 'g.pdf', b"0.263 0.263 0.263 rg")
|
||||
pageinfo = pdfinfo.PdfInfo(p)[0]
|
||||
assert _select_raster_device(pageinfo) == GhostscriptRasterDevice.PNGGRAY
|
||||
|
||||
|
||||
def test_select_device_color_mask(tmp_path):
|
||||
p = _make_image_mask_pdf(tmp_path / 'c.pdf', b"0.8 0.2 0.2 rg")
|
||||
pageinfo = pdfinfo.PdfInfo(p)[0]
|
||||
assert _select_raster_device(pageinfo) == GhostscriptRasterDevice.PNG16M
|
||||
|
||||
|
||||
def test_select_device_black_mask_stays_mono(tmp_path):
|
||||
p = _make_image_mask_pdf(tmp_path / 'b.pdf', b"0 g")
|
||||
pageinfo = pdfinfo.PdfInfo(p)[0]
|
||||
assert _select_raster_device(pageinfo) == GhostscriptRasterDevice.PNGMONOD
|
||||
|
||||
+196
-2
@@ -12,9 +12,11 @@ import pikepdf
|
||||
import pytest
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf._exec import ghostscript
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._plugin_manager import get_plugin_manager
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
|
||||
from .conftest import check_ocrmypdf
|
||||
|
||||
@@ -213,6 +215,95 @@ class TestRasterizerHookDirect:
|
||||
assert result == img
|
||||
assert img.exists()
|
||||
|
||||
@pytest.mark.skipif(not PYPDFIUM_AVAILABLE, reason="pypdfium2 not installed")
|
||||
def test_pypdfium_pngmonod_produces_1bit(self, resources, tmp_path):
|
||||
"""Pngmonod is treated like pngmono by pypdfium: it yields a 1-bit PNG."""
|
||||
pm = get_plugin_manager([])
|
||||
options = OcrOptions(
|
||||
input_file=resources / 'graph.pdf',
|
||||
output_file=tmp_path / 'out.pdf',
|
||||
rasterizer='pypdfium',
|
||||
)
|
||||
|
||||
img = tmp_path / 'pngmonod_test.png'
|
||||
result = pm.rasterize_pdf_page(
|
||||
input_file=resources / 'graph.pdf',
|
||||
output_file=img,
|
||||
raster_device='pngmonod',
|
||||
raster_dpi=Resolution(50, 50),
|
||||
page_dpi=Resolution(50, 50),
|
||||
pageno=1,
|
||||
rotation=0,
|
||||
filter_vector=False,
|
||||
stop_on_soft_error=True,
|
||||
options=options,
|
||||
use_cropbox=False,
|
||||
)
|
||||
assert result == img
|
||||
with Image.open(img) as im:
|
||||
assert im.mode == '1'
|
||||
|
||||
|
||||
def _make_text_mask_pdf(path, fill: bytes):
|
||||
"""Build a letter page with a large text image mask painted with ``fill``.
|
||||
|
||||
The mask is a 1-bit stencil; ``fill`` is the color operator sequence that
|
||||
sets the paint color (e.g. ``b"0.263 0.263 0.263 rg"``). With a gray fill
|
||||
this reproduces issue #1688: the text is mid-gray, which is dithered into
|
||||
noise if rasterized to 1-bit but reads correctly once promoted to gray.
|
||||
"""
|
||||
from importlib.resources import as_file, files
|
||||
|
||||
from PIL import ImageDraw, ImageFont
|
||||
|
||||
w, h = 1700, 600
|
||||
im = Image.new('1', (w, h), 1) # 1 = white = "do not paint" under Decode [0 1]
|
||||
draw = ImageDraw.Draw(im)
|
||||
# Use a font bundled with ocrmypdf so this test is portable across platforms;
|
||||
# system fonts like DejaVu are not present on macOS/Windows CI runners.
|
||||
with as_file(files('ocrmypdf.data') / 'NotoSans-Regular.ttf') as font_path:
|
||||
font = ImageFont.truetype(str(font_path), 220)
|
||||
draw.text((40, 120), "TESTING", fill=0, font=font)
|
||||
|
||||
packed = im.tobytes() # 1-bpc, rows byte-padded, MSB first
|
||||
pdf = pikepdf.Pdf.new()
|
||||
pdf.add_blank_page(page_size=(612, 792))
|
||||
mask = pikepdf.Stream(pdf, packed)
|
||||
mask.Type = pikepdf.Name.XObject
|
||||
mask.Subtype = pikepdf.Name.Image
|
||||
mask.Width = w
|
||||
mask.Height = h
|
||||
mask.ImageMask = True
|
||||
mask.BitsPerComponent = 1
|
||||
name = pdf.pages[0].add_resource(mask, pikepdf.Name.XObject)
|
||||
pdf.pages[0].Contents = pikepdf.Stream(
|
||||
pdf, b"q 560 0 0 200 26 500 cm %s %s Do Q" % (fill, bytes(name))
|
||||
)
|
||||
pdf.save(path)
|
||||
return path
|
||||
|
||||
|
||||
@pytest.mark.parametrize("rasterizer", ['ghostscript', 'pypdfium'])
|
||||
def test_gray_mask_ocrs_to_text(tmp_path, rasterizer):
|
||||
"""A gray-painted text mask OCRs to real text on both rasterizers (#1688)."""
|
||||
if rasterizer == 'pypdfium' and not PYPDFIUM_AVAILABLE:
|
||||
pytest.skip("pypdfium2 not installed")
|
||||
|
||||
src = _make_text_mask_pdf(tmp_path / 'mask.pdf', b"0.263 0.263 0.263 rg")
|
||||
out = tmp_path / 'out.pdf'
|
||||
sidecar = tmp_path / 'out.txt'
|
||||
check_ocrmypdf(
|
||||
src,
|
||||
out,
|
||||
'--rasterizer',
|
||||
rasterizer,
|
||||
'--sidecar',
|
||||
str(sidecar),
|
||||
'--oversample',
|
||||
'300',
|
||||
)
|
||||
assert 'TESTING' in sidecar.read_text().upper()
|
||||
|
||||
|
||||
def _create_gradient_image(width: int, height: int) -> Image.Image:
|
||||
"""Create an image with multiple gradients to detect rasterization errors.
|
||||
@@ -418,6 +509,110 @@ class TestRasterizerWithNonStandardBoxes:
|
||||
assert pdfium_size == (400, 500), f"pypdfium size: {pdfium_size}"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def pdf_with_offset_mediabox_origin(tmp_path):
|
||||
"""Create a single-page PDF whose MediaBox has a non-zero origin.
|
||||
|
||||
Tools that crop/rotate non-destructively (e.g. PDF Arranger) shift the
|
||||
MediaBox origin rather than re-rendering content, producing a MediaBox like
|
||||
``[0, 440, 600, 800]`` with the visible content still inside the box. This
|
||||
fixture reproduces that shape with a full-page gradient so the visible
|
||||
region is unambiguously non-blank. Regression fixture for issue #1709.
|
||||
"""
|
||||
# Full-page gradient so the entire MediaBox region carries content.
|
||||
img = _create_gradient_image(600, 800)
|
||||
img_bytes = BytesIO()
|
||||
img.save(img_bytes, format='PNG')
|
||||
img_bytes.seek(0)
|
||||
|
||||
pdf_bytes = BytesIO()
|
||||
img2pdf.convert(
|
||||
img_bytes.read(),
|
||||
layout_fun=img2pdf.get_fixed_dpi_layout_fun((72, 72)),
|
||||
outputstream=pdf_bytes,
|
||||
**IMG2PDF_KWARGS,
|
||||
)
|
||||
pdf_bytes.seek(0)
|
||||
|
||||
pdf_path = tmp_path / 'offset_mediabox_origin.pdf'
|
||||
with pikepdf.open(pdf_bytes) as pdf:
|
||||
page = pdf.pages[0]
|
||||
# Shift the lower-left y origin so the box is [0, 440, 600, 800]: a
|
||||
# 600x360 visible region whose content lies entirely within the box.
|
||||
page.MediaBox = pikepdf.Array([0, 440, 600, 800])
|
||||
page.CropBox = pikepdf.Array([0, 440, 600, 800])
|
||||
pdf.save(pdf_path)
|
||||
|
||||
return pdf_path
|
||||
|
||||
|
||||
def _nonwhite_fraction(pdf_path, png_path) -> float:
|
||||
"""Rasterize page 1 of pdf_path and return the fraction of non-white pixels."""
|
||||
ghostscript.rasterize_pdf(
|
||||
pdf_path,
|
||||
png_path,
|
||||
raster_device=GhostscriptRasterDevice.PNGGRAY,
|
||||
raster_dpi=Resolution(72, 72),
|
||||
pageno=1,
|
||||
rotation=0,
|
||||
)
|
||||
with Image.open(png_path) as im:
|
||||
gray = im.convert('L')
|
||||
histogram = gray.histogram()
|
||||
total = sum(histogram)
|
||||
# Treat near-white (>= 250) as background; everything else is page content.
|
||||
nonwhite = sum(histogram[:250])
|
||||
return nonwhite / total
|
||||
|
||||
|
||||
class TestOffsetMediaBoxOrigin:
|
||||
"""Regression tests for issue #1709.
|
||||
|
||||
A non-zero MediaBox origin (e.g. from PDF Arranger crops) must not cause
|
||||
--force-ocr to drop the page content and emit a blank page. Both rasterizers
|
||||
are covered because the bug surfaced regardless of which one rendered.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'rasterizer',
|
||||
[
|
||||
'ghostscript',
|
||||
pytest.param(
|
||||
'pypdfium',
|
||||
marks=pytest.mark.skipif(
|
||||
not PYPDFIUM_AVAILABLE, reason="pypdfium2 not installed"
|
||||
),
|
||||
),
|
||||
],
|
||||
)
|
||||
def test_force_ocr_preserves_offset_origin_content(
|
||||
self, pdf_with_offset_mediabox_origin, rasterizer, outpdf, tmp_path
|
||||
):
|
||||
"""--force-ocr must preserve content when the MediaBox origin is non-zero."""
|
||||
# Sanity check: the input genuinely has content in its visible region.
|
||||
input_fraction = _nonwhite_fraction(
|
||||
pdf_with_offset_mediabox_origin, tmp_path / 'input.png'
|
||||
)
|
||||
assert input_fraction > 0.5, "test fixture should have a non-blank page"
|
||||
|
||||
check_ocrmypdf(
|
||||
pdf_with_offset_mediabox_origin,
|
||||
outpdf,
|
||||
'--force-ocr',
|
||||
'--rasterizer',
|
||||
rasterizer,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
|
||||
# The output page must not be blank: the visible content survives.
|
||||
output_fraction = _nonwhite_fraction(outpdf, tmp_path / 'output.png')
|
||||
assert output_fraction > 0.5, (
|
||||
f"output page is blank (non-white fraction {output_fraction:.3f}); "
|
||||
"content was dropped for a non-zero MediaBox origin (issue #1709)"
|
||||
)
|
||||
|
||||
|
||||
class TestRasterizerWithRotationAndBoxes:
|
||||
"""Test rasterizer + rotation + nonstandard boxes combinations."""
|
||||
|
||||
@@ -582,8 +777,7 @@ class TestRasterizerWithRotationAndBoxes:
|
||||
expected = self._get_expected_size(rotation)
|
||||
|
||||
assert abs(gs_img.size[0] - expected[0]) <= 2, (
|
||||
f"GS width at {rotation}°: {gs_img.size[0]}, "
|
||||
f"expected {expected[0]}"
|
||||
f"GS width at {rotation}°: {gs_img.size[0]}, expected {expected[0]}"
|
||||
)
|
||||
assert abs(gs_img.size[1] - expected[1]) <= 2, (
|
||||
f"GS height at {rotation}°: {gs_img.size[1]}, "
|
||||
|
||||
@@ -0,0 +1,101 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from pikepdf import Dictionary, Name, String
|
||||
|
||||
from ocrmypdf._graft import discard_text_search_index
|
||||
|
||||
from .conftest import check_ocrmypdf
|
||||
|
||||
# pylint: disable=redefined-outer-name
|
||||
|
||||
|
||||
def _add_search_index(pdf: pikepdf.Pdf, *, other_owner: bool = False) -> None:
|
||||
"""Attach an Adobe-style embedded search index to the document catalog."""
|
||||
pieceinfo = Dictionary(
|
||||
SearchIndex=Dictionary(
|
||||
LastModified=String("D:20240101000000Z"),
|
||||
Private=Dictionary(IndexFile=String("dummy.pdx")),
|
||||
)
|
||||
)
|
||||
if other_owner:
|
||||
pieceinfo[Name.SomeOtherApp] = Dictionary(
|
||||
LastModified=String("D:20240101000000Z")
|
||||
)
|
||||
pdf.Root.PieceInfo = pdf.make_indirect(pieceinfo)
|
||||
|
||||
|
||||
def test_discard_text_search_index_removes_only_search_index(resources):
|
||||
with pikepdf.open(resources / 'francais.pdf') as pdf:
|
||||
# No PieceInfo at all -> nothing to do
|
||||
assert not discard_text_search_index(pdf)
|
||||
|
||||
_add_search_index(pdf, other_owner=True)
|
||||
assert discard_text_search_index(pdf), "Expected file to be modified"
|
||||
|
||||
# SearchIndex gone, but the other application's private data is preserved
|
||||
assert Name.SearchIndex not in pdf.Root.PieceInfo
|
||||
assert Name.SomeOtherApp in pdf.Root.PieceInfo
|
||||
|
||||
# Idempotent: a second call finds nothing to remove
|
||||
assert not discard_text_search_index(pdf)
|
||||
|
||||
|
||||
def test_discard_text_search_index_drops_empty_pieceinfo(resources):
|
||||
with pikepdf.open(resources / 'francais.pdf') as pdf:
|
||||
_add_search_index(pdf, other_owner=False)
|
||||
assert discard_text_search_index(pdf)
|
||||
# PieceInfo held only the SearchIndex, so the whole husk is removed
|
||||
assert Name.PieceInfo not in pdf.Root
|
||||
|
||||
|
||||
def test_discard_text_search_index_tolerates_malformed_pieceinfo(resources):
|
||||
with pikepdf.open(resources / 'francais.pdf') as pdf:
|
||||
pdf.Root.PieceInfo = String("not a dictionary")
|
||||
assert not discard_text_search_index(pdf)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def pdf_with_search_index(resources, outdir):
|
||||
out = outdir / 'with_search_index.pdf'
|
||||
with pikepdf.open(resources / 'graph.pdf') as pdf:
|
||||
_add_search_index(pdf, other_owner=False)
|
||||
assert Name.SearchIndex in pdf.Root.PieceInfo
|
||||
pdf.save(out)
|
||||
return out
|
||||
|
||||
|
||||
def test_search_index_discarded_end_to_end(pdf_with_search_index, outpdf, caplog):
|
||||
caplog.set_level(logging.DEBUG)
|
||||
check_ocrmypdf(
|
||||
pdf_with_search_index,
|
||||
outpdf,
|
||||
'--output-type',
|
||||
'pdf',
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert Name.PieceInfo not in pdf.Root
|
||||
assert 'search index' in caplog.text.lower()
|
||||
|
||||
|
||||
def test_search_index_discarded_with_ocr_engine_none(pdf_with_search_index, outpdf):
|
||||
# Even in pure image-processing mode, OCRmyPDF rewrites the PDF, which
|
||||
# invalidates the embedded index, so it must still be discarded.
|
||||
check_ocrmypdf(
|
||||
pdf_with_search_index,
|
||||
outpdf,
|
||||
'--ocr-engine',
|
||||
'none',
|
||||
'--output-type',
|
||||
'pdf',
|
||||
)
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert Name.PieceInfo not in pdf.Root
|
||||
@@ -48,6 +48,33 @@ def test_stdout(ocrmypdf_exec, resources, outpdf):
|
||||
assert check_pdf(output_file)
|
||||
|
||||
|
||||
def test_stdout_protected_from_pollution(ocrmypdf_exec, resources, outpdf):
|
||||
if 'COV_CORE_DATAFILE' in os.environ:
|
||||
pytest.skip("Coverage uses stdout")
|
||||
|
||||
input_file = str(resources / 'francais.pdf')
|
||||
output_file = str(outpdf)
|
||||
|
||||
# A plugin deliberately writes garbage to stdout during the run. With stdout
|
||||
# protection active, that garbage must be diverted to stderr and never reach
|
||||
# the PDF we are writing to stdout.
|
||||
with open(output_file, 'wb') as output_stream:
|
||||
p_args = ocrmypdf_exec + [
|
||||
input_file,
|
||||
'-',
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
'--plugin',
|
||||
'tests/plugins/stdout_polluter.py',
|
||||
]
|
||||
p = run(p_args, stdout=output_stream, stderr=PIPE, stdin=DEVNULL, check=True)
|
||||
|
||||
assert check_pdf(output_file), "PDF on stdout was corrupted"
|
||||
with open(output_file, 'rb') as f:
|
||||
assert b'POLLUTION' not in f.read(), "pollution leaked into the PDF"
|
||||
assert b'POLLUTION' in p.stderr, "pollution was not diverted to stderr"
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.name == 'nt', reason='Windows does not support /dev/null')
|
||||
def test_dev_null(resources):
|
||||
if 'COV_CORE_DATAFILE' in os.environ:
|
||||
|
||||
@@ -0,0 +1,71 @@
|
||||
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Tests for --mode strip (remove the OCR text layer in place)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
|
||||
from ocrmypdf.exceptions import BadArgsError
|
||||
from ocrmypdf.pdfinfo import PdfInfo
|
||||
|
||||
from .conftest import check_ocrmypdf, run_ocrmypdf_api
|
||||
|
||||
|
||||
def _image_raw_bytes(pdf_path):
|
||||
"""Return raw (still-compressed) stream bytes of each image on page 1."""
|
||||
out = []
|
||||
with pikepdf.open(pdf_path) as pdf:
|
||||
resources = pdf.pages[0].get('/Resources', {})
|
||||
for _name, xobj in resources.get('/XObject', {}).items():
|
||||
if xobj.get('/Subtype') == pikepdf.Name.Image:
|
||||
out.append(bytes(xobj.read_raw_bytes()))
|
||||
return out
|
||||
|
||||
|
||||
def test_mode_strip_removes_ocr_layer(resources, outpdf):
|
||||
"""--mode strip removes the invisible OCR layer without rasterizing.
|
||||
|
||||
The page image is preserved byte-for-byte and the output is no larger than
|
||||
the input.
|
||||
"""
|
||||
input_pdf = resources / 'graph_ocred.pdf'
|
||||
assert PdfInfo(input_pdf, detailed_analysis=True)[0].has_text
|
||||
|
||||
out = check_ocrmypdf(
|
||||
input_pdf, outpdf, '--mode', 'strip', '--output-type', 'pdf', '--optimize', '0'
|
||||
)
|
||||
|
||||
info = PdfInfo(out, detailed_analysis=True)
|
||||
assert len(info) == 1, "page count must be unchanged"
|
||||
assert not info[0].has_text, "OCR text layer should be removed"
|
||||
assert _image_raw_bytes(out) == _image_raw_bytes(input_pdf), (
|
||||
"page image must be preserved byte-for-byte (no rasterization)"
|
||||
)
|
||||
assert out.stat().st_size <= input_pdf.stat().st_size, (
|
||||
"removing the text layer must not grow the file"
|
||||
)
|
||||
|
||||
|
||||
def test_mode_strip_preserves_visible_text(resources, outpdf):
|
||||
"""--mode strip leaves visible/born-digital text untouched (render mode != 3).
|
||||
|
||||
type3_font_nomapping.pdf is born-digital text with no images (the #1608
|
||||
case): its visible text must survive strip, which only removes invisible
|
||||
OCR text.
|
||||
"""
|
||||
input_pdf = resources / 'type3_font_nomapping.pdf'
|
||||
out = check_ocrmypdf(
|
||||
input_pdf, outpdf, '--mode', 'strip', '--output-type', 'pdf', '--optimize', '0'
|
||||
)
|
||||
assert PdfInfo(out, detailed_analysis=True)[0].has_text
|
||||
|
||||
|
||||
def test_mode_strip_rejects_image_processing_options(resources, no_outpdf):
|
||||
"""Options requiring rasterization/OCR are rejected in strip mode."""
|
||||
with pytest.raises(BadArgsError, match=r'--deskew'):
|
||||
run_ocrmypdf_api(
|
||||
resources / 'graph_ocred.pdf', no_outpdf, '--mode', 'strip', '--deskew'
|
||||
)
|
||||
@@ -190,6 +190,9 @@ class TestSystemFontProviderAvailableFonts:
|
||||
assert 'NotoSansCJK-Regular' in fonts
|
||||
assert 'NotoSansArabic-Regular' in fonts
|
||||
assert 'NotoSansThai-Regular' in fonts
|
||||
# Per-language CJK families (modern Google Fonts / Homebrew naming)
|
||||
assert 'NotoSansSC-Regular' in fonts
|
||||
assert 'NotoSansJP-Regular' in fonts
|
||||
|
||||
def test_fallback_font_raises(self):
|
||||
"""Test that get_fallback_font raises NotImplementedError."""
|
||||
@@ -198,6 +201,166 @@ class TestSystemFontProviderAvailableFonts:
|
||||
provider.get_fallback_font()
|
||||
|
||||
|
||||
class TestSystemFontProviderVariableFonts:
|
||||
"""Test discovery of variable fonts and non-static filename variants.
|
||||
|
||||
Homebrew casks and Google Fonts ship Noto fonts as variable fonts with
|
||||
bracketed axis filenames (e.g. ``NotoSansArabic[wdth,wght].ttf``) rather
|
||||
than the static ``NotoSansArabic-Regular.ttf``. See issue #1652.
|
||||
"""
|
||||
|
||||
@pytest.fixture
|
||||
def real_font_bytes(self):
|
||||
"""Bytes of a real, loadable font (content is irrelevant to the test)."""
|
||||
font_path = (
|
||||
Path(__file__).parent.parent
|
||||
/ "src"
|
||||
/ "ocrmypdf"
|
||||
/ "data"
|
||||
/ "NotoSans-Regular.ttf"
|
||||
)
|
||||
if not font_path.exists():
|
||||
pytest.skip("Builtin font not available")
|
||||
return font_path.read_bytes()
|
||||
|
||||
def _provider_for(self, tmp_path, filenames, real_font_bytes):
|
||||
"""Build a provider whose only font dir is tmp_path with given files."""
|
||||
for name in filenames:
|
||||
(tmp_path / name).write_bytes(real_font_bytes)
|
||||
provider = SystemFontProvider()
|
||||
provider._font_dirs = [tmp_path]
|
||||
return provider
|
||||
|
||||
def test_finds_variable_font_with_axes(self, tmp_path, real_font_bytes):
|
||||
"""A bracketed variable font satisfies a request for the static name."""
|
||||
provider = self._provider_for(
|
||||
tmp_path, ['NotoSansArabic[wdth,wght].ttf'], real_font_bytes
|
||||
)
|
||||
font = provider.get_font('NotoSansArabic-Regular')
|
||||
assert font is not None
|
||||
assert font.font_path.name == 'NotoSansArabic[wdth,wght].ttf'
|
||||
|
||||
def test_finds_weight_only_variable_font(self, tmp_path, real_font_bytes):
|
||||
"""A variable font with only a weight axis is also discovered."""
|
||||
provider = self._provider_for(
|
||||
tmp_path, ['NotoSansHebrew[wght].ttf'], real_font_bytes
|
||||
)
|
||||
assert provider.get_font('NotoSansHebrew-Regular') is not None
|
||||
|
||||
def test_variable_font_does_not_cross_match_other_script(
|
||||
self, tmp_path, real_font_bytes
|
||||
):
|
||||
"""The generic NotoSans request must not match a script-specific font."""
|
||||
provider = self._provider_for(
|
||||
tmp_path, ['NotoSansArabic[wdth,wght].ttf'], real_font_bytes
|
||||
)
|
||||
# NotoSans (Latin) must NOT be satisfied by NotoSansArabic.
|
||||
assert provider.get_font('NotoSans-Regular') is None
|
||||
|
||||
def test_does_not_match_ui_or_bold_variants(self, tmp_path, real_font_bytes):
|
||||
"""Width/UI and weight variants must not satisfy the Regular request."""
|
||||
provider = self._provider_for(
|
||||
tmp_path,
|
||||
['NotoSansArabicUI-Regular.ttf', 'NotoSansArabic-Bold.ttf'],
|
||||
real_font_bytes,
|
||||
)
|
||||
assert provider.get_font('NotoSansArabic-Regular') is None
|
||||
|
||||
def test_prefers_static_regular_over_variable(self, tmp_path, real_font_bytes):
|
||||
"""When both exist, the static Regular is preferred for predictability."""
|
||||
provider = self._provider_for(
|
||||
tmp_path,
|
||||
['NotoSansArabic[wdth,wght].ttf', 'NotoSansArabic-Regular.ttf'],
|
||||
real_font_bytes,
|
||||
)
|
||||
font = provider.get_font('NotoSansArabic-Regular')
|
||||
assert font is not None
|
||||
assert font.font_path.name == 'NotoSansArabic-Regular.ttf'
|
||||
|
||||
# --- Modern per-language CJK families (NotoSansSC/TC/HK/JP/KR) ---
|
||||
# Homebrew casks (font-noto-sans-sc, ...) and Google Fonts ship CJK as
|
||||
# variable fonts under these bases rather than the legacy NotoSansCJK*.
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'filename',
|
||||
[
|
||||
'NotoSansSC[wght].ttf', # Simplified Chinese (Homebrew/Google)
|
||||
'NotoSansTC[wght].ttf', # Traditional Chinese
|
||||
'NotoSansHK[wght].ttf', # Hong Kong
|
||||
'NotoSansJP[wght].ttf', # Japanese
|
||||
'NotoSansKR[wght].ttf', # Korean
|
||||
],
|
||||
)
|
||||
def test_finds_modern_cjk_variable_font(self, tmp_path, real_font_bytes, filename):
|
||||
"""A modern per-language CJK variable font satisfies NotoSansCJK."""
|
||||
provider = self._provider_for(tmp_path, [filename], real_font_bytes)
|
||||
font = provider.get_font('NotoSansCJK-Regular')
|
||||
assert font is not None
|
||||
assert font.font_path.name == filename
|
||||
|
||||
def test_finds_static_cjk_language_variant(self, tmp_path, real_font_bytes):
|
||||
"""A static per-language CJK Regular also satisfies NotoSansCJK."""
|
||||
provider = self._provider_for(
|
||||
tmp_path, ['NotoSansTC-Regular.otf'], real_font_bytes
|
||||
)
|
||||
assert provider.get_font('NotoSansCJK-Regular') is not None
|
||||
|
||||
def test_prefers_pan_cjk_over_language_variant(self, tmp_path, real_font_bytes):
|
||||
"""The pan-CJK family is preferred over a single-language variant."""
|
||||
provider = self._provider_for(
|
||||
tmp_path,
|
||||
['NotoSansSC[wght].ttf', 'NotoSansCJK[wght].ttf'],
|
||||
real_font_bytes,
|
||||
)
|
||||
font = provider.get_font('NotoSansCJK-Regular')
|
||||
assert font is not None
|
||||
assert font.font_path.name == 'NotoSansCJK[wght].ttf'
|
||||
|
||||
def test_modern_cjk_does_not_cross_match_latin(self, tmp_path, real_font_bytes):
|
||||
"""A CJK variable font must not satisfy the generic NotoSans request."""
|
||||
provider = self._provider_for(
|
||||
tmp_path, ['NotoSansSC[wght].ttf'], real_font_bytes
|
||||
)
|
||||
assert provider.get_font('NotoSans-Regular') is None
|
||||
|
||||
# --- Per-language CJK families reachable by their own logical name ---
|
||||
# Needed so MultiFontManager can prefer the family matching the document
|
||||
# language (NotoSansJP for Japanese, NotoSansSC for Simplified Chinese, ...).
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
'logical,filename',
|
||||
[
|
||||
('NotoSansSC-Regular', 'NotoSansSC[wght].ttf'),
|
||||
('NotoSansTC-Regular', 'NotoSansTC[wght].ttf'),
|
||||
('NotoSansHK-Regular', 'NotoSansHK[wght].ttf'),
|
||||
('NotoSansJP-Regular', 'NotoSansJP[wght].ttf'),
|
||||
('NotoSansKR-Regular', 'NotoSansKR[wght].ttf'),
|
||||
],
|
||||
)
|
||||
def test_per_language_cjk_logical_name_resolves(
|
||||
self, tmp_path, real_font_bytes, logical, filename
|
||||
):
|
||||
"""Each per-language CJK family is reachable by its own logical name."""
|
||||
provider = self._provider_for(tmp_path, [filename], real_font_bytes)
|
||||
font = provider.get_font(logical)
|
||||
assert font is not None
|
||||
assert font.font_path.name == filename
|
||||
|
||||
def test_per_language_cjk_static_resolves(self, tmp_path, real_font_bytes):
|
||||
"""A static per-language Regular also resolves by logical name."""
|
||||
provider = self._provider_for(
|
||||
tmp_path, ['NotoSansJP-Regular.otf'], real_font_bytes
|
||||
)
|
||||
assert provider.get_font('NotoSansJP-Regular') is not None
|
||||
|
||||
def test_per_language_cjk_does_not_cross_match(self, tmp_path, real_font_bytes):
|
||||
"""A JP font must not satisfy an SC request (distinct families)."""
|
||||
provider = self._provider_for(
|
||||
tmp_path, ['NotoSansJP[wght].ttf'], real_font_bytes
|
||||
)
|
||||
assert provider.get_font('NotoSansSC-Regular') is None
|
||||
|
||||
|
||||
# --- ChainedFontProvider Tests ---
|
||||
|
||||
|
||||
|
||||
+52
-5
@@ -3,9 +3,12 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from pikepdf import Name
|
||||
|
||||
import ocrmypdf
|
||||
from ocrmypdf.pdfinfo import PdfInfo
|
||||
|
||||
|
||||
def test_block_tagged(resources):
|
||||
@@ -13,6 +16,25 @@ def test_block_tagged(resources):
|
||||
ocrmypdf.ocr(resources / 'tagged.pdf', '_.pdf')
|
||||
|
||||
|
||||
def test_detect_structure_tree(resources):
|
||||
assert PdfInfo(resources / 'tagged.pdf').has_structure_tree is True
|
||||
|
||||
|
||||
def test_structure_tree_without_markinfo_blocks(resources, tmp_path):
|
||||
"""A PDF with a structure tree but no /MarkInfo flag is still blocked."""
|
||||
untagged = tmp_path / 'struct_only.pdf'
|
||||
with pikepdf.open(resources / 'tagged.pdf') as pdf:
|
||||
del pdf.Root.MarkInfo
|
||||
pdf.save(untagged)
|
||||
|
||||
info = PdfInfo(untagged)
|
||||
assert info.is_tagged is False
|
||||
assert info.has_structure_tree is True
|
||||
|
||||
with pytest.raises(ocrmypdf.exceptions.TaggedPDFError):
|
||||
ocrmypdf.ocr(untagged, '_.pdf')
|
||||
|
||||
|
||||
def test_force_tagged_warns(resources, outpdf, caplog):
|
||||
caplog.set_level('WARNING')
|
||||
ocrmypdf.ocr(
|
||||
@@ -21,24 +43,31 @@ def test_force_tagged_warns(resources, outpdf, caplog):
|
||||
force_ocr=True,
|
||||
plugins=['tests/plugins/tesseract_noop.py'],
|
||||
)
|
||||
assert 'marked as a Tagged PDF' in caplog.text
|
||||
assert 'structural markup' in caplog.text
|
||||
|
||||
|
||||
def test_tagged_pdf_mode_ignore_with_skip_text(resources, outpdf, caplog):
|
||||
"""Ignore tagged_pdf_mode should warn but not error."""
|
||||
"""Ignore tagged_pdf_mode should warn but not error, and keep structure."""
|
||||
caplog.set_level('WARNING')
|
||||
ocrmypdf.ocr(
|
||||
resources / 'tagged.pdf',
|
||||
outpdf,
|
||||
tagged_pdf_mode='ignore',
|
||||
skip_text=True, # Tagged PDF has text, so skip pages with text
|
||||
# output_type=pdf avoids the Ghostscript PDF/A step, whose treatment of
|
||||
# the structure tree is version-dependent (Ghostscript >= 10 discards it,
|
||||
# 9.x preserves it). We only want to assert OCRmyPDF's own behavior here.
|
||||
output_type='pdf',
|
||||
plugins=['tests/plugins/tesseract_noop.py'],
|
||||
)
|
||||
assert 'marked as a Tagged PDF' in caplog.text
|
||||
assert 'structural markup' in caplog.text
|
||||
# skip-text leaves the text pages untouched, so OCRmyPDF keeps the structure tree
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert Name.StructTreeRoot in pdf.Root
|
||||
|
||||
|
||||
def test_tagged_pdf_mode_ignore_with_force(resources, outpdf, caplog):
|
||||
"""Ignore tagged_pdf_mode with force mode should warn."""
|
||||
"""Ignore tagged_pdf_mode with force mode should warn and discard structure."""
|
||||
caplog.set_level('WARNING')
|
||||
ocrmypdf.ocr(
|
||||
resources / 'tagged.pdf',
|
||||
@@ -47,4 +76,22 @@ def test_tagged_pdf_mode_ignore_with_force(resources, outpdf, caplog):
|
||||
force_ocr=True,
|
||||
plugins=['tests/plugins/tesseract_noop.py'],
|
||||
)
|
||||
assert 'marked as a Tagged PDF' in caplog.text
|
||||
assert 'structural markup' in caplog.text
|
||||
# force-ocr rasterizes every page, destroying the MCIDs the tree relies on
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert Name.StructTreeRoot not in pdf.Root
|
||||
assert Name.MarkInfo not in pdf.Root
|
||||
|
||||
|
||||
def test_tagged_pdf_mode_ignore_with_redo(resources, outpdf):
|
||||
"""Redo mode rewrites the text layer, so structure is discarded."""
|
||||
ocrmypdf.ocr(
|
||||
resources / 'tagged.pdf',
|
||||
outpdf,
|
||||
tagged_pdf_mode='ignore',
|
||||
redo_ocr=True,
|
||||
plugins=['tests/plugins/tesseract_noop.py'],
|
||||
)
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert Name.StructTreeRoot not in pdf.Root
|
||||
assert Name.MarkInfo not in pdf.Root
|
||||
|
||||
+15
-1
@@ -128,7 +128,6 @@ def test_timeout(caplog):
|
||||
(b'Error in boxClipToRectangle', ''),
|
||||
(b'an unexpected error', 'an unexpected error'),
|
||||
(b'a dire warning', 'a dire warning'),
|
||||
(b'read_params_file something', 'read_params_file'),
|
||||
(b'an innocent message', 'innocent'),
|
||||
(b'\x7f\x7f\x80innocent unicode failure', 'innocent'),
|
||||
],
|
||||
@@ -142,12 +141,27 @@ def test_tesseract_log_output(caplog, in_, logged):
|
||||
assert logged in caplog.text
|
||||
|
||||
|
||||
def test_tesseract_log_output_diacritics_raw(caplog):
|
||||
"""Diacritics branch keeps the interpreted hint and surfaces raw (#1566)."""
|
||||
caplog.set_level(logging.DEBUG)
|
||||
tesseract.tesseract_log_output(b'lots of diacritics blah blah')
|
||||
assert 'possibly poor OCR' in caplog.text # interpreted hint retained
|
||||
assert 'lots of diacritics blah blah' in caplog.text # raw message surfaced
|
||||
|
||||
|
||||
def test_tesseract_log_output_raises(caplog):
|
||||
with pytest.raises(tesseract.TesseractConfigError):
|
||||
tesseract.tesseract_log_output(b'parameter not found: moo')
|
||||
assert 'not found' in caplog.text
|
||||
|
||||
|
||||
def test_tesseract_log_output_raises_on_missing_config(caplog):
|
||||
with pytest.raises(tesseract.TesseractConfigError) as excinfo:
|
||||
tesseract.tesseract_log_output(b"read_params_file: Can't open hocr")
|
||||
assert 'hocr' in excinfo.value.args[0]
|
||||
assert 'read_params_file' in caplog.text
|
||||
|
||||
|
||||
def test_blocked_language(resources, no_outpdf):
|
||||
infile = resources / 'masks.pdf'
|
||||
for bad_lang in ['osd', 'equ']:
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
from pikepdf import Name
|
||||
|
||||
from ocrmypdf._graft import discard_page_thumbnails
|
||||
|
||||
from .conftest import check_ocrmypdf
|
||||
|
||||
# pylint: disable=redefined-outer-name
|
||||
|
||||
|
||||
def _add_thumbnail(pdf: pikepdf.Pdf, pageindex: int = 0) -> None:
|
||||
"""Attach a minimal /Thumb image XObject to a page."""
|
||||
width, height = 4, 4
|
||||
thumb = pikepdf.Stream(pdf, b'\x00' * (width * height))
|
||||
thumb.Type = Name.XObject
|
||||
thumb.Subtype = Name.Image
|
||||
thumb.Width = width
|
||||
thumb.Height = height
|
||||
thumb.ColorSpace = Name.DeviceGray
|
||||
thumb.BitsPerComponent = 8
|
||||
pdf.pages[pageindex].obj.Thumb = pdf.make_indirect(thumb)
|
||||
|
||||
|
||||
def test_discard_page_thumbnails_removes_thumbnails(resources):
|
||||
with pikepdf.open(resources / 'francais.pdf') as pdf:
|
||||
# No thumbnails -> nothing to do
|
||||
assert discard_page_thumbnails(pdf) == 0
|
||||
|
||||
_add_thumbnail(pdf, 0)
|
||||
assert Name.Thumb in pdf.pages[0].obj
|
||||
|
||||
assert discard_page_thumbnails(pdf) == 1
|
||||
assert Name.Thumb not in pdf.pages[0].obj
|
||||
|
||||
# Idempotent: a second call finds nothing to remove
|
||||
assert discard_page_thumbnails(pdf) == 0
|
||||
|
||||
|
||||
def test_discard_page_thumbnails_counts_each_page(resources):
|
||||
with pikepdf.open(resources / 'multipage.pdf') as pdf:
|
||||
assert len(pdf.pages) >= 2
|
||||
_add_thumbnail(pdf, 0)
|
||||
_add_thumbnail(pdf, 1)
|
||||
assert discard_page_thumbnails(pdf) == 2
|
||||
assert all(Name.Thumb not in page.obj for page in pdf.pages)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def pdf_with_thumbnail(resources, outdir):
|
||||
out = outdir / 'with_thumbnail.pdf'
|
||||
with pikepdf.open(resources / 'graph.pdf') as pdf:
|
||||
_add_thumbnail(pdf, 0)
|
||||
assert Name.Thumb in pdf.pages[0].obj
|
||||
pdf.save(out)
|
||||
return out
|
||||
|
||||
|
||||
def test_thumbnail_discarded_end_to_end(pdf_with_thumbnail, outpdf, caplog):
|
||||
caplog.set_level(logging.DEBUG)
|
||||
check_ocrmypdf(
|
||||
pdf_with_thumbnail,
|
||||
outpdf,
|
||||
'--output-type',
|
||||
'pdf',
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert all(Name.Thumb not in page.obj for page in pdf.pages)
|
||||
assert 'thumbnail' in caplog.text.lower()
|
||||
|
||||
|
||||
def test_thumbnail_discarded_with_ocr_engine_none(pdf_with_thumbnail, outpdf):
|
||||
# Even in pure image-processing mode, OCRmyPDF rewrites the PDF, which can
|
||||
# alter page appearance, so the stale thumbnail must still be discarded.
|
||||
check_ocrmypdf(
|
||||
pdf_with_thumbnail,
|
||||
outpdf,
|
||||
'--ocr-engine',
|
||||
'none',
|
||||
'--output-type',
|
||||
'pdf',
|
||||
)
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert all(Name.Thumb not in page.obj for page in pdf.pages)
|
||||
Reference in New Issue
Block a user