Compare commits
192
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b7377df7af | ||
|
|
6542d80064 | ||
|
|
7346e0f637 | ||
|
|
aa6a32e7d1 | ||
|
|
ea99758747 | ||
|
|
4942751a1b | ||
|
|
be06e3184a | ||
|
|
39bf09f1eb | ||
|
|
aaffc46f73 | ||
|
|
0277b3b3ba | ||
|
|
0817542883 | ||
|
|
6f4744dd20 | ||
|
|
5d49f75c56 | ||
|
|
5a824ddd8c | ||
|
|
54bf03a454 | ||
|
|
009754d137 | ||
|
|
f0a3a74374 | ||
|
|
178d339c8e | ||
|
|
d3f8d01227 | ||
|
|
b60df59c62 | ||
|
|
640b3062b2 | ||
|
|
ef903db360 | ||
|
|
9cda02317b | ||
|
|
92a2fe880a | ||
|
|
089f46690a | ||
|
|
e45c40b063 | ||
|
|
bbac5307f2 | ||
|
|
6167783696 | ||
|
|
3d291e72c0 | ||
|
|
efebe9ca2e | ||
|
|
12ec97f732 | ||
|
|
3f1aceade2 | ||
|
|
212b28e602 | ||
|
|
dfdb32995e | ||
|
|
273826377e | ||
|
|
5569d4db07 | ||
|
|
8de7b05fb9 | ||
|
|
72ce05768e | ||
|
|
3dc68778fc | ||
|
|
1aec92b919 | ||
|
|
43d3448709 | ||
|
|
7512b1042a | ||
|
|
efe83e8c54 | ||
|
|
a13d27bfb5 | ||
|
|
ea7ad7d683 | ||
|
|
8b20bb3c5b | ||
|
|
320876a6d1 | ||
|
|
dfbb4c9275 | ||
|
|
d4f5c2d160 | ||
|
|
263d6034be | ||
|
|
de403f6d5e | ||
|
|
86b6f2c907 | ||
|
|
e6fab76918 | ||
|
|
334918d0f7 | ||
|
|
d6329489ce | ||
|
|
e6d240ee93 | ||
|
|
ff45e54c07 | ||
|
|
e0ee0882ef | ||
|
|
3d17419a6c | ||
|
|
476ec12383 | ||
|
|
e99177ada7 | ||
|
|
e95ec9c497 | ||
|
|
82f30bfbec | ||
|
|
d1437e6bbc | ||
|
|
c669d30642 | ||
|
|
3613b30ca8 | ||
|
|
0d4c3bcdcf | ||
|
|
8a8d515933 | ||
|
|
11de13ecfe | ||
|
|
58642d8411 | ||
|
|
7e42d3c771 | ||
|
|
5cb5d7a682 | ||
|
|
37e71dece6 | ||
|
|
df84945773 | ||
|
|
b5a6a9f9f1 | ||
|
|
ed36aefe48 | ||
|
|
32013f4294 | ||
|
|
8f2bcc2c64 | ||
|
|
015b53ae30 | ||
|
|
164cf2dc8a | ||
|
|
98d6d02704 | ||
|
|
5efb98931d | ||
|
|
2f4e47213f | ||
|
|
94c8123bd7 | ||
|
|
0db130e1c3 | ||
|
|
91b6a818f5 | ||
|
|
6bc9499e68 | ||
|
|
09f2d6c386 | ||
|
|
87f918f58c | ||
|
|
80e77fb021 | ||
|
|
fa9c5b3fae | ||
|
|
3d17a60a54 | ||
|
|
c33f073d4f | ||
|
|
5d7b5742e4 | ||
|
|
c391b2b7d0 | ||
|
|
0250929150 | ||
|
|
9748208e68 | ||
|
|
e4b0c04be4 | ||
|
|
efb83ad64f | ||
|
|
08e40f96e8 | ||
|
|
3f6feb1dcc | ||
|
|
ab6553f4ff | ||
|
|
cedca9fa1f | ||
|
|
3f40118022 | ||
|
|
b18b1da6d0 | ||
|
|
14fb9f56e8 | ||
|
|
8709cf506b | ||
|
|
9a92eb40df | ||
|
|
0a59c210f9 | ||
|
|
0b370fdd15 | ||
|
|
1c16dd26f7 | ||
|
|
c355d927ba | ||
|
|
c993857752 | ||
|
|
84f5fe9ee0 | ||
|
|
3336d67e77 | ||
|
|
73e16e7821 | ||
|
|
6f1d37d78f | ||
|
|
2ed82de2e0 | ||
|
|
c43903fa14 | ||
|
|
1c89cacfef | ||
|
|
75714fe43e | ||
|
|
e371ce95ca | ||
|
|
716a2e22c3 | ||
|
|
10e6019ada | ||
|
|
89c76b5145 | ||
|
|
83c04e6399 | ||
|
|
7fdeeb3635 | ||
|
|
5be368fe75 | ||
|
|
91c5b1e480 | ||
|
|
73154b97ba | ||
|
|
76a40759ae | ||
|
|
12ce565e98 | ||
|
|
9f46126859 | ||
|
|
11849e5a70 | ||
|
|
e30c00cc26 | ||
|
|
001b403657 | ||
|
|
851c61ee85 | ||
|
|
f5ebd23b8f | ||
|
|
81118c6195 | ||
|
|
834b60a02a | ||
|
|
47e3b5b4d2 | ||
|
|
d9346cc3d8 | ||
|
|
4e974ebd46 | ||
|
|
6f2b8408c1 | ||
|
|
1dba941261 | ||
|
|
ef76625abb | ||
|
|
57bb554a70 | ||
|
|
5b9d6f979e | ||
|
|
b588e3bfd7 | ||
|
|
a35dd1f9ee | ||
|
|
bf46f4fe35 | ||
|
|
55b76338a8 | ||
|
|
2af7b1c179 | ||
|
|
69f4cca9b6 | ||
|
|
59190ef643 | ||
|
|
910ccccc7d | ||
|
|
0c15ff594c | ||
|
|
e19ea653aa | ||
|
|
a899f0d59a | ||
|
|
b4e8e9dac9 | ||
|
|
aca5eb626b | ||
|
|
bd4a74de0e | ||
|
|
10b71937c4 | ||
|
|
5890d1855e | ||
|
|
3da952a23d | ||
|
|
716ce6324c | ||
|
|
76fe2f7e28 | ||
|
|
c85c8941d3 | ||
|
|
9a0dadbd4c | ||
|
|
4d7e398c4b | ||
|
|
56c0b41f97 | ||
|
|
5c83dab8a7 | ||
|
|
e62e73e441 | ||
|
|
d68e2f6e34 | ||
|
|
1684982cde | ||
|
|
4d97dfd218 | ||
|
|
a35fcc9c43 | ||
|
|
3dd4cde7ce | ||
|
|
92beb474a5 | ||
|
|
9dcd882c83 | ||
|
|
9d8aa5a0c3 | ||
|
|
e036a902ae | ||
|
|
0a980fb11b | ||
|
|
3abe8f71c7 | ||
|
|
64f45b7fdb | ||
|
|
7e939ad44d | ||
|
|
297fb786a0 | ||
|
|
ad30dd94f7 | ||
|
|
e77f79ac6f | ||
|
|
c84fc56e45 | ||
|
|
4babdfcfbf | ||
|
|
8930efe787 |
+43
-8
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM ubuntu:25.04 AS base
|
||||
FROM ubuntu:26.04 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -40,7 +40,7 @@ RUN \
|
||||
WORKDIR /app
|
||||
|
||||
# Copy uv from ghcr
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9.8 /uv /uvx /bin/
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.21 /uv /uvx /bin/
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
@@ -55,15 +55,13 @@ RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
COPY . /app
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen \
|
||||
--extra webservice --extra watcher --no-dev \
|
||||
--extra webservice --extra watcher --extra webui --no-dev \
|
||||
--no-install-package pyarrow
|
||||
|
||||
FROM base
|
||||
|
||||
RUN apt-get update && apt-get install -y software-properties-common
|
||||
|
||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr5
|
||||
|
||||
# Tesseract 5 ships in the Ubuntu archive as of 24.04, so no third-party PPA is
|
||||
# needed. (Previously this used ppa:alex-p/tesseract-ocr5.)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ghostscript \
|
||||
fonts-droid-fallback \
|
||||
@@ -81,6 +79,18 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
unpaper \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Create a non-root user to run the application (defense in depth). The build
|
||||
# stages above need root to install packages, but the entrypoint should not.
|
||||
# A fixed uid/gid of 1000 keeps `--user`/`--userns keep-id` mappings predictable
|
||||
# and matches the --chown below. See docs/docker.md for the volume/permissions
|
||||
# implications under rootless vs rootful Docker.
|
||||
# The Ubuntu base ships a default "ubuntu" user at uid/gid 1000; remove it so
|
||||
# "app" can claim that uid for parity with the Alpine image.
|
||||
RUN userdel -r ubuntu 2>/dev/null; groupdel ubuntu 2>/dev/null; \
|
||||
groupadd -g 1000 app \
|
||||
&& useradd -u 1000 -g app -m -d /home/app app
|
||||
ENV HOME=/home/app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
||||
@@ -90,9 +100,34 @@ COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
ln -s /app/misc/watcher.py /app/watcher.py && \
|
||||
chown app:app /app
|
||||
|
||||
# Default working directory for bind-mounted data, so relative input/output
|
||||
# paths work without passing --workdir (e.g. `-v "$PWD:/data" in.pdf out.pdf`).
|
||||
# The webservice/watcher are run by absolute path (/app/*.py), unaffected by this.
|
||||
RUN mkdir -p /data && chown app:app /data
|
||||
|
||||
# Scratch space for the batch web interface (webui/). Uploads and results live
|
||||
# here and are deleted once their batch expires; nothing in it needs to
|
||||
# survive a restart, so it is a good candidate for a tmpfs mount.
|
||||
RUN mkdir -p /var/tmp/ocrmypdf-webui && chown app:app /var/tmp/ocrmypdf-webui
|
||||
|
||||
WORKDIR /data
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
# webui/ is a top-level package in the source tree rather than part of the
|
||||
# installed ocrmypdf distribution, so it has to be on the import path.
|
||||
ENV PYTHONPATH="/app"
|
||||
|
||||
# Batch web interface. Not published by the default entrypoint; start it with
|
||||
# docker run -p 8000:8000 --entrypoint python3 <image> -m webui
|
||||
EXPOSE 8000
|
||||
|
||||
# Drop privileges: run the entrypoint (ocrmypdf, or the webservice/watcher when
|
||||
# overridden) as the unprivileged app user. Override with `--user root` if you
|
||||
# need root inside a running container (e.g. to apt install extra packages).
|
||||
USER app
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM alpine:3.22 AS base
|
||||
FROM alpine:3.24 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -22,7 +22,7 @@ RUN apk add --no-cache \
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9.8 /uv /uvx /bin/
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.11.21 /uv /uvx /bin/
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
@@ -39,7 +39,7 @@ RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
COPY . /app
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen \
|
||||
--extra webservice --extra watcher --no-dev \
|
||||
--extra webservice --extra watcher --extra webui --no-dev \
|
||||
--no-install-package pyarrow
|
||||
|
||||
FROM base
|
||||
@@ -62,14 +62,48 @@ RUN apk add --no-cache \
|
||||
unpaper \
|
||||
&& rm -rf /var/cache/apk/*
|
||||
|
||||
# Create a non-root user to run the application (defense in depth). The build
|
||||
# stages above need root to install packages, but the entrypoint should not.
|
||||
# A fixed uid/gid of 1000 keeps `--user`/`--userns keep-id` mappings predictable
|
||||
# and matches the --chown below. See docs/docker.md for the volume/permissions
|
||||
# implications under rootless vs rootful Docker.
|
||||
RUN addgroup -g 1000 app \
|
||||
&& adduser -u 1000 -G app -D -h /home/app app
|
||||
ENV HOME=/home/app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
ln -s /app/misc/watcher.py /app/watcher.py && \
|
||||
chown app:app /app
|
||||
|
||||
# Default working directory for bind-mounted data, so relative input/output
|
||||
# paths work without passing --workdir (e.g. `-v "$PWD:/data" in.pdf out.pdf`).
|
||||
# The webservice/watcher are run by absolute path (/app/*.py), unaffected by this.
|
||||
RUN mkdir -p /data && chown app:app /data
|
||||
|
||||
# Scratch space for the batch web interface (webui/). Uploads and results live
|
||||
# here and are deleted once their batch expires; nothing in it needs to
|
||||
# survive a restart, so it is a good candidate for a tmpfs mount.
|
||||
RUN mkdir -p /var/tmp/ocrmypdf-webui && chown app:app /var/tmp/ocrmypdf-webui
|
||||
|
||||
WORKDIR /data
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
# webui/ is a top-level package in the source tree rather than part of the
|
||||
# installed ocrmypdf distribution, so it has to be on the import path.
|
||||
ENV PYTHONPATH="/app"
|
||||
|
||||
# Batch web interface. Not published by the default entrypoint; start it with
|
||||
# docker run -p 8000:8000 --entrypoint python3 <image> -m webui
|
||||
EXPOSE 8000
|
||||
|
||||
# Drop privileges: run the entrypoint (ocrmypdf, or the webservice/watcher when
|
||||
# overridden) as the unprivileged app user. Override with `--user root` if you
|
||||
# need root inside a running container (e.g. to apk add extra packages).
|
||||
USER app
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
|
||||
+72
-77
@@ -9,15 +9,34 @@ on:
|
||||
- ci
|
||||
- release/*
|
||||
- feature/*
|
||||
tags:
|
||||
- v*
|
||||
paths-ignore:
|
||||
- README*
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
lint:
|
||||
name: Lint (prek)
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Run prek
|
||||
run: |
|
||||
uv run prek run --all-files
|
||||
|
||||
test_linux:
|
||||
name: Test ${{ matrix.os }} with Python ${{ matrix.python }}
|
||||
needs: lint
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -33,9 +52,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -43,7 +60,7 @@ jobs:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
@@ -91,7 +108,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v5
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -100,6 +117,7 @@ jobs:
|
||||
|
||||
test_macos:
|
||||
name: Test macOS
|
||||
needs: lint
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -111,9 +129,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install Homebrew deps
|
||||
continue-on-error: true
|
||||
@@ -135,7 +151,7 @@ jobs:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
@@ -155,7 +171,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v5
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -164,6 +180,7 @@ jobs:
|
||||
|
||||
test_windows:
|
||||
name: Test Windows
|
||||
needs: lint
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
@@ -175,9 +192,7 @@ jobs:
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -185,7 +200,7 @@ jobs:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
@@ -204,7 +219,7 @@ jobs:
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v5
|
||||
uses: codecov/codecov-action@v7
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
@@ -215,9 +230,7 @@ jobs:
|
||||
name: Build sdist and wheels
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
@@ -228,71 +241,57 @@ jobs:
|
||||
run: |
|
||||
uv build --sdist --wheel
|
||||
|
||||
- uses: actions/upload-artifact@v6
|
||||
- uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: artifact
|
||||
path: |
|
||||
./dist/*.whl
|
||||
./dist/*.tar.gz
|
||||
|
||||
upload_pypi:
|
||||
name: Deploy artifacts to PyPI
|
||||
stage_release:
|
||||
name: Stage release artifacts
|
||||
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||
runs-on: ubuntu-latest
|
||||
environment: release
|
||||
if: github.ref == 'refs/heads/main'
|
||||
permissions:
|
||||
id-token: write # mandatory for PyPI publishing
|
||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||
steps:
|
||||
- uses: actions/download-artifact@v7
|
||||
with:
|
||||
name: artifact
|
||||
path: dist
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
|
||||
create_release:
|
||||
name: Create GitHub release
|
||||
needs: [upload_pypi]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||
permissions:
|
||||
# Required to create a release
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/download-artifact@v7
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- uses: actions/download-artifact@v8
|
||||
with:
|
||||
name: artifact
|
||||
path: dist
|
||||
|
||||
- name: Sign the dists with Sigstore
|
||||
uses: sigstore/gh-action-sigstore-python@v3.2.0
|
||||
with:
|
||||
inputs: |
|
||||
./dist/*.tar.gz
|
||||
./dist/*.whl
|
||||
- name: Read version from source
|
||||
id: version
|
||||
run: |
|
||||
VERSION=$(python3 -c "exec(open('src/ocrmypdf/_version.py').read()); print(__version__)")
|
||||
echo "version=$VERSION" >> $GITHUB_OUTPUT
|
||||
|
||||
- name: Create GitHub Release
|
||||
- name: Create or update draft release
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: >-
|
||||
gh release create
|
||||
"$GITHUB_REF_NAME"
|
||||
--repo "$GITHUB_REPOSITORY"
|
||||
--notes ""
|
||||
run: |
|
||||
TAG="v${{ steps.version.outputs.version }}"
|
||||
|
||||
- name: Upload artifact signatures to GitHub Release
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
# Upload to GitHub Release using the `gh` CLI.
|
||||
# `dist/` contains the built packages, and the
|
||||
# sigstore-produced signatures and certificates.
|
||||
run: >-
|
||||
gh release upload
|
||||
"$GITHUB_REF_NAME" dist/**
|
||||
--repo "$GITHUB_REPOSITORY"
|
||||
# If release.yml already published this version, _version.py may
|
||||
# still reflect it until the next version bump commit. Don't
|
||||
# re-draft an already-published release on later pushes to main.
|
||||
if [[ "$(gh release view "$TAG" --json isDraft --jq .isDraft 2>/dev/null)" == "false" ]]; then
|
||||
echo "Release $TAG is already published; skipping."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Delete existing draft release if it exists (ignore errors)
|
||||
gh release delete "$TAG" --yes 2>/dev/null || true
|
||||
|
||||
# Create new draft release with all artifacts
|
||||
gh release create "$TAG" \
|
||||
--draft \
|
||||
--title "$TAG" \
|
||||
--notes "Draft release - will be updated when tag is pushed" \
|
||||
dist/*
|
||||
|
||||
docker_ubuntu:
|
||||
name: Build Ubuntu-based Docker image
|
||||
@@ -313,22 +312,20 @@ jobs:
|
||||
- name: Set image name
|
||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
||||
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
username: jbarlow83
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
uses: docker/setup-qemu-action@v4
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@v4
|
||||
|
||||
- name: Print image tag
|
||||
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
||||
@@ -361,19 +358,17 @@ jobs:
|
||||
- name: Set image name
|
||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf-alpine" >> $GITHUB_ENV
|
||||
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
username: jbarlow83
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@v4
|
||||
|
||||
- name: Print image tag
|
||||
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
name: Publish Release
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "v*"
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
name: Publish release
|
||||
runs-on: ubuntu-latest
|
||||
environment:
|
||||
name: release
|
||||
url: https://pypi.org/p/ocrmypdf
|
||||
permissions:
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
|
||||
- name: Download artifacts from draft release
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
mkdir -p dist
|
||||
gh release download "$GITHUB_REF_NAME" --dir dist --pattern '*.whl'
|
||||
gh release download "$GITHUB_REF_NAME" --dir dist --pattern '*.tar.gz'
|
||||
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
|
||||
# PyPI doesn't support sigstore publishing, so generate after publishing to PyPI
|
||||
- name: Sign the dists with Sigstore
|
||||
uses: sigstore/gh-action-sigstore-python@v3.4.0
|
||||
with:
|
||||
inputs: |
|
||||
./dist/*.tar.gz
|
||||
./dist/*.whl
|
||||
|
||||
- name: Extract release notes
|
||||
run: |
|
||||
VERSION="${GITHUB_REF_NAME#v}"
|
||||
MAJOR="${VERSION%%.*}"
|
||||
MAJOR_PADDED=$(printf "%02d" "$MAJOR")
|
||||
RELEASE_FILE="docs/releasenotes/version${MAJOR_PADDED}.md"
|
||||
|
||||
python3 << EOF
|
||||
import re
|
||||
|
||||
version = "${VERSION}"
|
||||
release_file = "${RELEASE_FILE}"
|
||||
|
||||
try:
|
||||
with open(release_file) as f:
|
||||
content = f.read()
|
||||
|
||||
# Find the section for this version
|
||||
# Match from "## vX.Y.Z" until the next "## v" or end of file
|
||||
pattern = rf"## v{re.escape(version)}\n(.*?)(?=\n## v|\Z)"
|
||||
match = re.search(pattern, content, re.DOTALL)
|
||||
notes = match.group(1).strip() if match else ""
|
||||
except FileNotFoundError:
|
||||
notes = ""
|
||||
|
||||
with open("release_notes.md", "w") as f:
|
||||
f.write(notes)
|
||||
EOF
|
||||
|
||||
- name: Publish release (convert draft to published)
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
# Update release: remove draft status, add release notes
|
||||
gh release edit "$GITHUB_REF_NAME" \
|
||||
--draft=false \
|
||||
--notes-file release_notes.md
|
||||
|
||||
# Upload signatures to the release
|
||||
gh release upload "$GITHUB_REF_NAME" dist/*.sigstore.json --clobber
|
||||
|
||||
docker_tag:
|
||||
name: Tag Docker images with release version
|
||||
needs: [publish]
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4
|
||||
with:
|
||||
username: jbarlow83
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v4
|
||||
|
||||
- name: Tag ocrmypdf (Ubuntu) image
|
||||
run: |
|
||||
docker buildx imagetools create \
|
||||
--tag "jbarlow83/ocrmypdf:$GITHUB_REF_NAME" \
|
||||
"jbarlow83/ocrmypdf:latest"
|
||||
|
||||
- name: Tag ocrmypdf-ubuntu image
|
||||
run: |
|
||||
docker buildx imagetools create \
|
||||
--tag "jbarlow83/ocrmypdf-ubuntu:$GITHUB_REF_NAME" \
|
||||
"jbarlow83/ocrmypdf-ubuntu:latest"
|
||||
|
||||
- name: Tag ocrmypdf-alpine image
|
||||
run: |
|
||||
docker buildx imagetools create \
|
||||
--tag "jbarlow83/ocrmypdf-alpine:$GITHUB_REF_NAME" \
|
||||
"jbarlow83/ocrmypdf-alpine:latest"
|
||||
+1
-1
@@ -44,8 +44,8 @@ docs/_build/
|
||||
docs/_static/
|
||||
docs/_templates/
|
||||
docs/Makefile
|
||||
src/ocrmypdf/_version.py
|
||||
|
||||
.idea/
|
||||
.aider*
|
||||
CLAUDE.md
|
||||
.claude/
|
||||
@@ -1,27 +0,0 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v4.4.0
|
||||
hooks:
|
||||
- id: check-case-conflict
|
||||
- id: check-merge-conflict
|
||||
- id: check-toml
|
||||
- id: check-yaml
|
||||
- id: debug-statements
|
||||
- repo: https://github.com/astral-sh/ruff-pre-commit
|
||||
rev: "v0.14.11"
|
||||
hooks:
|
||||
- id: ruff-check
|
||||
args: [--fix]
|
||||
- id: ruff-format
|
||||
- repo: https://github.com/pre-commit/mirrors-mypy
|
||||
rev: v1.2.0
|
||||
hooks:
|
||||
- id: mypy
|
||||
additional_dependencies:
|
||||
- types-toml
|
||||
- types-setuptools
|
||||
- types-requests
|
||||
- types-Pillow
|
||||
+10
-8
@@ -15,11 +15,13 @@ sphinx:
|
||||
build:
|
||||
os: ubuntu-22.04
|
||||
tools:
|
||||
python: "3.11"
|
||||
|
||||
python:
|
||||
install:
|
||||
- method: pip
|
||||
path: .
|
||||
extra_requirements:
|
||||
- docs
|
||||
python: "3.13"
|
||||
jobs:
|
||||
pre_create_environment:
|
||||
- asdf plugin add uv
|
||||
- asdf install uv latest
|
||||
- asdf global uv latest
|
||||
create_environment:
|
||||
- uv venv "${READTHEDOCS_VIRTUALENV_PATH}"
|
||||
install:
|
||||
- UV_PROJECT_ENVIRONMENT="${READTHEDOCS_VIRTUALENV_PATH}" uv sync --frozen --group docs
|
||||
@@ -79,17 +79,24 @@ Linux, Windows, macOS and FreeBSD are supported. Docker images are also availabl
|
||||
|
||||
For everyone else, [see our documentation](https://ocrmypdf.readthedocs.io/en/latest/installation.html) for installation steps.
|
||||
|
||||
## Docker - WEbui
|
||||
```
|
||||
docker compose -f misc/docker-compose.webui.yml up --build
|
||||
# http://localhost:8772/
|
||||
#
|
||||
```
|
||||
|
||||
## Languages
|
||||
|
||||
OCRmyPDF uses Tesseract for OCR, and relies on its language packs. For Linux users, you can often find packages that provide language packs:
|
||||
|
||||
```bash
|
||||
# Display a list of all Tesseract language packs
|
||||
apt-cache search tesseract-ocr
|
||||
|
||||
# Debian/Ubuntu users
|
||||
apt-cache search tesseract-ocr # Display a list of all Tesseract language packs
|
||||
apt-get install tesseract-ocr-chi-sim # Example: Install Chinese Simplified language pack
|
||||
|
||||
|
||||
# Arch Linux users
|
||||
pacman -S tesseract-data-eng tesseract-data-deu # Example: Install the English and German language packs
|
||||
|
||||
@@ -99,6 +106,12 @@ pkg_add tesseract-cym # Example: Install the Welsh language pack
|
||||
|
||||
# brew macOS users
|
||||
brew install tesseract-lang
|
||||
|
||||
# Fedora users
|
||||
dnf search tesseract-langpack # Display a list of all Tesseract language packs
|
||||
dnf install tesseract-langpack-ita # Example: Install the Italian language pack
|
||||
|
||||
|
||||
```
|
||||
|
||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested.
|
||||
|
||||
+90
-76
@@ -1,8 +1,3 @@
|
||||
# OCRmyPDF 中文说明
|
||||
|
||||
我将为您创建一个中文版的 README.md 文件。以下是基于原始英文版本的中文翻译:
|
||||
|
||||
```markdown:/Users/huapai/PycharmProjects/OCRmyPDF/README.md
|
||||
<!-- SPDX-FileCopyrightText: 2014 Julien Pfefferkorn -->
|
||||
<!-- SPDX-FileCopyrightText: 2015 James R. Barlow -->
|
||||
<!-- SPDX-License-Identifier: CC-BY-SA-4.0 -->
|
||||
@@ -16,98 +11,109 @@
|
||||
[docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD"
|
||||
[pyversions]: https://img.shields.io/pypi/pyversions/ocrmypdf "支持的 Python 版本"
|
||||
|
||||
OCRmyPDF 为扫描的 PDF 文件添加 OCR 文本层,使其可以被搜索或复制粘贴。
|
||||
OCRmyPDF 会为扫描版 PDF 文件添加 OCR 文本层,使其可以搜索或复制粘贴。
|
||||
|
||||
```bash
|
||||
ocrmypdf # 这是一个可脚本化的命令行程序
|
||||
-l eng+fra # 支持多种语言
|
||||
--rotate-pages # 可以修正旋转错误的页面
|
||||
--deskew # 可以校正倾斜的 PDF!
|
||||
--title "My PDF" # 可以更改输出元数据
|
||||
--jobs 4 # 默认使用多核心处理
|
||||
--output-type pdfa # 默认生成 PDF/A 格式
|
||||
ocrmypdf # 它是一个可脚本化的命令行程序
|
||||
-l eng+fra # 它支持多种语言
|
||||
--rotate-pages # 它可以修正旋转方向错误的页面
|
||||
--deskew # 它可以校正歪斜的 PDF!
|
||||
--title "My PDF" # 它可以更改输出元数据
|
||||
--jobs 4 # 它默认使用多个 CPU 核心
|
||||
--output-type pdfa # 它默认生成 PDF/A
|
||||
input_scanned.pdf # 接受 PDF 输入(或图像)
|
||||
output_searchable.pdf # 生成经过验证的 PDF 输出
|
||||
```
|
||||
|
||||
[查看发布说明了解最新变更的详情](https://ocrmypdf.readthedocs.io/en/latest/release_notes.html)。
|
||||
[查看发布说明,了解最新变更详情](https://ocrmypdf.readthedocs.io/en/latest/release_notes.html)。
|
||||
|
||||
## 主要特点
|
||||
## 主要功能
|
||||
|
||||
- 从普通 PDF 生成可搜索的 [PDF/A](https://en.wikipedia.org/?title=PDF/A) 文件
|
||||
- 准确地将 OCR 文本放置在图像下方,便于复制/粘贴
|
||||
- 将 OCR 文本准确放置在图像下方,便于复制/粘贴
|
||||
- 保持原始嵌入图像的精确分辨率
|
||||
- 在可能的情况下,以"无损"操作方式插入 OCR 信息,不破坏任何其他内容
|
||||
- 在可能时,以“无损”操作插入 OCR 信息,不干扰任何其他内容
|
||||
- 优化 PDF 图像,通常生成比输入文件更小的文件
|
||||
- 如果需要,在执行 OCR 前对图像进行校正和/或清理
|
||||
- 按需在执行 OCR 前校正和/或清理图像
|
||||
- 验证输入和输出文件
|
||||
- 在所有可用的 CPU 核心上分配工作
|
||||
- 在所有可用 CPU 核心间分配工作
|
||||
- 使用 [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) 引擎识别超过 [100 种语言](https://github.com/tesseract-ocr/tessdata)
|
||||
- 保护您的私人数据安全
|
||||
- 适当扩展以处理包含数千页的文件
|
||||
- 在数百万 PDF 上经过实战测试
|
||||
- 保护你的私有数据。
|
||||
- 可以妥善扩展,处理包含数千页的文件。
|
||||
- 已在数百万份 PDF 上经过实战检验。
|
||||
|
||||
<img src="misc/screencast/demo.svg" alt="终端会话中的 OCRmyPDF 演示">
|
||||
<img src="misc/screencast/demo.svg" alt="OCRmyPDF 在终端会话中的演示">
|
||||
|
||||
详情请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/)。
|
||||
|
||||
## 开发动机
|
||||
## 动机
|
||||
|
||||
我在网上搜索免费的命令行工具来对 PDF 文件进行 OCR:我找到了很多,但没有一个真正令人满意:
|
||||
我曾在网上寻找一款免费的命令行工具来对 PDF 文件执行 OCR:我找到了很多,但没有一个真正令人满意:
|
||||
|
||||
- 要么它们生成的 PDF 文件中文本位置错误(使复制/粘贴变得不可能)
|
||||
- 要么它们不处理重音和多语言字符
|
||||
- 要么它们改变了嵌入图像的分辨率
|
||||
- 要么它们生成了体积巨大的 PDF 文件
|
||||
- 要么它们在尝试 OCR 时崩溃
|
||||
- 要么它们不生成有效的 PDF 文件
|
||||
- 最重要的是,它们都不生成 PDF/A 文件(专为长期存储设计的格式)
|
||||
- 要么生成的 PDF 文件中文本位于图像下方的错误位置(导致无法复制/粘贴)
|
||||
- 要么无法处理重音字符和多语言字符
|
||||
- 要么会改变嵌入图像的分辨率
|
||||
- 要么生成的 PDF 文件大得离谱
|
||||
- 要么在尝试 OCR 时崩溃
|
||||
- 要么无法生成有效的 PDF 文件
|
||||
- 除此之外,它们都不能生成 PDF/A 文件(专为长期存储设计的格式)
|
||||
|
||||
...所以我决定开发自己的工具。
|
||||
……所以我决定开发自己的工具。
|
||||
|
||||
## 安装
|
||||
|
||||
支持 Linux、Windows、macOS 和 FreeBSD。Docker 镜像也可用,同时支持 x64 和 ARM。
|
||||
支持 Linux、Windows、macOS 和 FreeBSD。也提供 Docker 镜像,同时支持 x64 和 ARM。
|
||||
|
||||
| 操作系统 | 安装命令 |
|
||||
| --------------------------- | ----------------------------- |
|
||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||
| 操作系统 | 安装命令 |
|
||||
| ----------------------------- | ------------------------------ |
|
||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| OpenBSD | ``pkg_add ocrmypdf`` |
|
||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||
|
||||
对于其他用户,[请参阅我们的文档](https://ocrmypdf.readthedocs.io/en/latest/installation.html)了解安装步骤。
|
||||
其他用户请[参阅我们的文档](https://ocrmypdf.readthedocs.io/en/latest/installation.html)了解安装步骤。
|
||||
|
||||
## 语言
|
||||
|
||||
OCRmyPDF 使用 Tesseract 进行 OCR,并依赖其语言包。对于 Linux 用户,您通常可以找到提供语言包的软件包:
|
||||
OCRmyPDF 使用 Tesseract 执行 OCR,并依赖其语言包。对于 Linux 用户,通常可以找到提供语言包的软件包:
|
||||
|
||||
```bash
|
||||
# 显示所有 Tesseract 语言包的列表
|
||||
apt-cache search tesseract-ocr
|
||||
|
||||
# Debian/Ubuntu 用户
|
||||
apt-get install tesseract-ocr-chi-sim # 示例:安装中文简体语言包
|
||||
apt-cache search tesseract-ocr # 显示所有 Tesseract 语言包列表
|
||||
apt-get install tesseract-ocr-chi-sim # 示例:安装简体中文语言包
|
||||
|
||||
|
||||
# Arch Linux 用户
|
||||
pacman -S tesseract-data-eng tesseract-data-deu # 示例:安装英语和德语语言包
|
||||
|
||||
# OpenBSD 用户
|
||||
pkg_info -aQ tesseract # 显示所有 Tesseract 语言包列表
|
||||
pkg_add tesseract-cym # 示例:安装威尔士语语言包
|
||||
|
||||
# brew macOS 用户
|
||||
brew install tesseract-lang
|
||||
|
||||
# Fedora 用户
|
||||
dnf search tesseract-langpack # 显示所有 Tesseract 语言包列表
|
||||
dnf install tesseract-langpack-ita # 示例:安装意大利语语言包
|
||||
|
||||
|
||||
```
|
||||
|
||||
然后,您可以向 OCRmyPDF 传递 `-l LANG` 参数,提示它应该搜索哪些语言。可以请求多种语言。
|
||||
随后可以向 OCRmyPDF 传递 `-l LANG` 参数,提示它应搜索哪些语言。可以同时请求多种语言。
|
||||
|
||||
OCRmyPDF 支持 Tesseract 4.1.1+。它会自动使用在 `PATH` 环境变量中首先找到的版本。在 Windows 上,如果 `PATH` 不提供 Tesseract 二进制文件,我们会根据 Windows 注册表使用已安装的最高版本号。
|
||||
OCRmyPDF 支持 Tesseract 4.1.1+。它会自动使用 `PATH` 环境变量中首先找到的版本。在 Windows 上,如果 `PATH` 中没有 Tesseract 二进制文件,我们会根据 Windows 注册表使用已安装的最高版本号。
|
||||
|
||||
## 文档和支持
|
||||
|
||||
安装 OCRmyPDF 后,可以通过以下方式访问内置帮助,解释命令语法和选项:
|
||||
安装 OCRmyPDF 后,可以通过以下命令访问内置帮助,了解命令语法和选项:
|
||||
|
||||
```bash
|
||||
ocrmypdf --help
|
||||
@@ -115,13 +121,13 @@ ocrmypdf --help
|
||||
|
||||
我们的[文档托管在 Read the Docs 上](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
|
||||
请在我们的 [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) 页面上报告问题,并遵循问题模板以获得快速响应。
|
||||
请在我们的 [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) 页面报告问题,并遵循 issue 模板以便快速获得响应。
|
||||
|
||||
## 功能演示
|
||||
|
||||
```bash
|
||||
# 添加 OCR 层并转换为 PDF/A
|
||||
ocrmypdf input.pdf output.pdf
|
||||
# 添加 OCR 层并要求输出 PDF/A
|
||||
ocrmypdf --output-type pdfa input.pdf output.pdf
|
||||
|
||||
# 将图像转换为单页 PDF
|
||||
ocrmypdf input.jpg output.pdf
|
||||
@@ -129,45 +135,53 @@ ocrmypdf input.jpg output.pdf
|
||||
# 就地为文件添加 OCR(仅在成功时修改文件)
|
||||
ocrmypdf myfile.pdf myfile.pdf
|
||||
|
||||
# 使用非英语语言进行 OCR(查找您语言的 ISO 639-3 代码)
|
||||
# 使用非英语语言执行 OCR(请查找对应语言的 ISO 639-3 代码)
|
||||
ocrmypdf -l fra LeParisien.pdf LeParisien.pdf
|
||||
|
||||
# OCR 多语言文档
|
||||
ocrmypdf -l eng+fra Bilingual-English-French.pdf Bilingual-English-French.pdf
|
||||
|
||||
# 校正(矫正倾斜的页面)
|
||||
# 校正歪斜页面
|
||||
ocrmypdf --deskew input.pdf output.pdf
|
||||
```
|
||||
|
||||
更多功能,请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
更多功能请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
|
||||
## 要求
|
||||
|
||||
除了所需的 Python 版本外,OCRmyPDF 还需要外部程序安装 Ghostscript 和 Tesseract OCR。OCRmyPDF 是纯 Python 编写的,几乎可以在所有平台上运行:Linux、macOS、Windows 和 FreeBSD。
|
||||
除所需的 Python 版本外,OCRmyPDF 还需要安装 Ghostscript 和 Tesseract OCR 这两个外部程序。OCRmyPDF 是纯 Python 项目,几乎可以在所有平台上运行:Linux、macOS、Windows 和 FreeBSD。
|
||||
|
||||
## 媒体报道
|
||||
## 插件
|
||||
|
||||
- [使用 OCRmyPDF 实现无纸化](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a)
|
||||
- [将扫描文档转换为带有编辑的压缩可搜索 PDF](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c)
|
||||
- [c't 1-2014, 第 59 页](https://heise.de/-2279695):在德国领先的 IT 杂志 c't 中详细介绍 OCRmyPDF v1.0
|
||||
- [heise Open Source, 09/2014: 使用 OCRmyPDF 进行文本识别](https://heise.de/-2356670)
|
||||
- [heise 使用 OCRmyPDF 创建可搜索的 PDF 文档](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [优秀实用工具:OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser 使用 OCRmyPDF 和 Scanbd 自动化文本识别](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [Y Combinator 讨论](https://news.ycombinator.com/item?id=32028752)
|
||||
OCRmyPDF 提供插件接口,允许扩展或替换其能力。以下是我们知道的一些插件:
|
||||
|
||||
## 商业咨询
|
||||
- [OCRmyPDF-AppleOCR](https://github.com/mkyt/ocrmypdf-AppleOCR):用 Apple Vision Framework 替换标准 Tesseract OCR 引擎。需要 macOS。
|
||||
- [OCRmyPDF-EasyOCR](https://github.com/ocrmypdf/OCRmyPDF-EasyOCR):用 EasyOCR 替换标准 Tesseract OCR 引擎;EasyOCR 是基于 PyTorch 的较新 OCR 引擎。强烈建议使用 GPU。
|
||||
- [OCRmyPDF-PaddleOCR](https://github.com/clefru/ocrmypdf-paddleocr):用 PaddleOCR 替换标准 Tesseract OCR 引擎;PaddleOCR 是功能强大的 GPU 加速 OCR 引擎。
|
||||
|
||||
如果没有公司和用户选择为功能开发和咨询提供支持,OCRmyPDF 就不会成为今天的软件。我们很乐意讨论所有咨询,无论是扩展现有功能集,还是将 OCRmyPDF 集成到更大的系统中。
|
||||
[paperless-ngx](https://docs.paperless-ngx.com/) 将 OCRmyPDF 集成到可搜索的文档管理系统中。
|
||||
|
||||
## 新闻与媒体
|
||||
|
||||
- [Going paperless with OCRmyPDF](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a)
|
||||
- [Converting a scanned document into a compressed searchable PDF with redactions](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c)
|
||||
- [c't 1-2014,第 59 页](https://heise.de/-2279695):德国领先 IT 杂志 c't 对 OCRmyPDF v1.0 的详细介绍
|
||||
- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670)
|
||||
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [Y Combinator discussion](https://news.ycombinator.com/item?id=32028752)
|
||||
|
||||
## 商务咨询
|
||||
|
||||
如果没有公司和用户选择支持功能开发与咨询服务,OCRmyPDF 不会成为今天的软件。无论是扩展现有功能集,还是将 OCRmyPDF 集成到更大的系统中,我们都很乐意讨论各类咨询需求。
|
||||
|
||||
## 许可证
|
||||
|
||||
OCRmyPDF 软件根据 Mozilla 公共许可证 2.0 (MPL-2.0) 授权。此许可证允许将 OCRmyPDF 与其他代码集成,包括商业和闭源代码,但要求您发布对 OCRmyPDF 所做的源代码级修改。
|
||||
OCRmyPDF 软件采用 Mozilla Public License 2.0 (MPL-2.0) 授权。该许可证允许将 OCRmyPDF 与其他代码集成,包括商业代码和闭源代码,但要求你发布对 OCRmyPDF 所做的源代码级修改。
|
||||
|
||||
OCRmyPDF 的某些组件有其他许可证,如标准 SPDX 许可证标识符或 DEP5 版权和许可信息文件所示。一般来说,非核心代码根据 MIT 许可,文档和测试文件根据 Creative Commons ShareAlike 4.0 (CC-BY-SA 4.0) 许可。
|
||||
OCRmyPDF 的某些组件采用其他许可证,具体由标准 SPDX 许可证标识符或 DEP5 版权与许可信息文件标明。一般来说,非核心代码采用 MIT 许可证,文档和测试文件采用 Creative Commons ShareAlike 4.0 (CC-BY-SA 4.0) 许可证。
|
||||
|
||||
## 免责声明
|
||||
|
||||
本软件按"原样"分发,不提供任何明示或暗示的保证或条件。
|
||||
|
||||
这份中文版 README.md 保留了原始文档的所有重要信息,包括功能介绍、安装说明、语言支持、使用示例等内容,同时保持了原始格式和结构。
|
||||
本软件按“原样”分发,不提供任何明示或暗示的保证或条件。
|
||||
|
||||
@@ -0,0 +1,408 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-FileCopyrightText: 2017-2019 Joe Rickerby and contributors
|
||||
# SPDX-License-Identifier: BSD-2-Clause
|
||||
|
||||
"""Bump the version number in all the right places."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import urllib.parse
|
||||
from pathlib import Path
|
||||
|
||||
import cyclopts
|
||||
from packaging.version import InvalidVersion, Version
|
||||
|
||||
try:
|
||||
from github import Auth, Github, GithubException
|
||||
except ImportError:
|
||||
Auth = None # type: ignore
|
||||
Github = None # type: ignore
|
||||
GithubException = Exception # type: ignore
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
config = [
|
||||
# file path, version find/replace format
|
||||
("src/ocrmypdf/_version.py", '__version__ = "{}"'),
|
||||
("pyproject.toml", 'version = "{}"'),
|
||||
]
|
||||
|
||||
RED = "\u001b[31m"
|
||||
GREEN = "\u001b[32m"
|
||||
YELLOW = "\u001b[33m"
|
||||
OFF = "\u001b[0m"
|
||||
|
||||
REPO_NAME = "ocrmypdf/OCRmyPDF"
|
||||
|
||||
|
||||
def validate_release_notes(new_version: str) -> bool:
|
||||
"""Check that the version appears in the release notes.
|
||||
|
||||
Returns True if the version is found, False otherwise.
|
||||
"""
|
||||
version_obj = Version(new_version)
|
||||
major = version_obj.major
|
||||
release_notes_path = Path(f"docs/releasenotes/version{major:02d}.md")
|
||||
|
||||
if not release_notes_path.exists():
|
||||
print(f"{RED}error:{OFF} Release notes file not found: {release_notes_path}")
|
||||
return False
|
||||
|
||||
content = release_notes_path.read_text(encoding="utf8")
|
||||
version_header = f"## v{new_version}"
|
||||
|
||||
if version_header not in content:
|
||||
print(
|
||||
f"{RED}error:{OFF} Version v{new_version} not found in {release_notes_path}"
|
||||
)
|
||||
print(f" Expected to find: {version_header}")
|
||||
return False
|
||||
|
||||
print(f"{GREEN}Found v{new_version} in {release_notes_path}{OFF}")
|
||||
return True
|
||||
|
||||
|
||||
def get_github_client():
|
||||
"""Get an authenticated GitHub client."""
|
||||
if Github is None or Auth is None:
|
||||
print(f"{RED}error:{OFF} PyGithub is not installed")
|
||||
print(" Install with: pip install PyGithub")
|
||||
return None
|
||||
|
||||
# Try GITHUB_TOKEN env var first
|
||||
token = os.environ.get("GITHUB_TOKEN")
|
||||
|
||||
# Fall back to gh CLI
|
||||
if not token:
|
||||
try:
|
||||
result = subprocess.run(
|
||||
["gh", "auth", "token"],
|
||||
capture_output=True,
|
||||
encoding="utf8",
|
||||
check=True,
|
||||
)
|
||||
token = result.stdout.strip()
|
||||
except (FileNotFoundError, subprocess.CalledProcessError):
|
||||
print(f"{RED}error:{OFF} No GitHub authentication found")
|
||||
print(" Set GITHUB_TOKEN env var or run: gh auth login")
|
||||
return None
|
||||
|
||||
try:
|
||||
return Github(auth=Auth.Token(token))
|
||||
except GithubException as e:
|
||||
print(f"{RED}error:{OFF} Failed to authenticate with GitHub: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def wait_for_ci_completion(commit_sha: str, timeout_minutes: int = 30) -> bool:
|
||||
"""Wait for CI to complete on the given commit.
|
||||
|
||||
Returns True if CI passed, False otherwise.
|
||||
"""
|
||||
gh = get_github_client()
|
||||
if gh is None:
|
||||
return False
|
||||
|
||||
try:
|
||||
repo = gh.get_repo(REPO_NAME)
|
||||
except GithubException as e:
|
||||
print(f"{RED}error:{OFF} Failed to access repository: {e}")
|
||||
return False
|
||||
|
||||
workflow_name = "Test and deploy"
|
||||
start_time = time.time()
|
||||
timeout_seconds = timeout_minutes * 60
|
||||
poll_interval = 30 # seconds
|
||||
|
||||
print(f"Waiting for CI workflow '{workflow_name}' on commit {commit_sha[:8]}...")
|
||||
|
||||
# First, wait for the workflow run to appear
|
||||
run = None
|
||||
while time.time() - start_time < timeout_seconds:
|
||||
try:
|
||||
runs = repo.get_workflow_runs(head_sha=commit_sha)
|
||||
for r in runs:
|
||||
if r.name == workflow_name:
|
||||
run = r
|
||||
break
|
||||
if run:
|
||||
break
|
||||
except GithubException as e:
|
||||
print(f"{YELLOW}Warning:{OFF} Error checking workflow runs: {e}")
|
||||
|
||||
elapsed = int(time.time() - start_time)
|
||||
print(f" Waiting for workflow to start... ({elapsed}s)")
|
||||
time.sleep(poll_interval)
|
||||
|
||||
if not run:
|
||||
print(
|
||||
f"{RED}error:{OFF} Workflow run not found within {timeout_minutes} minutes"
|
||||
)
|
||||
return False
|
||||
|
||||
print(f" Found workflow run #{run.run_number} (ID: {run.id})")
|
||||
|
||||
# Now wait for the workflow to complete
|
||||
while time.time() - start_time < timeout_seconds:
|
||||
try:
|
||||
run = repo.get_workflow_run(run.id) # Refresh the run
|
||||
except GithubException as e:
|
||||
print(f"{YELLOW}Warning:{OFF} Error refreshing workflow run: {e}")
|
||||
time.sleep(poll_interval)
|
||||
continue
|
||||
|
||||
status = run.status
|
||||
conclusion = run.conclusion
|
||||
|
||||
elapsed = int(time.time() - start_time)
|
||||
if status == "completed":
|
||||
if conclusion == "success":
|
||||
print(f"{GREEN}CI passed!{OFF} (took {elapsed}s)")
|
||||
return True
|
||||
else:
|
||||
print(f"{RED}CI failed!{OFF} Conclusion: {conclusion}")
|
||||
print(f" View details: {run.html_url}")
|
||||
return False
|
||||
else:
|
||||
print(f" Status: {status} ({elapsed}s elapsed)")
|
||||
time.sleep(poll_interval)
|
||||
|
||||
print(f"{RED}error:{OFF} CI did not complete within {timeout_minutes} minutes")
|
||||
return False
|
||||
|
||||
|
||||
def push_and_wait_for_ci(branch: str) -> bool:
|
||||
"""Push to remote and wait for CI tests to pass."""
|
||||
print("Pushing to GitHub...")
|
||||
|
||||
push_result = subprocess.run(
|
||||
["git", "push", "origin", branch],
|
||||
capture_output=True,
|
||||
encoding="utf8",
|
||||
)
|
||||
|
||||
if push_result.returncode != 0:
|
||||
print(f"{RED}error:{OFF} Failed to push: {push_result.stderr}")
|
||||
return False
|
||||
|
||||
# Get the commit SHA we just pushed
|
||||
sha_result = subprocess.run(
|
||||
["git", "rev-parse", "HEAD"],
|
||||
capture_output=True,
|
||||
encoding="utf8",
|
||||
check=True,
|
||||
)
|
||||
commit_sha = sha_result.stdout.strip()
|
||||
|
||||
print(f"Pushed commit {commit_sha[:8]}")
|
||||
|
||||
return wait_for_ci_completion(commit_sha)
|
||||
|
||||
|
||||
def push_tag(tag: str) -> bool:
|
||||
"""Push the tag to trigger release workflow."""
|
||||
print(f"Pushing tag {tag} to trigger release...")
|
||||
|
||||
result = subprocess.run(
|
||||
["git", "push", "origin", tag],
|
||||
capture_output=True,
|
||||
encoding="utf8",
|
||||
)
|
||||
|
||||
if result.returncode != 0:
|
||||
print(f"{RED}error:{OFF} Failed to push tag: {result.stderr}")
|
||||
return False
|
||||
|
||||
print(f"{GREEN}Tag {tag} pushed successfully!{OFF}")
|
||||
return True
|
||||
|
||||
|
||||
def bump_version() -> None:
|
||||
"""Bump the version number in all the right places."""
|
||||
current_version = ocrmypdf.__version__ # type: ignore
|
||||
try:
|
||||
commit_date_str = subprocess.run(
|
||||
[
|
||||
"git",
|
||||
"show",
|
||||
"--no-patch",
|
||||
"--pretty=format:%ci",
|
||||
f"v{current_version}^{{commit}}",
|
||||
],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
encoding="utf8",
|
||||
).stdout
|
||||
cd_date, cd_time, cd_tz = commit_date_str.split(" ")
|
||||
|
||||
url_opts = urllib.parse.urlencode(
|
||||
{"q": f"is:pr merged:>{cd_date}T{cd_time}{cd_tz}"}
|
||||
)
|
||||
url = f"https://github.com/{REPO_NAME}/pulls?{url_opts}"
|
||||
|
||||
print(f"PRs merged since last release:\n {url}")
|
||||
print()
|
||||
except subprocess.CalledProcessError as e:
|
||||
print(e)
|
||||
print("Failed to get previous version tag information.")
|
||||
print("Is the virtual environment active?")
|
||||
sys.exit(1)
|
||||
|
||||
git_changes_result = subprocess.run(["git diff-index --quiet HEAD --"], shell=True)
|
||||
repo_has_uncommitted_changes = git_changes_result.returncode != 0
|
||||
|
||||
if repo_has_uncommitted_changes:
|
||||
print("error: Uncommitted changes detected.")
|
||||
sys.exit(1)
|
||||
|
||||
# fmt: off
|
||||
print( 'Current version:', current_version)
|
||||
new_version = input(' New version: ').strip()
|
||||
# fmt: on
|
||||
|
||||
try:
|
||||
Version(new_version)
|
||||
except InvalidVersion:
|
||||
print("error: This version doesn't conform to PEP440")
|
||||
print(" https://www.python.org/dev/peps/pep-0440/")
|
||||
sys.exit(1)
|
||||
|
||||
# Validate release notes contain this version
|
||||
if not validate_release_notes(new_version):
|
||||
print()
|
||||
print("Please add release notes for this version before proceeding.")
|
||||
print(f"Edit: docs/releasenotes/version{Version(new_version).major:02d}.md")
|
||||
sys.exit(1)
|
||||
|
||||
actions = []
|
||||
|
||||
for path_pattern, version_pattern in config:
|
||||
paths = list(Path().glob(path_pattern))
|
||||
|
||||
if not paths:
|
||||
print(f"error: Pattern {path_pattern} didn't match any files")
|
||||
sys.exit(1)
|
||||
|
||||
find_pattern = version_pattern.format(current_version)
|
||||
replace_pattern = version_pattern.format(new_version)
|
||||
found_at_least_one_file_needing_update = False
|
||||
|
||||
for path in paths:
|
||||
contents = path.read_text(encoding="utf8")
|
||||
if find_pattern in contents:
|
||||
found_at_least_one_file_needing_update = True
|
||||
actions.append(
|
||||
(
|
||||
path,
|
||||
find_pattern,
|
||||
replace_pattern,
|
||||
)
|
||||
)
|
||||
|
||||
if not found_at_least_one_file_needing_update:
|
||||
print(
|
||||
f'''error: Didn't find any occurrences of "{find_pattern}" '''
|
||||
f'''in "{path_pattern}"'''
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
print()
|
||||
print("Here's the plan:")
|
||||
print()
|
||||
|
||||
for action in actions:
|
||||
path, find, replace = action
|
||||
print(f"{path} {RED}{find}{OFF} → {GREEN}{replace}{OFF}")
|
||||
|
||||
print(f"Then commit, and tag as v{new_version}")
|
||||
|
||||
answer = input("Proceed? [y/N] ").strip()
|
||||
|
||||
if answer != "y":
|
||||
print("Aborted")
|
||||
sys.exit(1)
|
||||
|
||||
for path, find, replace in actions:
|
||||
contents = path.read_text(encoding="utf8")
|
||||
contents = contents.replace(find, replace)
|
||||
path.write_text(contents, encoding="utf8")
|
||||
|
||||
# Format only after every file (including pyproject.toml) reflects the new
|
||||
# version. Running `uv run` while pyproject.toml still had the old version
|
||||
# would leave its post-bump environment/lockfile resync to happen for the
|
||||
# first time during the commit's pre-commit hooks instead of here, which
|
||||
# then aborts the commit with a spurious "files were modified by this
|
||||
# hook" error.
|
||||
for path, _find, _replace in actions:
|
||||
if path.suffix == ".py":
|
||||
subprocess.run(["uv", "run", "ruff", "format", str(path)], check=True)
|
||||
|
||||
print("Files updated.")
|
||||
print()
|
||||
|
||||
while input('Type "done" to continue: ').strip().lower() != "done":
|
||||
pass
|
||||
|
||||
subprocess.run(
|
||||
[
|
||||
"git",
|
||||
"commit",
|
||||
"--all",
|
||||
f"--message=Bump version: v{new_version}",
|
||||
],
|
||||
check=True,
|
||||
)
|
||||
|
||||
subprocess.run(
|
||||
[
|
||||
"git",
|
||||
"tag",
|
||||
"--annotate",
|
||||
f"--message=v{new_version}",
|
||||
f"v{new_version}",
|
||||
],
|
||||
check=True,
|
||||
)
|
||||
|
||||
print("Commit and tag created locally.")
|
||||
print()
|
||||
|
||||
# Get current branch
|
||||
branch_result = subprocess.run(
|
||||
["git", "rev-parse", "--abbrev-ref", "HEAD"],
|
||||
capture_output=True,
|
||||
encoding="utf8",
|
||||
check=True,
|
||||
)
|
||||
branch = branch_result.stdout.strip()
|
||||
|
||||
# Push commit and wait for CI
|
||||
if not push_and_wait_for_ci(branch):
|
||||
print()
|
||||
print(f"{RED}CI failed. The tag was NOT pushed.{OFF}")
|
||||
print("Fix the issues, then manually push the tag:")
|
||||
print(f" git push origin v{new_version}")
|
||||
sys.exit(1)
|
||||
|
||||
# Push tag to trigger release
|
||||
if not push_tag(f"v{new_version}"):
|
||||
print(f"{RED}Failed to push tag.{OFF} Push manually:")
|
||||
print(f" git push origin v{new_version}")
|
||||
sys.exit(1)
|
||||
|
||||
print()
|
||||
print(f"{GREEN}Done! Release workflow has been triggered.{OFF}")
|
||||
print()
|
||||
|
||||
release_url = f"https://github.com/{REPO_NAME}/releases/tag/v{new_version}"
|
||||
print("Monitor the release at:")
|
||||
print(f" {release_url}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
os.chdir(Path(__file__).parent.parent.resolve())
|
||||
cyclopts.run(bump_version)
|
||||
+131
-3
@@ -121,6 +121,30 @@ representation. This is useful for redoing OCR, for fixing OCR text
|
||||
with a damaged character map (text is selectable but not searchable),
|
||||
and destroying redacted information.
|
||||
|
||||
### Tagged PDFs and structural markup
|
||||
|
||||
Some PDFs carry a logical structure tree (`/StructTreeRoot`), the markup that
|
||||
makes a "Tagged PDF" — typically the result of layout analysis or a born-digital
|
||||
export. By default OCRmyPDF treats this as a signal that the document may not need
|
||||
OCR and exits, in the same way it stops on PDFs that already contain text. Use
|
||||
`--tagged-pdf-mode ignore`, or one of `--mode skip`/`redo`/`force`, to process
|
||||
such a file anyway.
|
||||
|
||||
OCRmyPDF cannot rebuild a structure tree to match newly recognized text. When
|
||||
`--force-ocr` rasterizes pages, or `--redo-ocr` strips and rewrites the text layer,
|
||||
the structure tree no longer corresponds to the page content, so it is discarded.
|
||||
`--mode skip` leaves text pages untouched, so their structural markup is preserved.
|
||||
|
||||
:::{note}
|
||||
Preservation under `--mode skip` only holds when the output is not converted to
|
||||
PDF/A. PDF/A conversion is performed by Ghostscript, and Ghostscript 10.x discards
|
||||
the structure tree during conversion (Ghostscript 9.x preserved it). Because the
|
||||
default `--output-type auto` may fall back to Ghostscript, use
|
||||
`--output-type pdf` if you need to guarantee that a Tagged PDF's structural markup
|
||||
survives. For best results, install veraPDF so that speculative PDF/A
|
||||
conversion can sidestep this issue entirely in most real cases.
|
||||
:::
|
||||
|
||||
### Time and image size limits
|
||||
|
||||
By default, OCRmyPDF permits tesseract to run for three minutes (180
|
||||
@@ -187,6 +211,13 @@ include:
|
||||
Overrides the path to Tesseract's data files. This can allow
|
||||
simultaneous installation of the "best" and "fast" training data
|
||||
sets. OCRmyPDF does not manage this environment variable.
|
||||
|
||||
If you point ``TESSDATA_PREFIX`` at a hand-assembled ``tessdata``
|
||||
folder (for example, individual ``.traineddata`` files downloaded
|
||||
from tessdata_best), make sure it also contains the ``configs/``
|
||||
subdirectory with the ``hocr`` and ``txt`` files. OCRmyPDF requires
|
||||
these; without them Tesseract produces no output. See
|
||||
:ref:`Tesseract cannot open its config file <tesseract-config-missing>`.
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
@@ -419,6 +450,70 @@ curves. In this case, you may want to use a different color conversion
|
||||
strategy. The `--color-conversion-strategy` option allows you to select a
|
||||
different strategy, such as `RGB`.
|
||||
|
||||
## Advanced Ghostscript tuning
|
||||
|
||||
:::{versionadded} 17.5.0
|
||||
:::
|
||||
|
||||
OCRmyPDF intentionally hides most Ghostscript controls because Ghostscript
|
||||
is a legacy code path. The preferred PDF/A pipeline in v17+ uses pypdfium2
|
||||
as the rasterizer and verapdf to validate speculative PDF/A output, with
|
||||
Ghostscript reserved as a fallback for PDFs that cannot be made compliant
|
||||
without it. OCRmyPDF's separate optimizer (controlled by `--optimize`,
|
||||
`--jpeg-quality`, `--png-quality`, etc.) is the supported way to shrink
|
||||
output PDFs: it gives consistent results across input files, and isolates
|
||||
Ghostscript so it can focus on producing a PDF/A with as few image
|
||||
transformations as possible.
|
||||
|
||||
The two options below are exposed for advanced users who want to tune
|
||||
Ghostscript's intermediate PDF/A output directly. Most users will get
|
||||
more predictable results from the optimizer.
|
||||
|
||||
### `--ghostscript-jpeg-quality Q`
|
||||
|
||||
Sets Ghostscript's `-dJPEGQ` switch for images that Ghostscript chooses
|
||||
to recompress to JPEG while building a PDF/A. `Q=0` requests maximum
|
||||
compression and `Q=100` requests best quality; if the flag is omitted,
|
||||
OCRmyPDF passes `95` (the historical default). This only affects images
|
||||
Ghostscript transcodes — existing JPEGs pass through unchanged on modern
|
||||
Ghostscript releases. For end-to-end JPEG quality tuning, prefer
|
||||
`--jpeg-quality`, which is implemented by the OCRmyPDF optimizer and is
|
||||
applied independently of whatever Ghostscript decides to do.
|
||||
|
||||
Note: setting both `--ghostscript-jpeg-quality` and `--jpeg-quality` can
|
||||
result in double JPEG recompression, since the optimizer may re-encode
|
||||
images that Ghostscript already recompressed. This can degrade quality
|
||||
in subtle ways.
|
||||
|
||||
### `--ghostscript-jpeg-maxdpi DPI`
|
||||
|
||||
Enables Ghostscript's image downsampling and caps color, grayscale, and
|
||||
monochrome image resolution to `DPI`. The downsample threshold is set to
|
||||
`1.0`, so any image whose effective DPI exceeds the cap will be
|
||||
downsampled.
|
||||
|
||||
Reducing JPEG quality is almost always a better trade than downsampling
|
||||
at the same compression budget: a 400 DPI JPEG at modest quality usually
|
||||
looks much better than a 200 DPI JPEG, because the JPEG codec can spend
|
||||
bits where they count. Downsampling is also dangerous for PDFs that
|
||||
combine a low-resolution color image with a high-resolution monochrome
|
||||
mask — capping the mask resolution can produce visible quality loss.
|
||||
For these reasons, prefer `--jpeg-quality` over `--ghostscript-jpeg-maxdpi`
|
||||
unless you specifically want to force a hard DPI cap.
|
||||
|
||||
Example:
|
||||
|
||||
```bash
|
||||
ocrmypdf --output-type pdfa \
|
||||
--ghostscript-jpeg-quality 80 \
|
||||
--ghostscript-jpeg-maxdpi 150 \
|
||||
in.pdf out.pdf
|
||||
```
|
||||
|
||||
These options only take effect when Ghostscript is invoked for PDF/A
|
||||
conversion (`--output-type pdfa`, `pdfa-1`, `pdfa-2`, or `pdfa-3`, or
|
||||
when `--output-type auto` falls back to Ghostscript).
|
||||
|
||||
## PDF/A output modes
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
@@ -438,6 +533,36 @@ OCRmyPDF can produce PDF/A compliant output for long-term archival. The
|
||||
| `pdf` | Standard PDF, no PDF/A conversion |
|
||||
| `none` | No output file (useful with `--sidecar`) |
|
||||
|
||||
### Non-embedded fonts and PDF/A
|
||||
|
||||
:::{versionadded} 17.8.0
|
||||
OCRmyPDF now refuses to corrupt non-embedded CID text layers during PDF/A
|
||||
conversion.
|
||||
:::
|
||||
|
||||
PDF/A requires every font to be embedded. If your input already has a text
|
||||
layer that uses *non-embedded* CID fonts — most commonly a CJK
|
||||
(Chinese-Japanese-Korean) OCR layer
|
||||
produced by Adobe Acrobat, which relies on the reader's system fonts —
|
||||
Ghostscript would have to substitute and re-embed a replacement font to make
|
||||
the file PDF/A. For CID-keyed (CJK) fonts this routinely corrupts the
|
||||
character-to-Unicode mapping, so the text silently becomes garbage or stops
|
||||
being searchable even though the page still *looks* correct.
|
||||
|
||||
Rather than emit corrupted output, OCRmyPDF detects this situation and:
|
||||
|
||||
- with `--output-type auto` (the default), produces a regular PDF instead of
|
||||
PDF/A, preserving the existing text layer exactly;
|
||||
- with an explicit `--output-type pdfa` (or `pdfa-1`/`pdfa-2`/`pdfa-3`), stops
|
||||
with an error.
|
||||
|
||||
This is a Ghostscript limitation that OCRmyPDF cannot repair, because a
|
||||
non-embedded font cannot be made PDF/A-compliant without re-embedding it. To
|
||||
keep the existing text layer, use `--output-type pdf`. To produce PDF/A anyway,
|
||||
re-run OCR with `--force-ocr`, which discards the original text layer and
|
||||
rebuilds it with embedded fonts. Text layers whose fonts are *already embedded*
|
||||
are converted to PDF/A normally.
|
||||
|
||||
### Speculative PDF/A conversion
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
@@ -451,9 +576,12 @@ fast "speculative" PDF/A conversion that avoids Ghostscript when possible:
|
||||
3. If validation passes, Ghostscript is skipped entirely
|
||||
4. If validation fails or verapdf is unavailable, falls back to Ghostscript
|
||||
|
||||
This approach is faster and avoids some Ghostscript limitations (such as
|
||||
image transcoding), but only works for PDFs that are already "mostly"
|
||||
PDF/A compliant.
|
||||
This fast path avoids some Ghostscript limitations (such as image
|
||||
transcoding) and is used whenever it can produce valid PDF/A. When it
|
||||
cannot — for example when veraPDF is not installed, or the input needs real
|
||||
conversion — `auto` falls back to Ghostscript so that it still produces
|
||||
PDF/A by default, matching OCRmyPDF 16 and earlier. If even Ghostscript
|
||||
cannot safely produce PDF/A, `auto` outputs a regular PDF instead of failing.
|
||||
|
||||
### PDF/A conversion flow
|
||||
|
||||
|
||||
@@ -95,10 +95,12 @@ from multiprocessing import Process
|
||||
import ocrmypdf
|
||||
from ocrmypdf import OcrOptions
|
||||
|
||||
|
||||
def ocrmypdf_process():
|
||||
options = OcrOptions(input_file='input.pdf', output_file='output.pdf')
|
||||
ocrmypdf.ocr(options)
|
||||
|
||||
|
||||
def call_ocrmypdf_from_my_app():
|
||||
p = Process(target=ocrmypdf_process)
|
||||
p.start()
|
||||
|
||||
@@ -33,9 +33,6 @@ should be mainly of interest to plugin developers.
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf.helpers
|
||||
:members:
|
||||
:noindex: deprecated
|
||||
|
||||
.. autodecorator:: deprecated
|
||||
```
|
||||
|
||||
## ocrmypdf.hocrtransform
|
||||
|
||||
+9
-1
@@ -174,7 +174,15 @@ docker run \
|
||||
--env PYTHONUNBUFFERED=1 \
|
||||
--interactive --tty --entrypoint python3 \
|
||||
jbarlow83/ocrmypdf \
|
||||
watcher.py
|
||||
/app/watcher.py
|
||||
:::
|
||||
|
||||
:::{note}
|
||||
The image runs as the non-root `app` user (uid 1000) by default, so it
|
||||
may not be able to write to the `/output` and `/processed` volumes unless
|
||||
you add a `--user` argument. The correct value depends on whether you use
|
||||
rootful Docker, rootless Docker, or Podman -- see
|
||||
{ref}`Bind-mounted volumes <docker-volumes>` for details.
|
||||
:::
|
||||
|
||||
This service will watch for a file that matches `/input/\*.pdf`, convert
|
||||
|
||||
@@ -42,6 +42,7 @@ extensions = [
|
||||
'sphinx.ext.napoleon',
|
||||
'sphinx.ext.imgconverter', # PDF docs needs this for SVG to PNG conversion
|
||||
'sphinx_issues',
|
||||
'sphinx_reredirects',
|
||||
'sphinxcontrib.mermaid',
|
||||
]
|
||||
|
||||
@@ -51,6 +52,9 @@ myst_enable_extensions = ['colon_fence', 'attrs_block', 'attrs_inline', 'substit
|
||||
intersphinx_mapping = {'python': ('https://docs.python.org/3', None)}
|
||||
napoleon_use_rtype = False
|
||||
issues_github_path = "ocrmypdf/OCRmyPDF"
|
||||
redirects = {
|
||||
"release_notes": "releasenotes/index.html",
|
||||
}
|
||||
|
||||
# Add any paths that contain templates here, relative to this directory.
|
||||
templates_path = ['_templates']
|
||||
@@ -174,6 +178,20 @@ html_theme = 'sphinx_rtd_theme'
|
||||
#
|
||||
html_theme_options = {}
|
||||
|
||||
# ReadTheDocs used to inject the "Edit on GitHub" context automatically, but
|
||||
# dropped it when it switched to Addons, so set it explicitly here. This makes
|
||||
# sphinx_rtd_theme add an "Edit on GitHub" link to each page that points at the
|
||||
# corresponding source file in the repository, replacing the static
|
||||
# "View page source" (_sources/*.txt) link. See
|
||||
# https://github.com/ocrmypdf/OCRmyPDF/issues/1490
|
||||
html_context = {
|
||||
'display_github': True,
|
||||
'github_user': 'ocrmypdf',
|
||||
'github_repo': 'OCRmyPDF',
|
||||
'github_version': 'main',
|
||||
'conf_py_path': '/docs/',
|
||||
}
|
||||
|
||||
# Add any paths that contain custom themes here, relative to this directory.
|
||||
# html_theme_path = []
|
||||
|
||||
|
||||
+47
-12
@@ -31,6 +31,16 @@ ocrmypdf --output-type pdf input.pdf output.pdf
|
||||
ocrmypdf --output-type pdfa --pdfa-image-compression jpeg input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Reduce JPEG quality with the optimizer
|
||||
|
||||
This is the recommended way to shrink JPEG content in the output. The
|
||||
optimizer applies regardless of `--output-type`, so it works on both
|
||||
plain PDFs and Ghostscript-produced PDF/A files.
|
||||
|
||||
```bash
|
||||
ocrmypdf --optimize 2 --jpeg-quality 60 input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Modify a file in place
|
||||
|
||||
The file will only be overwritten if OCRmyPDF is successful.
|
||||
@@ -239,19 +249,34 @@ case. Use `--tesseract-non-ocr-timeout` to control the timeout for
|
||||
non-OCR operations, if needed.
|
||||
:::
|
||||
|
||||
### Remove all text or OCR from my PDF
|
||||
### Remove the OCR text layer from my PDF
|
||||
|
||||
This is getting ridiculous, but OCRmyPDF can complete strip all textual
|
||||
information from a PDF and reconstruct it as a \"bag of images\" PDF.
|
||||
To remove the invisible OCR text layer while keeping the original pages
|
||||
exactly as they are -- no rasterizing, no change to images or visible
|
||||
content, and a smaller output file -- use `--mode strip`:
|
||||
|
||||
```bash
|
||||
ocrmypdf --mode strip input.pdf output.pdf
|
||||
```
|
||||
|
||||
Why would you want to do this? Perhaps you have a PDF where OCR failed to
|
||||
produce useful results and you simply want to get rid of it.
|
||||
|
||||
`--mode strip` removes only text drawn as *invisible* (PDF text render
|
||||
mode 3), which is how OCRmyPDF and most OCR tools add a searchable layer
|
||||
over a scanned page. Some OCR products -- and OCRmyPDF v2.2 and earlier --
|
||||
instead draw *visible* text and paint an opaque image on top of it. That
|
||||
text is part of the visible page, so `--mode strip` cannot remove it
|
||||
without altering the page's appearance.
|
||||
|
||||
To strip *all* text, including such visible text, rasterize the whole page
|
||||
into a \"bag of images\" PDF instead (this rebuilds every page as an image,
|
||||
so the file usually grows and vector content is lost):
|
||||
|
||||
```bash
|
||||
ocrmypdf --ocr-engine none --force-ocr input.pdf output.pdf
|
||||
```
|
||||
|
||||
Why would you want to do this? Perhaps you have a PDF where OCR fails to
|
||||
produce useful results, and just want to get rid of all OCR information.
|
||||
This command also removes OCR generated by third party tools.
|
||||
|
||||
### Optimize images without performing OCR
|
||||
|
||||
You can also optimize all images without performing any OCR:
|
||||
@@ -333,12 +358,22 @@ Hyphens denote a range of pages and commas separate page numbers. If you
|
||||
prefer to use spaces, quote all of the page numbers:
|
||||
`--pages '2, 3, 5, 7'`.
|
||||
|
||||
The token `end` (case-insensitive) is an alias for the last page in the
|
||||
document. For example, `--pages 3-end` OCRs from page 3 through the
|
||||
final page, and `--pages end` OCRs only the last page:
|
||||
|
||||
```bash
|
||||
ocrmypdf --pages 3-end input.pdf output.pdf
|
||||
ocrmypdf --pages end input.pdf output.pdf
|
||||
```
|
||||
|
||||
OCRmyPDF will warn if your list of page numbers contains duplicates or
|
||||
overlapping pages. OCRmyPDF does not currently account for document page
|
||||
numbers, such as an introduction section of a book that uses Roman
|
||||
numerals. It simply counts the number of virtual pieces of paper since
|
||||
the start. If your list of pages is out of numerical order, OCRmyPDF
|
||||
will sort it for you.
|
||||
overlapping pages. (Repeated page numbers are de-duplicated automatically,
|
||||
since the underlying set of pages is what matters.) OCRmyPDF does not
|
||||
currently account for document page numbers, such as an introduction
|
||||
section of a book that uses Roman numerals. It simply counts the number
|
||||
of virtual pieces of paper since the start. If your list of pages is out
|
||||
of numerical order, OCRmyPDF will sort it for you.
|
||||
|
||||
Regardless of the argument to `--pages`, OCRmyPDF will optimize all
|
||||
pages/images in the file and convert it to PDF/A, unless you disable
|
||||
|
||||
+136
-23
@@ -71,15 +71,29 @@ application (as opposed to the more conventional case, where a Docker
|
||||
container runs as a server). For that reason we usually use the `--rm`
|
||||
argument to delete the container when it exits.
|
||||
|
||||
:::{note}
|
||||
The image runs as a non-root user (`app`, uid/gid 1000) by default,
|
||||
rather than as root. This is a defense-in-depth measure: a flaw in
|
||||
OCRmyPDF or one of its dependencies cannot trivially act as root inside
|
||||
the container. The examples below assume **rootless Docker** or
|
||||
**Podman**; the differences for traditional *rootful* Docker are
|
||||
described separately under *Special case: rootful Docker* below.
|
||||
:::
|
||||
|
||||
To start a Docker container (instance of the image):
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm -i jbarlow83/ocrmypdf-alpine (... all other arguments here...) - -
|
||||
:::
|
||||
|
||||
For convenience, create a shell alias to hide the Docker command. It is
|
||||
easier to send the input file as stdin and read the output from stdout
|
||||
-- **this avoids the messy permission issues with Docker entirely**.
|
||||
### Recommended: pipe through stdin and stdout
|
||||
|
||||
The easiest and most portable way to use the image is to send the input
|
||||
file on stdin and read the output from stdout. This **avoids file
|
||||
permission issues entirely** -- nothing is written to a mounted
|
||||
directory, so it does not matter which user the container runs as, nor
|
||||
whether you use rootless or rootful Docker. For convenience, create a
|
||||
shell alias to hide the Docker command:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i jbarlow83/ocrmypdf-alpine'
|
||||
@@ -90,28 +104,42 @@ docker_ocrmypdf - - <input.pdf >output.pdf
|
||||
Or in the wonderful [fish shell](https://fishshell.com/):
|
||||
|
||||
:::{code} fish
|
||||
alias docker_ocrmypdf 'docker run --rm jbarlow83/ocrmypdf-alpine'
|
||||
alias docker_ocrmypdf 'docker run --rm -i jbarlow83/ocrmypdf-alpine'
|
||||
funcsave docker_ocrmypdf
|
||||
:::
|
||||
|
||||
Alternately, you could mount the local current working directory as a
|
||||
Docker volume:
|
||||
{#docker-volumes}
|
||||
### Bind-mounted volumes
|
||||
|
||||
If you would rather mount a directory and pass file paths, you need to
|
||||
consider which user owns the files OCRmyPDF writes back into that
|
||||
directory. The image's default working directory is `/data`, so mounting
|
||||
your files there lets you pass plain relative paths without an explicit
|
||||
`--workdir`. Because the container runs as the non-root `app` user, the
|
||||
right invocation otherwise depends on your container runtime.
|
||||
|
||||
**Rootless Docker (the assumed default).** Your own account runs the
|
||||
daemon, so the container's `root` maps back to *your* unprivileged host
|
||||
user, while every other container uid -- including the image's default
|
||||
`app`/1000 -- maps to a *subordinate* uid. A directory you own on the
|
||||
host therefore appears owned by `root` inside the container, so the
|
||||
default `app` user usually **cannot write to it at all**. Run the job as
|
||||
container-`root`, which under rootless Docker is still your ordinary host
|
||||
user, so the write succeeds and the output is owned by you:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" --workdir /data -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias docker_ocrmypdf='docker run --rm -i --user 0:0 -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
## Podman
|
||||
|
||||
Especially if you use [Podman](https://podman.io/) (or use Docker in
|
||||
rootless mode), you may need to add `--userns keep-id` there,
|
||||
otherwise you may get access errors, because the user ID is otherwise not
|
||||
mapped to the same UID as on the host:
|
||||
**Podman.** Podman provides `--userns keep-id`, which maps your host uid
|
||||
straight through into the container. Combined with `--user`, you run as
|
||||
your own uid and own the output directly, otherwise you may get access
|
||||
errors because the user ID is not mapped to the same UID as on the host:
|
||||
|
||||
:::{code} bash
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id --workdir /data -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
If you have SELinux enabled, you may additionally need to add the `:Z` [suffix to
|
||||
@@ -124,10 +152,27 @@ the end of the linked podman documentation for details. This results in
|
||||
the following full command:
|
||||
|
||||
:::{code} bash
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id --workdir /data -v "$PWD:/data" --security-opt label=disable jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id -v "$PWD:/data" --security-opt label=disable jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
{#docker-rootful}
|
||||
### Special case: rootful Docker
|
||||
|
||||
With a traditional root daemon, container uid *N* is the *same* uid *N*
|
||||
on the host. Running the container as root would therefore fill your
|
||||
mounted directory with root-owned files and -- more importantly -- a
|
||||
container escape would run as real host root. Drop to your own uid so the
|
||||
output is owned by you and the process stays unprivileged:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf input.pdf output.pdf
|
||||
:::
|
||||
|
||||
The non-root default and the `--user` override both reduce the risk here,
|
||||
but rootless Docker or Podman remain the safer choice when available.
|
||||
|
||||
{#docker-lang-packs}
|
||||
## Adding languages to the Docker image
|
||||
|
||||
@@ -139,8 +184,12 @@ creating a new Dockerfile based on the public one.
|
||||
:::{code} dockerfile
|
||||
FROM jbarlow83/ocrmypdf
|
||||
|
||||
# The image runs as the non-root "app" user, so switch back to root for
|
||||
# build steps that install packages, then drop back to "app".
|
||||
USER root
|
||||
# Example: add Italian
|
||||
RUN apt install tesseract-ocr-ita
|
||||
RUN apt-get update && apt-get install -y tesseract-ocr-ita
|
||||
USER app
|
||||
:::
|
||||
|
||||
To install language packs (training data) such as the
|
||||
@@ -179,7 +228,11 @@ Extending the Docker image
|
||||
--------------------------
|
||||
|
||||
You can extend the Docker image with your own customizations, similar to
|
||||
the way it is extended to add language packs.
|
||||
the way it is extended to add language packs. Because the image runs as
|
||||
the non-root `app` user, switch to `USER root` for any build steps that
|
||||
require root (installing packages, writing to system directories) and
|
||||
back to `USER app` afterwards, as shown in the language pack example
|
||||
above.
|
||||
|
||||
Note that the Docker image is subject to change at any time. For
|
||||
example, the base image may be updated to a newer version of Ubuntu or
|
||||
@@ -196,7 +249,7 @@ Executing the test suite
|
||||
The OCRmyPDF test suite is installed with image. To run it:
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm --entrypoint python jbarlow83/ocrmypdf -m pytest
|
||||
docker run --rm --workdir /app --entrypoint python jbarlow83/ocrmypdf -m pytest
|
||||
:::
|
||||
|
||||
Accessing the shell
|
||||
@@ -205,7 +258,15 @@ Accessing the shell
|
||||
To use the shell in the Docker image:
|
||||
|
||||
:::{code} bash
|
||||
docker run -it --entrypoint sh jbarlow83/ocrmypdf
|
||||
docker run -it --entrypoint sh jbarlow83/ocrmypdf-alpine
|
||||
:::
|
||||
|
||||
This shell runs as the non-root `app` user. If you need root inside the
|
||||
container -- for example to install extra packages with `apk` or `apt` --
|
||||
add `--user root`:
|
||||
|
||||
:::{code} bash
|
||||
docker run -it --user root --entrypoint sh jbarlow83/ocrmypdf-alpine
|
||||
:::
|
||||
|
||||
Using the OCRmyPDF web service wrapper
|
||||
@@ -215,7 +276,7 @@ The OCRmyPDF Docker image includes an example, barebones HTTP web
|
||||
service. The webservice may be launched as follows:
|
||||
|
||||
:::{code} bash
|
||||
docker run --entrypoint python -p 5000:5000 jbarlow83/ocrmypdf webservice.py
|
||||
docker run --entrypoint python -p 5000:5000 jbarlow83/ocrmypdf /app/webservice.py
|
||||
:::
|
||||
|
||||
We omit the `--rm` parameter so that the container will not be
|
||||
@@ -249,3 +310,55 @@ Affero GPLv3 (AGPLv3) since Ghostscript is also licensed in this way.
|
||||
In addition to the above, please read our
|
||||
`general remarks on using OCRmyPDF as a service <ocr-service>`{.interpreted-text
|
||||
role="ref"}.
|
||||
|
||||
Using the batch web interface
|
||||
-----------------------------
|
||||
|
||||
The Docker image also includes a batch web interface, in `webui/`. It
|
||||
accepts many files in one submission, OCRs them in the background, and
|
||||
returns the results individually or as a single zip archive. Start it
|
||||
with:
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm -p 8000:8000 --entrypoint python3 jbarlow83/ocrmypdf -m webui
|
||||
:::
|
||||
|
||||
Then open <http://localhost:8000/>.
|
||||
|
||||
A Compose file with sensible resource limits is provided:
|
||||
|
||||
:::{code} bash
|
||||
docker compose -f misc/docker-compose.webui.yml up --build
|
||||
:::
|
||||
|
||||
It is configured entirely through environment variables:
|
||||
|
||||
| Variable | Default | Meaning |
|
||||
| --- | --- | --- |
|
||||
| `OCRMYPDF_WEBUI_PORT` | `8000` | Port to listen on |
|
||||
| `OCRMYPDF_WEBUI_HOST` | `0.0.0.0` | Address to bind |
|
||||
| `OCRMYPDF_WEBUI_WORKERS` | half the CPUs, max 4 | Files OCR'd at once |
|
||||
| `OCRMYPDF_WEBUI_OCR_JOBS` | CPUs ÷ workers | `--jobs` for each file |
|
||||
| `OCRMYPDF_WEBUI_MAX_FILES` | `50` | Files allowed per submission |
|
||||
| `OCRMYPDF_WEBUI_MAX_UPLOAD_MB` | `500` | Size limit per file |
|
||||
| `OCRMYPDF_WEBUI_BATCH_TTL_SECONDS` | `3600` | Retention before deletion |
|
||||
| `OCRMYPDF_WEBUI_JOB_TIMEOUT_SECONDS` | `1800` | Limit for one file |
|
||||
| `OCRMYPDF_WEBUI_WORK_DIR` | `/var/tmp/ocrmypdf-webui` | Scratch space |
|
||||
|
||||
`WORKERS × OCR_JOBS` should be roughly the number of cores available to
|
||||
the container.
|
||||
|
||||
:::{warning}
|
||||
Like the Streamlit webservice above, the batch web interface has **no
|
||||
authentication and no rate limiting**. Run it on a trusted network, or
|
||||
behind a reverse proxy that terminates TLS and authenticates users.
|
||||
Uploaded files and their results are readable by anyone who can reach
|
||||
the server until their batch expires.
|
||||
:::
|
||||
|
||||
Because batch state is held in memory, the server must run as a single
|
||||
process. To handle more load, increase `OCRMYPDF_WEBUI_WORKERS` rather
|
||||
than starting additional server workers.
|
||||
|
||||
This interface is also licensed under the Affero GPLv3, for the same
|
||||
reason as the webservice above.
|
||||
|
||||
@@ -49,3 +49,31 @@ pdftk input.pdf cat output output.pdf
|
||||
|
||||
Sometimes Acrobat can repair PDFs with its [Preflight
|
||||
tool](https://helpx.adobe.com/acrobat/using/correcting-problem-areas-preflight-tool.html).
|
||||
|
||||
(tesseract-config-missing)=
|
||||
|
||||
## Tesseract cannot open its config file \'hocr\' or \'txt\'
|
||||
|
||||
:::{code}
|
||||
ERROR - Tesseract cannot open its config file 'hocr'.
|
||||
:::
|
||||
|
||||
OCRmyPDF asks Tesseract to produce `hocr` and `txt` output. Tesseract
|
||||
reads the instructions for these output formats from configuration files
|
||||
named `hocr` and `txt` that live in the `configs/` subdirectory of its
|
||||
`tessdata` folder. If those files are missing, Tesseract prints
|
||||
`read_params_file: Can't open hocr`, exits without error, and produces no
|
||||
output.
|
||||
|
||||
This usually happens when a `tessdata` directory was assembled by hand --
|
||||
for example, by downloading individual `.traineddata` files from
|
||||
[tessdata_best](https://github.com/tesseract-ocr/tessdata_best) and
|
||||
pointing `TESSDATA_PREFIX` at them -- because those repositories do not
|
||||
include the `configs/` directory. A complete Tesseract installation from
|
||||
your operating system\'s package manager includes it.
|
||||
|
||||
To fix this, ensure the `configs/hocr` and `configs/txt` files exist in
|
||||
the `tessdata` directory that Tesseract is using. Copying the `configs/`
|
||||
directory from a full Tesseract installation is sufficient. See
|
||||
{envvar}`TESSDATA_PREFIX` for more on selecting an alternate `tessdata`
|
||||
folder.
|
||||
|
||||
+1
-1
@@ -17,7 +17,7 @@ image processing and OCR (recognized, searchable text) to existing PDFs.
|
||||
:maxdepth: 1
|
||||
|
||||
introduction
|
||||
release_notes
|
||||
releasenotes/index
|
||||
installation
|
||||
languages
|
||||
jbig2
|
||||
|
||||
+191
-151
@@ -1,42 +1,42 @@
|
||||
---
|
||||
myst:
|
||||
substitutions:
|
||||
deb_11: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/debian_11/ocrmypdf.svg
|
||||
:alt: Debian 11
|
||||
:::
|
||||
deb_12: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/debian_12/ocrmypdf.svg
|
||||
:alt: Debian 12
|
||||
:::
|
||||
deb_13: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/debian_13/ocrmypdf.svg
|
||||
:alt: Debian 13
|
||||
:::
|
||||
deb_unstable: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/debian_unstable/ocrmypdf.svg
|
||||
:alt: Debian unstable
|
||||
:::
|
||||
fedora_38: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/fedora_38/ocrmypdf.svg
|
||||
:alt: Fedora 38
|
||||
fedora_40: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/fedora_40/ocrmypdf.svg
|
||||
:alt: Fedora 40
|
||||
:::
|
||||
fedora_39: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/fedora_39/ocrmypdf.svg
|
||||
:alt: Fedora 39
|
||||
fedora_41: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/fedora_41/ocrmypdf.svg
|
||||
:alt: Fedora 41
|
||||
:::
|
||||
fedora_rawhide: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
||||
:alt: Fedore Rawhide
|
||||
:alt: Fedora Rawhide
|
||||
:::
|
||||
latest: |-
|
||||
:::{image} https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||
:alt: OCRmyPDF latest released version on PyPI
|
||||
:::
|
||||
ubu_2004: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
||||
:alt: Ubuntu 20.04 LTS
|
||||
:::
|
||||
ubu_2204: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/ubuntu_22_04/ocrmypdf.svg
|
||||
:alt: Ubuntu 22.04 LTS
|
||||
:::
|
||||
ubu_2404: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/ubuntu_24_04/ocrmypdf.svg
|
||||
:alt: Ubuntu 24.04 LTS
|
||||
:::
|
||||
---
|
||||
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
@@ -54,18 +54,16 @@ These platforms have one-liner installs:
|
||||
:::{list-table}
|
||||
:header-rows: 0
|
||||
|
||||
* - Homebrew (macOS and Linux)
|
||||
- ``brew install ocrmypdf``
|
||||
* - Debian, Ubuntu
|
||||
- ``apt install ocrmypdf``
|
||||
* - Windows Subsystem for Linux
|
||||
- ``apt install ocrmypdf``
|
||||
* - Fedora
|
||||
- ``dnf install ocrmypdf tesseract-osd``
|
||||
* - macOS (Homebrew)
|
||||
- ``brew install ocrmypdf``
|
||||
* - macOS (MacPorts)
|
||||
- ``port install ocrmypdf``
|
||||
* - LinuxBrew
|
||||
- ``brew install ocrmypdf``
|
||||
* - FreeBSD
|
||||
- ``pkg install textproc/py-ocrmypdf``
|
||||
* - Snap (snapcraft packaging)
|
||||
@@ -82,15 +80,15 @@ install, or install a more recent version than your platform provides, read on.
|
||||
|
||||
## Installing on Linux
|
||||
|
||||
### Debian and Ubuntu 20.04 or newer
|
||||
### Debian and Ubuntu 22.04 or newer
|
||||
|
||||
:::{list-table}
|
||||
:header-rows: 1
|
||||
|
||||
* - OCRmyPDF versions in Debian & Ubuntu
|
||||
* - {{ latest }}
|
||||
* - {{ deb_11 }} {{ deb_12 }} {{ deb_unstable }}
|
||||
* - {{ ubu_2004 }} {{ ubu_2204 }}
|
||||
* - {{ deb_12 }} {{ deb_13 }} {{ deb_unstable }}
|
||||
* - {{ ubu_2204 }} {{ ubu_2404 }}
|
||||
:::
|
||||
|
||||
Users of Debian or Ubuntu may simply
|
||||
@@ -112,9 +110,9 @@ For full details on version availability for your platform, check the
|
||||
:::{note}
|
||||
OCRmyPDF for Debian and Ubuntu currently omit the JBIG2 encoder.
|
||||
OCRmyPDF works fine without it but will produce larger output files.
|
||||
If you build jbig2enc from source, ocrmypdf will
|
||||
automatically detect it (specifically the `jbig2` binary) on the
|
||||
`PATH`. To add JBIG2 encoding, see {ref}`jbig2`.
|
||||
All JBIG2 patents expired in 2017, so if you build jbig2enc from source,
|
||||
OCRmyPDF will automatically detect it on the `PATH`.
|
||||
To add JBIG2 encoding, see {ref}`jbig2`.
|
||||
:::
|
||||
|
||||
### Fedora
|
||||
@@ -124,7 +122,7 @@ automatically detect it (specifically the `jbig2` binary) on the
|
||||
|
||||
* - OCRmyPDF version
|
||||
* - {{latest}}
|
||||
* - {{fedora_38}} {{fedora_39}} {{fedora_rawhide}}
|
||||
* - {{fedora_40}} {{fedora_41}} {{fedora_rawhide}}
|
||||
:::
|
||||
|
||||
Users of Fedora may simply
|
||||
@@ -141,21 +139,20 @@ to install the latest version from source. See [Installing HEAD revision
|
||||
from sources](#installing-head-revision-from-sources).
|
||||
|
||||
:::{note}
|
||||
OCRmyPDF for Fedora currently omits the JBIG2 encoder due to patent
|
||||
issues. OCRmyPDF works fine without it but will produce larger output
|
||||
files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later
|
||||
will automatically detect it on the `PATH`. To add JBIG2 encoding,
|
||||
see {ref}`Installing the JBIG2 encoder <jbig2>`.
|
||||
OCRmyPDF for Fedora currently omits the JBIG2 encoder. All JBIG2 patents
|
||||
expired in 2017. OCRmyPDF works fine without it but will produce larger
|
||||
output files. If you build jbig2enc from source, OCRmyPDF will automatically
|
||||
detect it on the `PATH`. To add JBIG2 encoding, see {ref}`jbig2`.
|
||||
:::
|
||||
|
||||
(ubuntu-lts-latest)=
|
||||
|
||||
### RHEL 9
|
||||
|
||||
Prepare the environment by getting Python 3.11:
|
||||
Prepare the environment by getting Python 3.12:
|
||||
|
||||
```bash
|
||||
dnf install python3.11 python3.11-pip
|
||||
dnf install python3.12 python3.12-pip
|
||||
```
|
||||
|
||||
Then, follow [Requirements for pip and HEAD install](#requirements-for-pip-and-head-install) to install dependencies:
|
||||
@@ -167,42 +164,47 @@ dnf install ghostscript tesseract
|
||||
and build ocrmypdf in virtual environment:
|
||||
|
||||
```bash
|
||||
python3.11 -m venv .venv
|
||||
python3.12 -m venv .venv
|
||||
```
|
||||
|
||||
To add JBIG2 encoding, see {ref}`Installing the JBIG2 encoder <jbig2>`.
|
||||
|
||||
Note Fedora packages for language data haven't been branched for RHEL/EPEL, but you can get traineddata files directly from [tesseract](https://github.com/tesseract-ocr/tessdata/) and place them in `/usr/share/tesseract/tessdata`.
|
||||
|
||||
### Installing the latest version on Ubuntu 22.04 LTS
|
||||
### Installing the latest version on Ubuntu 22.04/24.04 LTS
|
||||
|
||||
Ubuntu 22.04 includes ocrmypdf 13.4.0 - you can install that with
|
||||
`apt install ocrmypdf`. To install a more recent version for the current
|
||||
user, follow these steps:
|
||||
Ubuntu includes an older version of OCRmyPDF - you can install that with
|
||||
`apt install ocrmypdf`. To install the latest version, we recommend using uv:
|
||||
|
||||
```bash
|
||||
# Install system dependencies first
|
||||
sudo apt-get update
|
||||
sudo apt-get -y install ocrmypdf python3-pip
|
||||
sudo apt-get -y install ocrmypdf
|
||||
|
||||
pip install --user --upgrade ocrmypdf
|
||||
# Install uv and upgrade to the latest OCRmyPDF
|
||||
pip install uv
|
||||
uv pip install --user --upgrade ocrmypdf
|
||||
```
|
||||
|
||||
If you get the message `WARNING: The script ocrmypdf is installed in
|
||||
'/home/$USER/.local/bin' which is not on PATH.`, you may need to re-login
|
||||
or open a new shell, or manually adjust your PATH.
|
||||
Alternatively, use Homebrew on Linux for a full-featured installation (see below).
|
||||
|
||||
To add JBIG2 encoding, see {ref}`jbig2`.
|
||||
|
||||
### Ubuntu 20.04 LTS
|
||||
### Ubuntu 20.04 LTS (and other older distributions)
|
||||
|
||||
Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with `apt`. The
|
||||
most convenient way to install recent OCRmyPDF on older Ubuntu is to use
|
||||
Homebrew on Linux (Linuxbrew).
|
||||
:::{note}
|
||||
Ubuntu 20.04 is approaching end of life. Consider upgrading to Ubuntu 22.04 or 24.04 LTS.
|
||||
:::
|
||||
|
||||
For older distributions, the most convenient way to install a recent version of
|
||||
OCRmyPDF is to use Homebrew on Linux:
|
||||
|
||||
```bash
|
||||
brew install ocrmypdf
|
||||
```
|
||||
|
||||
See {ref}`homebrew-linux` for more information on using Homebrew on Linux.
|
||||
|
||||
### Arch Linux (AUR)
|
||||
|
||||
:::{image} https://repology.org/badge/version-for-repo/aur/ocrmypdf.svg
|
||||
@@ -300,29 +302,45 @@ In general, first install the OCRmyPDF package for your system, then
|
||||
optionally use the procedure [Installing with Python
|
||||
pip](#installing-with-python-pip) to install a more recent version.
|
||||
|
||||
## Installing on macOS
|
||||
(homebrew-linux)=
|
||||
|
||||
### Homebrew
|
||||
## Installing with Homebrew (macOS and Linux)
|
||||
|
||||
:::{image} https://img.shields.io/homebrew/v/ocrmypdf.svg
|
||||
:alt: homebrew
|
||||
:target: https://formulae.brew.sh/formula/ocrmypdf
|
||||
:::
|
||||
|
||||
OCRmyPDF is now a standard [Homebrew](https://brew.sh) formula. To
|
||||
install on macOS:
|
||||
[Homebrew](https://brew.sh) provides a full-featured OCRmyPDF installation
|
||||
on both macOS and Linux with all recommended dependencies. This is often
|
||||
the easiest way to get a complete, up-to-date installation.
|
||||
|
||||
```bash
|
||||
brew install ocrmypdf
|
||||
```
|
||||
|
||||
This will include only the English language pack. If you need other
|
||||
languages you can optionally install them all:
|
||||
This includes Tesseract, Ghostscript, and all required dependencies. English
|
||||
language support is included by default. For other languages:
|
||||
|
||||
```bash
|
||||
brew install tesseract-lang # Optional: Install all language packs
|
||||
```
|
||||
|
||||
:::{tip}
|
||||
**For Linux users:** Homebrew on Linux is an excellent choice when your
|
||||
distribution's package is outdated or missing optional dependencies like
|
||||
jbig2enc, pngquant, or unpaper. Homebrew provides a consistent, full-featured
|
||||
installation that works across many Linux distributions.
|
||||
|
||||
Install Homebrew on Linux: https://brew.sh
|
||||
:::
|
||||
|
||||
## Installing on macOS
|
||||
|
||||
### Homebrew
|
||||
|
||||
See {ref}`homebrew-linux` above - the installation is identical on macOS.
|
||||
|
||||
### MacPorts
|
||||
|
||||
:::{image} https://img.shields.io/badge/dynamic/json?url=https%3A%2F%2Fports.macports.org%2Fapi%2Fv1%2Fports%2Focrmypdf%2F%3Fformat%3Djson&query=version&label=MacPorts
|
||||
@@ -330,7 +348,7 @@ brew install tesseract-lang # Optional: Install all language packs
|
||||
:target: https://ports.macports.org/port/ocrmypdf
|
||||
:::
|
||||
|
||||
OCRmyPDF is includes in MacPorts:
|
||||
OCRmyPDF is included in MacPorts:
|
||||
|
||||
```bash
|
||||
sudo port install ocrmypdf
|
||||
@@ -341,14 +359,13 @@ the appropriate tesseract [language ports](https://ports.macports.org/search/?se
|
||||
|
||||
### Manual installation on macOS
|
||||
|
||||
These instructions probably work on all macOS supported by Homebrew, and are
|
||||
for installing a more current version of OCRmyPDF than is available from
|
||||
Homebrew. Note that the Homebrew versions usually track the release versions
|
||||
fairly closely.
|
||||
These instructions are for installing a more current version of OCRmyPDF than
|
||||
is available from Homebrew. Note that Homebrew versions usually track
|
||||
releases fairly closely.
|
||||
|
||||
If it's not already present, [install Homebrew](http://brew.sh/).
|
||||
|
||||
Update Homebrew:
|
||||
Update Homebrew and install dependencies:
|
||||
|
||||
```bash
|
||||
brew update
|
||||
@@ -367,16 +384,11 @@ packs. If you need other languages you can optionally install them all:
|
||||
> brew install tesseract-lang # Option 2: for all language packs
|
||||
> ```
|
||||
|
||||
Update the homebrew pip:
|
||||
Install uv and OCRmyPDF:
|
||||
|
||||
```bash
|
||||
pip install --upgrade pip
|
||||
```
|
||||
|
||||
You can then install OCRmyPDF from PyPI for the current user:
|
||||
|
||||
```bash
|
||||
pip install --user ocrmypdf
|
||||
pip install uv
|
||||
uv pip install --user ocrmypdf
|
||||
```
|
||||
|
||||
The command line program should now be available:
|
||||
@@ -405,7 +417,7 @@ You must install the following for Windows:
|
||||
Using the [winget](https://docs.microsoft.com/en-us/windows/package-manager/winget/)
|
||||
package manager:
|
||||
|
||||
- `winget install -e --id Python.Python.3.11`
|
||||
- `winget install -e --id Python.Python.3.12`
|
||||
- `winget install -e --id UB-Mannheim.TesseractOCR`
|
||||
|
||||
You will need to install Ghostscript manually, [since it does not support automated
|
||||
@@ -452,13 +464,6 @@ override the versions OCRmyPDF selects, you can modify the `PATH` environment
|
||||
variable. [Follow these directions](https://www.computerhope.com/issues/ch000549.htm#dospath)
|
||||
to change the PATH.
|
||||
|
||||
:::{warning}
|
||||
As of early 2021, users have reported problems with the Microsoft Store version of
|
||||
Python and OCRmyPDF. These issues affect many other third party Python packages.
|
||||
Please download Python from Python.org or a package manager instead of the
|
||||
Microsoft Store version.
|
||||
:::
|
||||
|
||||
:::{warning}
|
||||
32-bit Windows is not supported.
|
||||
:::
|
||||
@@ -551,23 +556,35 @@ See [Installing the Docker image](docker) for more information.
|
||||
|
||||
(installing-with-python-pip)=
|
||||
|
||||
## Installing with Python pip
|
||||
## Installing with uv (recommended)
|
||||
|
||||
OCRmyPDF is delivered by PyPI because it is a convenient way to install
|
||||
the latest version. However, PyPI and `pip` cannot address the fact
|
||||
that `ocrmypdf` depends on certain non-Python system libraries and
|
||||
programs being installed.
|
||||
We recommend using [uv](https://docs.astral.sh/uv/) for installing OCRmyPDF from PyPI.
|
||||
uv is a fast, modern Python package manager that provides better dependency resolution
|
||||
and consistent behavior across all platforms.
|
||||
|
||||
For best results, first install [your platform's
|
||||
version](https://repology.org/metapackage/ocrmypdf/versions) of
|
||||
`ocrmypdf`, using the instructions elsewhere in this document. Then
|
||||
you can use `pip` to get the latest version if your platform version
|
||||
is out of date. Chances are that this will satisfy most dependencies.
|
||||
`ocrmypdf` using the instructions elsewhere in this document to satisfy system
|
||||
dependencies. Then use uv to get the latest OCRmyPDF version.
|
||||
|
||||
```bash
|
||||
# Install uv if you don't have it
|
||||
pip install uv
|
||||
|
||||
# Install ocrmypdf in a virtual environment (recommended)
|
||||
uv venv
|
||||
source .venv/bin/activate # On Windows: .venv\Scripts\activate
|
||||
uv pip install ocrmypdf
|
||||
|
||||
# Or install globally
|
||||
uv pip install --system ocrmypdf
|
||||
```
|
||||
|
||||
Use `ocrmypdf --version` to confirm what version was installed.
|
||||
|
||||
Then you can install the latest OCRmyPDF from the Python wheels. First
|
||||
try:
|
||||
### Installing with pip
|
||||
|
||||
If you prefer pip, you can still use it:
|
||||
|
||||
```bash
|
||||
pip install --user ocrmypdf
|
||||
@@ -576,21 +593,20 @@ pip install --user ocrmypdf
|
||||
(If the message appears `Requirement already satisfied: ocrmypdf in...`,
|
||||
you will need to use `pip install --user --upgrade ocrmypdf`.)
|
||||
|
||||
You should then be able to run `ocrmypdf --version` and see that the
|
||||
latest version was located.
|
||||
### Installing with pipx
|
||||
|
||||
## Installing with pipx
|
||||
Some users may prefer pipx for isolated command-line tool installations:
|
||||
|
||||
Some users may prefer pipx. As with the method above, you will need to
|
||||
satisfy all non-Python dependencies. Then if pipx is installed, you
|
||||
can use
|
||||
```bash
|
||||
pipx install ocrmypdf
|
||||
```
|
||||
|
||||
Or run without permanent installation:
|
||||
|
||||
```bash
|
||||
pipx run ocrmypdf
|
||||
```
|
||||
|
||||
(If not installed, pipx will install first.)
|
||||
|
||||
(requirements-for-pip-and-head-install)=
|
||||
|
||||
### Requirements for pip and HEAD install
|
||||
@@ -606,27 +622,66 @@ and verapdf can validate speculative PDF/A conversion.
|
||||
|
||||
The following versions are required:
|
||||
|
||||
- Python 3.11 or newer
|
||||
- Python 3.11 or newer (3.12+ recommended)
|
||||
- Tesseract 4.1.1 or newer
|
||||
- One of: Ghostscript 9.54+ **or** pypdfium2 (Python package)
|
||||
- One of: Ghostscript 9.54+ **or** verapdf (for PDF/A output)
|
||||
- fpdf2 2.8 or newer (Python package)
|
||||
- uharfbuzz (Python package)
|
||||
- fonts-noto or equivalent (system package, recommended)
|
||||
- jbig2enc 0.29 or newer (optional)
|
||||
- pngquant 2.5 or newer (optional)
|
||||
- unpaper 6.1 (optional)
|
||||
|
||||
:::{note}
|
||||
For the best user experience, install both Ghostscript and pypdfium2.
|
||||
pypdfium2 is faster for rasterization, while Ghostscript provides
|
||||
broader compatibility and is required for certain PDF/A conversions.
|
||||
For the best user experience, install both Ghostscript and pypdfium2. pypdfium2 is
|
||||
faster for rasterization, while Ghostscript provides is required for certain PDF/A
|
||||
conversions.
|
||||
:::
|
||||
|
||||
**Dependency summary:**
|
||||
|
||||
| Feature | Option 1 | Option 2 | Notes |
|
||||
|---------|----------|----------|-------|
|
||||
| PDF rasterization | pypdfium2 (Python) | Ghostscript (binary) | pypdfium2 preferred when available |
|
||||
| PDF/A conversion | verapdf + pikepdf | Ghostscript | verapdf validates speculative conversion |
|
||||
| Text rendering | fpdf2 + uharfbuzz | - | Required |
|
||||
| OCR | tesseract-ocr | `--ocr-engine none` | Can be skipped entirely |
|
||||
|
||||
**Minimum viable installation:**
|
||||
tesseract-ocr + (pypdfium2 OR Ghostscript) + fpdf2 + uharfbuzz
|
||||
|
||||
**Recommended installation:**
|
||||
tesseract-ocr + pypdfium2 + Ghostscript + verapdf + fpdf2 + uharfbuzz + fonts-noto + unpaper + pngquant + jbig2enc
|
||||
|
||||
We recommend 64-bit versions of all software. (32-bit versions are not
|
||||
supported, although on Linux, they may still work.)
|
||||
|
||||
**fpdf2** is a required dependency that provides the text layer
|
||||
rendering engine. It replaces the legacy hOCR-based renderer with improved
|
||||
multilingual support. Install with: `pip install fpdf2`
|
||||
**fpdf2** and **uharfbuzz** are required dependencies that provide the text
|
||||
layer rendering engine. fpdf2 generates the PDF text layer, while uharfbuzz
|
||||
provides text shaping for proper multilingual support. These replace the
|
||||
legacy hOCR-based renderer. Install with: `pip install fpdf2 uharfbuzz`
|
||||
|
||||
**fonts-noto** (or an equivalent comprehensive font package) is recommended
|
||||
for proper text rendering, especially for non-Latin scripts. OCRmyPDF bundles
|
||||
a Latin font only, and discovers the rest from the fonts installed on your
|
||||
system.
|
||||
|
||||
- Debian/Ubuntu: `apt install fonts-noto`
|
||||
- Fedora: `dnf install google-noto-fonts-all`
|
||||
- macOS with Homebrew: Homebrew has no single Noto package; each family is a
|
||||
separate cask. Install at least
|
||||
`brew install --cask font-noto-sans font-noto-serif`, plus a cask per
|
||||
additional script you OCR, for example
|
||||
`brew install --cask font-noto-sans-arabic font-noto-sans-cjk`. Run
|
||||
`brew search font-noto` to list them all.
|
||||
|
||||
If OCRmyPDF warns that no installed font has glyphs for some of the text, the
|
||||
message names the characters it could not render, for example
|
||||
`'Ꮳ' U+13E3 CHEROKEE LETTER TSA`. Install the Noto font for that script — here,
|
||||
`fonts-noto-core` on Debian or `font-noto-sans-cherokee` on Homebrew. The text
|
||||
layer remains searchable and copyable either way; only its appearance when
|
||||
highlighted in a PDF viewer is affected.
|
||||
|
||||
**pypdfium2**, if present, provides fast PDF page rasterization using
|
||||
the pdfium library (the same library used by Google Chrome). It is
|
||||
@@ -642,10 +697,10 @@ or visit [verapdf.org](https://verapdf.org/).
|
||||
**jbig2enc**, if present, will be used to optimize the encoding of
|
||||
monochrome images. This can significantly reduce the file size of the
|
||||
output file. It is not required.
|
||||
[jbig2enc](https://github.com/agl/jbig2enc) is not generally
|
||||
available for Ubuntu or Debian due to lingering concerns about patent
|
||||
issues, but can easily be built from source. To add JBIG2 encoding, see
|
||||
{ref}`jbig2`.
|
||||
[jbig2enc](https://github.com/agl/jbig2enc) is not available in some
|
||||
distributions due to historical patent concerns, but all JBIG2 patents
|
||||
expired in 2017. It can easily be built from source. To add JBIG2 encoding,
|
||||
see {ref}`jbig2`.
|
||||
|
||||
:::{warning}
|
||||
Lossy JBIG2 encoding (`--jbig2-lossy`) has been removed in v17.0.0 due to
|
||||
@@ -668,8 +723,8 @@ unfortunately, the `pip install` command cannot satisfy all of them.
|
||||
|
||||
## Installing HEAD revision from sources
|
||||
|
||||
If you have `git` and Python 3.11 or newer installed, you can install
|
||||
from source. When the `pip` installer runs, it will alert you if
|
||||
If you have `git` and Python 3.12 or newer installed, you can install
|
||||
from source. (Python 3.11 is supported but 3.12+ is recommended.) When the `pip` installer runs, it will alert you if
|
||||
dependencies are missing.
|
||||
|
||||
If you prefer to build every from source, you will need to [build
|
||||
@@ -677,33 +732,39 @@ pikepdf from
|
||||
source](https://pikepdf.readthedocs.io/en/latest/installation.html#building-from-source).
|
||||
First ensure you can build and install pikepdf.
|
||||
|
||||
To install the HEAD revision from sources in the current Python 3
|
||||
environment:
|
||||
We recommend using uv to install from sources:
|
||||
|
||||
```bash
|
||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
cd OCRmyPDF
|
||||
pip install uv # If not already installed
|
||||
uv sync
|
||||
```
|
||||
|
||||
This creates a virtual environment and installs all dependencies. Activate
|
||||
the environment to use ocrmypdf:
|
||||
|
||||
```bash
|
||||
source .venv/bin/activate
|
||||
ocrmypdf --help
|
||||
```
|
||||
|
||||
Alternatively, install directly from GitHub using pip:
|
||||
|
||||
```bash
|
||||
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
```
|
||||
|
||||
Or, to install in editable mode
|
||||
allowing customization of OCRmyPDF, use the `-e` flag:
|
||||
|
||||
```bash
|
||||
pip install -e git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
```
|
||||
|
||||
You may find it easiest to install in a virtual environment, rather than
|
||||
system-wide:
|
||||
Or, to install in editable mode allowing customization:
|
||||
|
||||
```bash
|
||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
python3 -m venv .venv
|
||||
source .venv/bin/activate
|
||||
cd OCRmyPDF
|
||||
pip install .
|
||||
pip install -e .
|
||||
```
|
||||
|
||||
However, `ocrmypdf` will only be accessible on the system PATH when
|
||||
you activate the virtual environment.
|
||||
Note: `ocrmypdf` will only be accessible when the virtual environment
|
||||
is activated.
|
||||
|
||||
To run the program:
|
||||
|
||||
@@ -727,18 +788,14 @@ User features are available as optional dependencies. Install them with `uv` (re
|
||||
```bash
|
||||
# Using uv (recommended)
|
||||
uv sync --extra watcher # File watching service
|
||||
uv sync --extra webservice # Streamlit web UI
|
||||
uv sync --extra webservice # Streamlit web UI (single file)
|
||||
uv sync --extra webui # Batch web interface (multi-file upload)
|
||||
uv sync --extra watcher --extra webservice # Multiple features
|
||||
|
||||
# Using pip (also works)
|
||||
pip install ocrmypdf[watcher]
|
||||
pip install ocrmypdf[webservice]
|
||||
pip install ocrmypdf[watcher,webservice]
|
||||
```
|
||||
|
||||
### Development Tools (uv only)
|
||||
### Development Tools
|
||||
|
||||
Development tools use dependency groups and require `uv`:
|
||||
Development tools use dependency groups:
|
||||
|
||||
```bash
|
||||
# Testing infrastructure
|
||||
@@ -754,11 +811,6 @@ uv sync --group streamlit-dev
|
||||
uv sync
|
||||
```
|
||||
|
||||
:::{note}
|
||||
**User features** (`watcher`, `webservice`) work with both `uv` and `pip`.
|
||||
**Developer tools** (`test`, `docs`, `streamlit-dev`) require `uv` and use dependency groups (PEP 735).
|
||||
:::
|
||||
|
||||
**Why use uv?**
|
||||
|
||||
- Modern, fast Python package manager
|
||||
@@ -766,7 +818,7 @@ uv sync
|
||||
- Better dependency resolution
|
||||
- Consistent across all platforms
|
||||
|
||||
Install uv: `pip install uv` or visit https://docs.astral.sh/uv/
|
||||
Install uv: `curl -LsSf https://astral.sh/uv/install.sh | sh` or visit https://docs.astral.sh/uv/
|
||||
|
||||
### For development
|
||||
|
||||
@@ -775,12 +827,9 @@ To install all of the development and test requirements:
|
||||
```bash
|
||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
cd OCRmyPDF
|
||||
pip install uv # Install uv if not already installed
|
||||
uv sync --group test
|
||||
uv sync --all-groups
|
||||
```
|
||||
|
||||
Note: Development requires `uv`. The old `pip install -e .[test]` method is no longer supported.
|
||||
|
||||
To add JBIG2 encoding, see {ref}`jbig2`.
|
||||
|
||||
## Shell completions
|
||||
@@ -800,14 +849,5 @@ To manually install the `fish` completion, copy
|
||||
|
||||
## Note on 32-bit support
|
||||
|
||||
Many Python libraries no longer provide 32-bit binary wheels for Linux. This
|
||||
includes many of the libraries that OCRmyPDF depends on, such as
|
||||
Pillow. The easiest way to express this to end users is to say we don't
|
||||
support 32-bit Linux.
|
||||
|
||||
However, if your Linux distribution still supports 32-bit binaries, you
|
||||
can still install and use OCRmyPDF. A warning message will appear.
|
||||
In practice, OCRmyPDF may need more than 32-bit memory space to run when
|
||||
large documents are processed, so there are practical limitations to what
|
||||
users can accomplish with it. Still, for the common use case of an 32-bit
|
||||
ARM NAS or Raspberry Pi processing small documents, it should work.
|
||||
We don't support any 32-bit system, including 32-bit Python or 32-bit
|
||||
Ghostscript on Windows.
|
||||
+16
-4
@@ -178,11 +178,23 @@ v17 addresses through alternative codepaths. When Ghostscript is used:
|
||||
encoding, which may introduce compression artifacts, if Ghostscript
|
||||
PDF/A is enabled.
|
||||
- Ghostscript may transcode grayscale and color images, potentially
|
||||
lossily, based on an internal algorithm. This
|
||||
behavior can be suppressed by setting `--pdfa-image-compression` to
|
||||
`jpeg` or `lossless` to set all images to one type or the other.
|
||||
Ghostscript lacks an option to maintain the input image's format.
|
||||
lossily, based on an internal algorithm. By default
|
||||
(`--pdfa-image-compression=auto`) OCRmyPDF selects lossless image
|
||||
compression at `-O0` so Ghostscript will not transcode lossless images
|
||||
to JPEG. At `-O1` (the default optimization level) and above, `auto`
|
||||
defers to Ghostscript's heuristic instead; `-O1` is a historical
|
||||
exception, kept for backwards compatibility because coercing it to
|
||||
lossless can substantially bloat output. You can override this by
|
||||
setting `--pdfa-image-compression` to `jpeg` or `lossless` to force all
|
||||
images to one type or the other. `lossless` passes existing JPEGs
|
||||
through untouched (re-encoding them losslessly would only inflate them)
|
||||
while encoding non-JPEG images losslessly.
|
||||
(Modern Ghostscript can copy JPEG images without transcoding them.)
|
||||
Advanced users can also tune Ghostscript's image recompression with
|
||||
`--ghostscript-jpeg-quality` and `--ghostscript-jpeg-maxdpi`; see
|
||||
[Advanced Ghostscript tuning](advanced.md#advanced-ghostscript-tuning).
|
||||
Most users should prefer `--jpeg-quality` (applied by the OCRmyPDF
|
||||
optimizer) over those Ghostscript-scoped controls.
|
||||
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
||||
one of the standard XMP metadata namespaces for PDFs. In particular,
|
||||
PRISM Metadata is removed.
|
||||
|
||||
+3
-1
@@ -47,7 +47,9 @@ OCRmyPDF has the following runtime dependencies:
|
||||
**For text rendering** (expressing OCR results in PDF):
|
||||
- `fpdf2` (Python package) - Required for text layer rendering
|
||||
- `uharfbuzz` (Python package) - Required for text layer rendering
|
||||
- `font-noto` (system package) - Recommended for text layer rendering
|
||||
- Noto fonts (system package) - Recommended for text layer rendering.
|
||||
`fonts-noto` on Debian/Ubuntu, `google-noto-fonts-all` on Fedora; Homebrew
|
||||
has no single Noto package, only per-family casks such as `font-noto-sans`.
|
||||
|
||||
**Other dependencies**:
|
||||
- `unpaper` (system binary) - Optional, enables `--clean` and `--clean-final`
|
||||
|
||||
+10
-1
@@ -98,7 +98,16 @@ If `pngquant` is installed, OCRmyPDF will use it to perform quantize
|
||||
paletted images to reduce their size.
|
||||
|
||||
The quality of JPEGs may be lowered, on the assumption that a lower
|
||||
quality image may be suitable for storage after OCR.
|
||||
quality image may be suitable for storage after OCR. Use `--jpeg-quality`
|
||||
to control the optimizer's JPEG quality target. The optimizer is the
|
||||
recommended way to reduce JPEG image sizes: it applies consistently
|
||||
regardless of whether Ghostscript was used to produce a PDF/A.
|
||||
|
||||
If you specifically need to tune Ghostscript's own PDF/A image handling
|
||||
(for example, to force a hard DPI cap), see
|
||||
[Advanced Ghostscript tuning](advanced.md#advanced-ghostscript-tuning)
|
||||
for the separate `--ghostscript-jpeg-quality` and
|
||||
`--ghostscript-jpeg-maxdpi` options.
|
||||
|
||||
It is not possible to optimize all image types. Uncommon image types may
|
||||
be skipped by the optimizer.
|
||||
|
||||
+13
-13
@@ -120,6 +120,7 @@ A plugin may provide the following hooks. Hooks must be decorated with
|
||||
```python
|
||||
from ocrmypdf import hookimpl
|
||||
|
||||
|
||||
@hookimpl
|
||||
def add_options(parser):
|
||||
pass
|
||||
@@ -205,12 +206,11 @@ from ocrmypdf._options import OcrOptions
|
||||
|
||||
```python
|
||||
# Before (v16 and earlier)
|
||||
def check_options(options: argparse.Namespace) -> None:
|
||||
...
|
||||
def check_options(options: argparse.Namespace) -> None: ...
|
||||
|
||||
|
||||
# After (v17+)
|
||||
def check_options(options: OcrOptions) -> None:
|
||||
...
|
||||
def check_options(options: OcrOptions) -> None: ...
|
||||
```
|
||||
|
||||
**Attribute access unchanged:**
|
||||
@@ -229,6 +229,7 @@ options.tesseract_timeout
|
||||
def check_options(options):
|
||||
options.some_computed_value = compute_value(options)
|
||||
|
||||
|
||||
# After (v17 pattern - compute at point of use)
|
||||
def some_function(options):
|
||||
computed = compute_value(options)
|
||||
@@ -336,19 +337,17 @@ from ocrmypdf import OcrElement, OcrClass, BoundingBox
|
||||
|
||||
# OcrElement - represents any OCR structural unit
|
||||
page = OcrElement(
|
||||
ocr_class=OcrClass.PAGE,
|
||||
bbox=BoundingBox(0, 0, 612, 792),
|
||||
children=[...]
|
||||
ocr_class=OcrClass.PAGE, bbox=BoundingBox(0, 0, 612, 792), children=[...]
|
||||
)
|
||||
|
||||
# BoundingBox - axis-aligned bounding box (left, top, right, bottom)
|
||||
bbox = BoundingBox(left=100, top=50, right=300, bottom=80)
|
||||
|
||||
# OcrClass - constants for element types
|
||||
OcrClass.PAGE # "ocr_page"
|
||||
OcrClass.LINE # "ocr_line"
|
||||
OcrClass.WORD # "ocrx_word"
|
||||
OcrClass.PARAGRAPH # "ocr_par"
|
||||
OcrClass.PAGE # "ocr_page"
|
||||
OcrClass.LINE # "ocr_line"
|
||||
OcrClass.WORD # "ocrx_word"
|
||||
OcrClass.PARAGRAPH # "ocr_par"
|
||||
```
|
||||
|
||||
**Navigating the tree:**
|
||||
@@ -378,6 +377,7 @@ from pathlib import Path
|
||||
from ocrmypdf.pluginspec import OcrEngine
|
||||
from ocrmypdf import OcrElement, OcrClass, BoundingBox
|
||||
|
||||
|
||||
class MyOcrEngine(OcrEngine):
|
||||
def generate_ocr(
|
||||
self,
|
||||
@@ -402,10 +402,10 @@ class MyOcrEngine(OcrEngine):
|
||||
text="Hello",
|
||||
),
|
||||
# ... more words
|
||||
]
|
||||
],
|
||||
),
|
||||
# ... more lines
|
||||
]
|
||||
],
|
||||
)
|
||||
|
||||
def supports_generate_ocr(self) -> bool:
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,34 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Release notes
|
||||
|
||||
OCRmyPDF uses [semantic versioning](http://semver.org/) for its
|
||||
command line interface and its public API.
|
||||
|
||||
OCRmyPDF's output messages are not considered part of the stable interface -
|
||||
that is, output messages may be improved at any release level, so parsing them
|
||||
may be unreliable. Use the API to depend on precise behavior.
|
||||
|
||||
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||
wish to use some of its features for working with PDFs.
|
||||
|
||||
The most recent release of OCRmyPDF is . Any newer versions
|
||||
referred to in these notes may exist the main branch but have not been
|
||||
tagged yet.
|
||||
|
||||
OCRmyPDF typically supports the three most recent Python versions.
|
||||
|
||||
:::{note}
|
||||
Attention maintainers: these release notes may be updated with information
|
||||
about a forthcoming release that has not been tagged yet. A release is only
|
||||
official when it's tagged and posted to PyPI.
|
||||
:::
|
||||
|
||||
```{toctree}
|
||||
:glob: true
|
||||
:maxdepth: 1
|
||||
:reversed: true
|
||||
|
||||
version*
|
||||
```
|
||||
@@ -0,0 +1,14 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v2
|
||||
|
||||
## v2.2-stable (2014-09-29)
|
||||
|
||||
OCRmyPDF versions 1 and 2 were implemented as shell scripts. OCRmyPDF
|
||||
3.0+ is a fork that gradually replaced all shell scripts with Python
|
||||
while maintaining the existing command line arguments. No one is
|
||||
maintaining old versions.
|
||||
|
||||
For details on older versions, see the [final version of its release
|
||||
notes](https://github.com/fritz-hh/OCRmyPDF/blob/7fd3dbdf42ca53a619412ce8add7532c5e81a9d1/RELEASE_NOTES.md).
|
||||
@@ -0,0 +1,219 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v3
|
||||
|
||||
## v3.2.1
|
||||
|
||||
Changes
|
||||
|
||||
- Fixed {issue}`47`
|
||||
"convert() got and unexpected keyword argument 'dpi'" by upgrading to
|
||||
img2pdf 0.2
|
||||
- Tweaked the Dockerfiles
|
||||
|
||||
## v3.2
|
||||
|
||||
New features
|
||||
|
||||
- Lossless reconstruction: when possible, OCRmyPDF will inject text
|
||||
layers without otherwise manipulating the content and layout of a PDF
|
||||
page. For example, a PDF containing a mix of vector and raster
|
||||
content would see the vector content preserved. Images may still be
|
||||
transcoded during PDF/A conversion. (`--deskew` and
|
||||
`--clean-final` disable this mode, necessarily.)
|
||||
- New argument `--tesseract-pagesegmode` allows you to pass page
|
||||
segmentation arguments to Tesseract OCR. This helps for two column
|
||||
text and other situations that confuse Tesseract.
|
||||
- Added a new "polyglot" version of the Docker image, that generates
|
||||
Tesseract with all languages packs installed, for the polyglots among
|
||||
us. It is much larger.
|
||||
|
||||
Changes
|
||||
|
||||
- JPEG transcoding quality is now 95 instead of the default 75. Bigger
|
||||
file sizes for less degradation.
|
||||
|
||||
## v3.1.1
|
||||
|
||||
Changes
|
||||
|
||||
- Fixed bug that caused incorrect page size and DPI calculations on
|
||||
documents with mixed page sizes
|
||||
|
||||
## v3.1
|
||||
|
||||
Changes
|
||||
|
||||
- Default output format is now PDF/A-2b instead of PDF/A-1b
|
||||
- Python 3.5 and macOS El Capitan are now supported platforms - no
|
||||
changes were needed to implement support
|
||||
- Improved some error messages related to missing input files
|
||||
- Fixed {issue}`20`: uppercase .PDF extension not accepted
|
||||
- Fixed an issue where OCRmyPDF failed to text that certain pages
|
||||
contained previously OCR'ed text, such as OCR text produced by
|
||||
Tesseract 3.04
|
||||
- Inserts /Creator tag into PDFs so that errors can be traced back to
|
||||
this project
|
||||
- Added new option `--pdf-renderer=auto`, to let OCRmyPDF pick the
|
||||
best PDF renderer. Currently it always chooses the 'hocrtransform'
|
||||
renderer but that behavior may change.
|
||||
- Set up Travis CI automatic integration testing
|
||||
|
||||
## v3.0
|
||||
|
||||
New features
|
||||
|
||||
- Easier installation with a Docker container or Python's `pip`
|
||||
package manager
|
||||
- Eliminated many external dependencies, so it's easier to setup
|
||||
- Now installs `ocrmypdf` to `/usr/local/bin` or equivalent for
|
||||
system-wide access and easier typing
|
||||
- Improved command line syntax and usage help (`--help`)
|
||||
- Tesseract 3.03+ PDF page rendering can be used instead for better
|
||||
positioning of recognized text (`--pdf-renderer tesseract`)
|
||||
- PDF metadata (title, author, keywords) are now transferred to the
|
||||
output PDF
|
||||
- PDF metadata can also be set from the command line (`--title`,
|
||||
etc.)
|
||||
- Automatic repairs malformed input PDFs if possible
|
||||
- Added test cases to confirm everything is working
|
||||
- Added option to skip extremely large pages that take too long to OCR
|
||||
and are often not OCRable (e.g. large scanned maps or diagrams);
|
||||
other pages are still processed (`--skip-big`)
|
||||
- Added option to kill Tesseract OCR process if it seems to be taking
|
||||
too long on a page, while still processing other pages
|
||||
(`--tesseract-timeout`)
|
||||
- Less common colorspaces (CMYK, palette) are now supported by
|
||||
conversion to RGB
|
||||
- Multiple images on the same PDF page are now supported
|
||||
|
||||
Changes
|
||||
|
||||
- New, robust rewrite in Python 3.4+ with
|
||||
[ruffus](http://www.ruffus.org.uk/index.html) pipelines
|
||||
|
||||
- Now uses Ghostscript 9.14's improved color conversion model to
|
||||
preserve PDF colors
|
||||
|
||||
- OCR text is now rendered in the PDF as invisible text. Previous
|
||||
versions of OCRmyPDF incorrectly rendered visible text with an image
|
||||
on top.
|
||||
|
||||
- All "tasks" in the pipeline can be executed in parallel on any
|
||||
available CPUs, increasing performance
|
||||
|
||||
- The `-o DPI` argument has been phased out, in favor of
|
||||
`--oversample DPI`, in case we need `-o OUTPUTFILE` in the future
|
||||
|
||||
- Removed several dependencies, so it's easier to install. We no longer
|
||||
use:
|
||||
|
||||
- GNU [parallel](https://www.gnu.org/software/parallel/)
|
||||
- [ImageMagick](http://www.imagemagick.org/script/index.php)
|
||||
- Python 2.7
|
||||
- Poppler
|
||||
- [MuPDF](http://mupdf.com/docs/) tools
|
||||
- shell scripts
|
||||
- Java and [JHOVE](http://jhove.sourceforge.net/)
|
||||
- libxml2
|
||||
|
||||
- Some new external dependencies are required or optional, compared to
|
||||
v2.x:
|
||||
|
||||
- Ghostscript 9.14+
|
||||
- [qpdf](http://qpdf.sourceforge.net/) 5.0.0+
|
||||
- [Unpaper](https://github.com/Flameeyes/unpaper) 6.1 (optional)
|
||||
- some automatically managed Python packages
|
||||
|
||||
Release candidates^
|
||||
|
||||
- rc9:
|
||||
|
||||
- Fix
|
||||
{issue}`118`:
|
||||
report error if ghostscript iccprofiles are missing
|
||||
- fixed another issue related to
|
||||
{issue}`111`: PDF
|
||||
rasterized to palette file
|
||||
- add support image files with a palette
|
||||
- don't try to validate PDF file after an exception occurs
|
||||
|
||||
- rc8:
|
||||
|
||||
- Fix
|
||||
{issue}`111`:
|
||||
exception thrown if PDF is missing DocumentInfo dictionary
|
||||
|
||||
- rc7:
|
||||
|
||||
- fix error when installing direct from pip, "no such file
|
||||
'requirements.txt'"
|
||||
|
||||
- rc6:
|
||||
|
||||
- dropped libxml2 (Python lxml) since Python 3's internal XML parser
|
||||
is sufficient
|
||||
- set up Docker container
|
||||
- fix Unicode errors if recognized text contains Unicode characters
|
||||
and system locale is not UTF-8
|
||||
|
||||
- rc5:
|
||||
|
||||
- dropped Java and JHOVE in favour of qpdf
|
||||
- improved command line error output
|
||||
- additional tests and bug fixes
|
||||
- tested on Ubuntu 14.04 LTS
|
||||
|
||||
- rc4:
|
||||
|
||||
- dropped MuPDF in favour of qpdf
|
||||
- fixed some installer issues and errors in installation
|
||||
instructions
|
||||
- improve performance: run Ghostscript with multithreaded rendering
|
||||
- improve performance: use multiple cores by default
|
||||
- bug fix: checking for wrong exception on process timeout
|
||||
|
||||
- rc3: skipping version number intentionally to avoid confusion with
|
||||
Tesseract
|
||||
|
||||
- rc2: first release for public testing to test-PyPI, Github
|
||||
|
||||
- rc1: testing release process
|
||||
|
||||
## Compatibility notes
|
||||
|
||||
- `./OCRmyPDF.sh` script is still available for now
|
||||
- Stacking the verbosity option like `-vvv` is no longer supported
|
||||
- The configuration file `config.sh` has been removed. Instead, you
|
||||
can feed a file to the arguments for common settings:
|
||||
|
||||
```
|
||||
ocrmypdf input.pdf output.pdf @settings.txt
|
||||
```
|
||||
|
||||
where `settings.txt` contains *one argument per line*, for example:
|
||||
|
||||
```
|
||||
-l
|
||||
deu
|
||||
--author
|
||||
A. Merkel
|
||||
--pdf-renderer
|
||||
tesseract
|
||||
```
|
||||
|
||||
Fixes
|
||||
|
||||
- Handling of filenames containing spaces: fixed
|
||||
|
||||
Notes and known issues
|
||||
|
||||
- Some dependencies may work with lower versions than tested, so try
|
||||
overriding dependencies if they are "in the way" to see if they work.
|
||||
- `--pdf-renderer tesseract` will output files with an incorrect page
|
||||
size in Tesseract 3.03, due to a bug in Tesseract.
|
||||
- PDF files containing "inline images" are not supported and won't be
|
||||
for the 3.0 release. Scanned images almost never contain inline
|
||||
images.
|
||||
|
||||
@@ -0,0 +1,408 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v4
|
||||
|
||||
## v4.5.6
|
||||
|
||||
- Fixed {issue}`156`,
|
||||
'NoneType' object has no attribute 'getObject' on pages with no
|
||||
optional /Contents record. This should resolve all issues related to
|
||||
pages with no /Contents record.
|
||||
- Fixed {issue}`158`, ocrmypdf
|
||||
now stops and terminates if Ghostscript fails on an intermediate
|
||||
step, as it is not possible to proceed.
|
||||
- Fixed {issue}`160`,
|
||||
exception thrown on certain invalid arguments instead of error
|
||||
message
|
||||
|
||||
## v4.5.5
|
||||
|
||||
- Automated update of macOS homebrew tap
|
||||
- Fixed {issue}`154`, KeyError
|
||||
'/Contents' when searching for text on blank pages that have no
|
||||
/Contents record. Note: incomplete fix for this issue.
|
||||
|
||||
## v4.5.4
|
||||
|
||||
- Fixed `--skip-big` raising an exception if a page contains no images
|
||||
({issue}`152`) (thanks
|
||||
to @TomRaz)
|
||||
- Fixed an issue where pages with no images might trigger "cannot write
|
||||
mode P as JPEG"
|
||||
({issue}`151`)
|
||||
|
||||
## v4.5.3
|
||||
|
||||
- Added a workaround for Ghostscript 9.21 and probably earlier versions
|
||||
would fail with the error message "VMerror -25", due to a Ghostscript
|
||||
bug in XMP metadata handling
|
||||
- High Unicode characters (U+10000 and up) are no longer accepted for
|
||||
setting metadata on the command line, as Ghostscript may not handle
|
||||
them correctly.
|
||||
- Fixed an issue where the `tess4` renderer would duplicate content
|
||||
onto output pages if tesseract failed or timed out
|
||||
- Fixed `tess4` renderer not recognized when lossless reconstruction
|
||||
is possible
|
||||
|
||||
## v4.5.2
|
||||
|
||||
- Fixed {issue}`147`,
|
||||
`--pdf-renderer tess4 --clean` will produce an oversized page
|
||||
containing the original image in the bottom left corner, due to loss
|
||||
DPI information.
|
||||
- Make "using Tesseract 4.0" warning less ominous
|
||||
- Set up machinery for homebrew OCRmyPDF tap
|
||||
|
||||
## v4.5.1
|
||||
|
||||
- Fixed {issue}`137`,
|
||||
proportions of images with a non-square pixel aspect ratio would be
|
||||
distorted in output for `--force-ocr` and some other combinations
|
||||
of flags
|
||||
|
||||
## v4.5
|
||||
|
||||
- PDFs containing "Form XObjects" are now supported (issue
|
||||
{issue}`134`; PDF
|
||||
reference manual 8.10), and images they contain are taken into
|
||||
account when determining the resolution for rasterizing
|
||||
- The Tesseract 4 Docker image no longer includes all languages,
|
||||
because it took so long to build something would tend to fail
|
||||
- OCRmyPDF now warns about using `--pdf-renderer tesseract` with
|
||||
Tesseract 3.04 or lower due to issues with Ghostscript corrupting the
|
||||
OCR text in these cases
|
||||
|
||||
## v4.4.2
|
||||
|
||||
- The Docker images (ocrmypdf, ocrmypdf-polyglot, ocrmypdf-tess4) are
|
||||
now based on Ubuntu 16.10 instead of Debian stretch
|
||||
|
||||
- This makes supporting the Tesseract 4 image easier
|
||||
- This could be a disruptive change for any Docker users who built
|
||||
customized these images with their own changes, and made those
|
||||
changes in a way that depends on Debian and not Ubuntu
|
||||
|
||||
- OCRmyPDF now prevents running the Tesseract 4 renderer with Tesseract
|
||||
3.04, which was permitted in v4.4 and v4.4.1 but will not work
|
||||
|
||||
## v4.4.1
|
||||
|
||||
- To prevent a [TIFF output
|
||||
error](https://github.com/python-pillow/Pillow/issues/2206) caused
|
||||
by img2pdf >= 0.2.1 and Pillow \<= 3.4.2, dependencies have been
|
||||
tightened
|
||||
- The Tesseract 4.00 simultaneous process limit was increased from 1 to
|
||||
2, since it was observed that 1 lowers performance
|
||||
- Documentation improvements to describe the `--tesseract-config`
|
||||
feature
|
||||
- Added test cases and fixed error handling for `--tesseract-config`
|
||||
- Tweaks to setup.py to deal with issues in the v4.4 release
|
||||
|
||||
## v4.4
|
||||
|
||||
- Tesseract 4.00 is now supported on an experimental basis.
|
||||
|
||||
- A new rendering option `--pdf-renderer tess4` exploits Tesseract
|
||||
4's new text-only output PDF mode. See the documentation on PDF
|
||||
Renderers for details.
|
||||
- The `--tesseract-oem` argument allows control over the Tesseract
|
||||
4 OCR engine mode (tesseract's `--oem`). Use
|
||||
`--tesseract-oem 2` to enforce the new LSTM mode.
|
||||
- Fixed poor performance with Tesseract 4.00 on Linux
|
||||
|
||||
- Fixed an issue that caused corruption of output to stdout in some
|
||||
cases
|
||||
|
||||
- Removed test for Pillow JPEG and PNG support, as the minimum
|
||||
supported version of Pillow now enforces this
|
||||
|
||||
- OCRmyPDF now tests that the intended destination file is writable
|
||||
before proceeding
|
||||
|
||||
- The test suite now requires `pytest-helpers-namespace` to run (but
|
||||
not install)
|
||||
|
||||
- Significant code reorganization to make OCRmyPDF re-entrant and
|
||||
improve performance. All changes should be backward compatible for
|
||||
the v4.x series.
|
||||
|
||||
- However, OCRmyPDF's dependency "ruffus" is not re-entrant, so no
|
||||
Python API is available. Scripts should continue to use the
|
||||
command line interface.
|
||||
|
||||
## v4.3.5
|
||||
|
||||
- Update documentation to confirm Python 3.6.0 compatibility. No code
|
||||
changes were needed, so many earlier versions are likely supported.
|
||||
|
||||
## v4.3.4
|
||||
|
||||
- Fixed "decimal.InvalidOperation: quantize result has too many digits"
|
||||
for high DPI images
|
||||
|
||||
## v4.3.3
|
||||
|
||||
- Fixed PDF/A creation with Ghostscript 9.20 properly
|
||||
- Fixed an exception on inline stencil masks with a missing optional
|
||||
parameter
|
||||
|
||||
## v4.3.2
|
||||
|
||||
- Fixed a PDF/A creation issue with Ghostscript 9.20 (note: this fix
|
||||
did not actually work)
|
||||
|
||||
## v4.3.1
|
||||
|
||||
- Fixed an issue where pages produced by the "hocr" renderer after a
|
||||
Tesseract timeout would be rotated incorrectly if the input page was
|
||||
rotated with a /Rotate marker
|
||||
- Fixed a file handle leak in LeptonicaErrorTrap that would cause a
|
||||
"too many open files" error for files around hundred pages of pages
|
||||
long when `--deskew` or `--remove-background` or other Leptonica
|
||||
based image processing features were in use, depending on the system
|
||||
value of `ulimit -n`
|
||||
- Ability to specify multiple languages for multilingual documents is
|
||||
now advertised in documentation
|
||||
- Reduced the file sizes of some test resources
|
||||
- Cleaned up debug output
|
||||
- Tesseract caching in test cases is now more cautious about false
|
||||
cache hits and reproducing exact output, not that any problems were
|
||||
observed
|
||||
|
||||
## v4.3
|
||||
|
||||
- New feature `--remove-background` to detect and erase the
|
||||
background of color and grayscale images
|
||||
|
||||
- Better documentation
|
||||
|
||||
- Fixed an issue with PDFs that draw images when the raster stack depth
|
||||
is zero
|
||||
|
||||
- ocrmypdf can now redirect its output to stdout for use in a shell
|
||||
pipeline
|
||||
|
||||
- This does not improve performance since temporary files are still
|
||||
used for buffering
|
||||
- Some output validation is disabled in this mode
|
||||
|
||||
## v4.2.5
|
||||
|
||||
- Fixed an issue
|
||||
({issue}`100`) with
|
||||
PDFs that omit the optional /BitsPerComponent parameter on images
|
||||
- Removed non-free file milk.pdf
|
||||
|
||||
## v4.2.4
|
||||
|
||||
- Fixed an error
|
||||
({issue}`90`) caused by
|
||||
PDFs that use stencil masks properly
|
||||
- Fixed handling of PDFs that try to draw images or stencil masks
|
||||
without properly setting up the graphics state (such images are now
|
||||
ignored for the purposes of calculating DPI)
|
||||
|
||||
## v4.2.3
|
||||
|
||||
- Fixed an issue with PDFs that store page rotation (/Rotate) in an
|
||||
indirect object
|
||||
|
||||
- Integrated a few fixes to simplify downstream packaging (Debian)
|
||||
|
||||
- The test suite no longer assumes it is installed
|
||||
- If running Linux, skip a test that passes Unicode on the command
|
||||
line
|
||||
|
||||
- Added a test case to check explicit masks and stencil masks
|
||||
|
||||
- Added a test case for indirect objects and linearized PDFs
|
||||
|
||||
- Deprecated the OCRmyPDF.sh shell script
|
||||
|
||||
## v4.2.2
|
||||
|
||||
- Improvements to documentation
|
||||
|
||||
## v4.2.1
|
||||
|
||||
- Fixed an issue where PDF pages that contained stencil masks would
|
||||
report an incorrect DPI and cause Ghostscript to abort
|
||||
- Implemented stdin streaming
|
||||
|
||||
## v4.2
|
||||
|
||||
- ocrmypdf will now try to convert single image files to PDFs if they
|
||||
are provided as input
|
||||
({issue}`15`)
|
||||
|
||||
- This is a basic convenience feature. It only supports a single
|
||||
image and always makes the image fill the whole page.
|
||||
- For better control over image to PDF conversion, use `img2pdf`
|
||||
(one of ocrmypdf's dependencies)
|
||||
|
||||
- New argument `--output-type {pdf|pdfa}` allows disabling
|
||||
Ghostscript PDF/A generation
|
||||
|
||||
- `pdfa` is the default, consistent with past behavior
|
||||
- `pdf` provides a workaround for users concerned about the
|
||||
increase in file size from Ghostscript forcing JBIG2 images to
|
||||
CCITT and transcoding JPEGs
|
||||
- `pdf` preserves as much as it can about the original file,
|
||||
including problems that PDF/A conversion fixes
|
||||
|
||||
- PDFs containing images with "non-square" pixel aspect ratios, such as
|
||||
200x100 DPI, are now handled and converted properly (fixing a bug
|
||||
that caused to be cropped)
|
||||
|
||||
- `--force-ocr` rasterizes pages even if they contain no images
|
||||
|
||||
- supports users who want to use OCRmyPDF to reconstruct text
|
||||
information in PDFs with damaged Unicode maps (copy and paste text
|
||||
does not match displayed text)
|
||||
- supports reinterpreting PDFs where text was rendered as curves for
|
||||
printing, and text needs to be recovered
|
||||
- fixes issue
|
||||
{issue}`82`
|
||||
|
||||
- Fixes an issue where, with certain settings, monochrome images in
|
||||
PDFs would be converted to 8-bit grayscale, increasing file size
|
||||
({issue}`79`)
|
||||
|
||||
- Support for Ubuntu 12.04 LTS "precise" has been dropped in favor of
|
||||
(roughly) Ubuntu 14.04 LTS "trusty"
|
||||
|
||||
- Some Ubuntu "PPAs" (backports) are needed to make it work
|
||||
|
||||
- Support for some older dependencies dropped
|
||||
|
||||
- Ghostscript 9.15 or later is now required (available in Ubuntu
|
||||
trusty with backports)
|
||||
- Tesseract 3.03 or later is now required (available in Ubuntu
|
||||
trusty)
|
||||
|
||||
- Ghostscript now runs in "safer" mode where possible
|
||||
|
||||
## v4.1.4
|
||||
|
||||
- Bug fix: monochrome images with an ICC profile attached were
|
||||
incorrectly converted to full color images if lossless reconstruction
|
||||
was not possible due to other settings; consequence was increased
|
||||
file size for these images
|
||||
|
||||
## v4.1.3
|
||||
|
||||
- More helpful error message for PDFs with version 4 security handler
|
||||
- Update usage instructions for Windows/Docker users
|
||||
- Fixed order of operations for matrix multiplication (no effect on most
|
||||
users)
|
||||
- Add a few leptonica wrapper functions (no effect on most users)
|
||||
|
||||
## v4.1.2
|
||||
|
||||
- Replace IEC sRGB ICC profile with Debian's sRGB (from
|
||||
icc-profiles-free) which is more compatible with the MIT license
|
||||
- More helpful error message for an error related to certain types of
|
||||
malformed PDFs
|
||||
|
||||
## v4.1
|
||||
|
||||
- `--rotate-pages` now only rotates pages when reasonably confidence
|
||||
in the orientation. This behavior can be adjusted with the new
|
||||
argument `--rotate-pages-threshold`
|
||||
- Fixed problems in error checking if `unpaper` is uninstalled or
|
||||
missing at run-time
|
||||
- Fixed problems with "RethrownJobError" errors during error handling
|
||||
that suppressed the useful error messages
|
||||
|
||||
## v4.0.7
|
||||
|
||||
- Minor correction to Ghostscript output settings
|
||||
|
||||
## v4.0.6
|
||||
|
||||
- Update install instructions
|
||||
- Provide a sRGB profile instead of using Ghostscript's
|
||||
|
||||
## v4.0.5
|
||||
|
||||
- Remove some verbose debug messages from v4.0.4
|
||||
- Fixed temporary that wasn't being deleted
|
||||
- DPI is now calculated correctly for cropped images, along with other
|
||||
image transformations
|
||||
- Inline images are now checked during DPI calculation instead of
|
||||
rejecting the image
|
||||
|
||||
## v4.0.4
|
||||
|
||||
Released with verbose debug message turned on. Do not use. Skip to
|
||||
v4.0.5.
|
||||
|
||||
## v4.0.3
|
||||
|
||||
New features
|
||||
|
||||
- Page orientations detected are now reported in a summary comment
|
||||
|
||||
Fixes
|
||||
|
||||
- Show stack trace if unexpected errors occur
|
||||
- Treat "too few characters" error message from Tesseract as a reason
|
||||
to skip that page rather than abort the file
|
||||
- Docker: fix blank JPEG2000 issue by insisting on Ghostscript versions
|
||||
that have this fixed
|
||||
|
||||
## v4.0.2
|
||||
|
||||
Fixes
|
||||
|
||||
- Fixed compatibility with Tesseract 3.04.01 release, particularly its
|
||||
different way of outputting orientation information
|
||||
- Improved handling of Tesseract errors and crashes
|
||||
- Fixed use of chmod on Docker that broke most test cases
|
||||
|
||||
## v4.0.1
|
||||
|
||||
Fixes
|
||||
|
||||
- Fixed a KeyError if tesseract fails to find page orientation
|
||||
information
|
||||
|
||||
## v4.0
|
||||
|
||||
New features
|
||||
|
||||
- Automatic page rotation (`-r`) is now available. It uses ignores
|
||||
any prior rotation information on PDFs and sets rotation based on the
|
||||
dominant orientation of detectable text. This feature is fairly
|
||||
reliable but some false positives occur especially if there is not
|
||||
much text to work with.
|
||||
({issue}`4`)
|
||||
- Deskewing is now performed using Leptonica instead of unpaper.
|
||||
Leptonica is faster and more reliable at image deskewing than
|
||||
unpaper.
|
||||
|
||||
Fixes
|
||||
|
||||
- Fixed an issue where lossless reconstruction could cause some pages
|
||||
to be appear incorrectly if the page was rotated by the user in
|
||||
Acrobat after being scanned (specifically if it a /Rotate tag)
|
||||
- Fixed an issue where lossless reconstruction could misalign the
|
||||
graphics layer with respect to text layer if the page had been
|
||||
cropped such that its origin is not (0, 0)
|
||||
({issue}`49`)
|
||||
|
||||
Changes
|
||||
|
||||
- Logging output is now much easier to read
|
||||
- `--deskew` is now performed by Leptonica instead of unpaper
|
||||
({issue}`25`)
|
||||
- libffi is now required
|
||||
- Some changes were made to the Docker and Travis build environments to
|
||||
support libffi
|
||||
- `--pdf-renderer=tesseract` now displays a warning if the Tesseract
|
||||
version is less than 3.04.01, the planned release that will include
|
||||
fixes to an important OCR text rendering bug in Tesseract 3.04.00.
|
||||
You can also manually install ./share/sharp2.ttf on top of pdf.ttf in
|
||||
your Tesseract tessdata folder to correct the problem.
|
||||
|
||||
@@ -0,0 +1,210 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v5
|
||||
|
||||
## v5.7.0
|
||||
|
||||
- Fixed an issue that caused poor CPU utilization on machines with more
|
||||
than 4 cores when running Tesseract 4. (Related to {issue}`217`.)
|
||||
|
||||
- The 'hocr' renderer has been improved. The 'sandwich' and 'tesseract'
|
||||
renderers are still better for most use cases, but 'hocr' may be
|
||||
useful for people who work with the PDF.js renderer in English/ASCII
|
||||
languages. ({issue}`225`)
|
||||
|
||||
- It now formats text in a matter that is easier for certain PDF
|
||||
viewers to select and extract copy and paste text. This should
|
||||
help macOS Preview and PDF.js in particular.
|
||||
- The appearance of selected text and behavior of selecting text is
|
||||
improved.
|
||||
- The PDF content stream now uses relative moves, making it more
|
||||
compact and easier for viewers to determine when two words on the
|
||||
same line.
|
||||
- It can now deal with text on a skewed baseline.
|
||||
- Thanks to @cforcey for the pull request, @jbreiden for many
|
||||
helpful suggestions, @ctbarbour for another round of improvements,
|
||||
and @acaloiaro for an independent review.
|
||||
|
||||
## v5.6.3
|
||||
|
||||
- Suppress two debug messages that were too verbose
|
||||
|
||||
## v5.6.2
|
||||
|
||||
- Development branch accidentally tagged as release. Do not use.
|
||||
|
||||
## v5.6.1
|
||||
|
||||
- Fixed {issue}`219`: change
|
||||
how the final output file is created to avoid triggering permission
|
||||
errors when the output is a special file such as `/dev/null`
|
||||
- Fixed test suite failures due to a qpdf 8.0.0 regression and Python
|
||||
3.5's handling of symlink
|
||||
- The "encrypted PDF" error message was different depending on the type
|
||||
of PDF encryption. Now a single clear message appears for all types
|
||||
of PDF encryption.
|
||||
- ocrmypdf is now in Homebrew. Homebrew users are advised to the
|
||||
version of ocrmypdf in the official homebrew-core formulas rather
|
||||
than the private tap.
|
||||
- Some linting
|
||||
|
||||
## v5.6.0
|
||||
|
||||
- Fixed {issue}`216`: preserve
|
||||
"text as curves" PDFs without rasterizing file
|
||||
- Related to the above, messages about rasterizing are more consistent
|
||||
- For consistency versions minor releases will now get the trailing .0
|
||||
they always should have had.
|
||||
|
||||
## v5.5
|
||||
|
||||
- Add new argument `--max-image-mpixels`. Pillow 5.0 now raises an
|
||||
exception when images may be decompression bombs. This argument can
|
||||
be used to override the limit Pillow sets.
|
||||
- Fixed output page cropped when using the sandwich renderer and OCR is
|
||||
skipped on a rotated and image-processed page
|
||||
- A warning is now issued when old versions of Ghostscript are used in
|
||||
cases known to cause issues with non-Latin characters
|
||||
- Fixed a few parameter validation checks for `-output-type pdfa-1` and
|
||||
`pdfa-2`
|
||||
|
||||
## v5.4.4
|
||||
|
||||
- Fixed {issue}`181`: fix
|
||||
final merge failure for PDFs with more pages than the system file
|
||||
handle limit (`ulimit -n`)
|
||||
- Fixed {issue}`200`: an
|
||||
uncommon syntax for formatting decimal numbers in a PDF would cause
|
||||
qpdf to issue a warning, which ocrmypdf treated as an error. Now this
|
||||
the warning is relayed.
|
||||
- Fixed an issue where intermediate PDFs would be created at version 1.3
|
||||
instead of the version of the original file. It's possible but
|
||||
unlikely this had side effects.
|
||||
- A warning is now issued when older versions of qpdf are used since
|
||||
issues like
|
||||
{issue}`200` cause
|
||||
qpdf to infinite-loop
|
||||
- Address issue
|
||||
{issue}`140`: if
|
||||
Tesseract outputs invalid UTF-8, escape it and print its message
|
||||
instead of aborting with a Unicode error
|
||||
- Adding previously unlisted setup requirement, pytest-runner
|
||||
- Update documentation: fix an error in the example script for Synology
|
||||
with Docker images, improved security guidance, advised
|
||||
`pip install --user`
|
||||
|
||||
## v5.4.3
|
||||
|
||||
- If a subprocess fails to report its version when queried, exit
|
||||
cleanly with an error instead of throwing an exception
|
||||
- Added test to confirm that the system locale is Unicode-aware and
|
||||
fail early if it's not
|
||||
- Clarified some copyright information
|
||||
- Updated pinned requirements.txt so the homebrew formula captures more
|
||||
recent versions
|
||||
|
||||
## v5.4.2
|
||||
|
||||
- Fixed a regression from v5.4.1 that caused sidecar files to be
|
||||
created as empty files
|
||||
|
||||
## v5.4.1
|
||||
|
||||
- Add workaround for Tesseract v4.00alpha crash when trying to obtain
|
||||
orientation and the latest language packs are installed
|
||||
|
||||
## v5.4
|
||||
|
||||
- Change wording of a deprecation warning to improve clarity
|
||||
- Added option to generate PDF/A-1b output if desired
|
||||
(`--output-type pdfa-1`); default remains PDF/A-2b generation
|
||||
- Update documentation
|
||||
|
||||
## v5.3.3
|
||||
|
||||
- Fixed missing error message that should occur when trying to force
|
||||
`--pdf-renderer sandwich` on old versions of Tesseract
|
||||
- Update copyright information in test files
|
||||
- Set system `LANG` to UTF-8 in Dockerfiles to avoid UTF-8 encoding
|
||||
errors
|
||||
|
||||
## v5.3.2
|
||||
|
||||
- Fixed a broken test case related to language packs
|
||||
|
||||
## v5.3.1
|
||||
|
||||
- Fixed wrong return code given for missing Tesseract language packs
|
||||
- Fixed "brew audit" crashing on Travis when trying to auto-brew
|
||||
|
||||
## v5.3
|
||||
|
||||
- Added `--user-words` and `--user-patterns` arguments which are
|
||||
forwarded to Tesseract OCR as words and regular expressions
|
||||
respective to use to guide OCR. Supplying a list of subject-domain
|
||||
words should assist Tesseract with resolving words.
|
||||
({issue}`165`)
|
||||
- Using a non Latin-1 language with the "hocr" renderer now warns about
|
||||
possible OCR quality and recommends workarounds
|
||||
({issue}`176`)
|
||||
- Output file path added to error message when that location is not
|
||||
writable
|
||||
({issue}`175`)
|
||||
- Otherwise valid PDFs with leading whitespace at the beginning of the
|
||||
file are now accepted
|
||||
|
||||
## v5.2
|
||||
|
||||
- When using Tesseract 3.05.01 or newer, OCRmyPDF will select the
|
||||
"sandwich" PDF renderer by default, unless another PDF renderer is
|
||||
specified with the `--pdf-renderer` argument. The previous behavior
|
||||
was to select `--pdf-renderer=hocr`.
|
||||
- The "tesseract" PDF renderer is now deprecated, since it can cause
|
||||
problems with Ghostscript on Tesseract 3.05.00
|
||||
- The "tess4" PDF renderer has been renamed to "sandwich". "tess4" is
|
||||
now a deprecated alias for "sandwich".
|
||||
|
||||
## v5.1
|
||||
|
||||
- Files with pages larger than 200" (5080 mm) in either dimension are
|
||||
now supported with `--output-type=pdf` with the page size preserved
|
||||
(in the PDF specification this feature is called UserUnit scaling).
|
||||
Due to Ghostscript limitations this is not available in conjunction
|
||||
with PDF/A output.
|
||||
|
||||
## v5.0.1
|
||||
|
||||
- Fixed {issue}`169`,
|
||||
exception due to failure to create sidecar text files on some
|
||||
versions of Tesseract 3.04, including the jbarlow83/ocrmypdf Docker
|
||||
image
|
||||
|
||||
## v5.0
|
||||
|
||||
- Backward incompatible changes
|
||||
|
||||
> - Support for Python 3.4 dropped. Python 3.5 is now required.
|
||||
> - Support for Tesseract 3.02 and 3.03 dropped. Tesseract 3.04 or
|
||||
> newer is required. Tesseract 4.00 (alpha) is supported.
|
||||
> - The OCRmyPDF.sh script was removed.
|
||||
|
||||
- Add a new feature, `--sidecar`, which allows creating "sidecar"
|
||||
text files which contain the OCR results in plain text. These OCR
|
||||
text is more reliable than extracting text from PDFs. Closes
|
||||
{issue}`126`.
|
||||
|
||||
- New feature: `--pdfa-image-compression`, which allows overriding
|
||||
Ghostscript's lossy-or-lossless image encoding heuristic and making
|
||||
all images JPEG encoded or lossless encoded as desired. Fixes
|
||||
{issue}`163`.
|
||||
|
||||
- Fixed {issue}`143`, added
|
||||
`--quiet` to suppress "INFO" messages
|
||||
|
||||
- Fixed {issue}`164`, a typo
|
||||
|
||||
- Removed the command line parameters `-n` and `--just-print` since
|
||||
they have not worked for some time (reported as Ubuntu bug
|
||||
[#1687308](https://bugs.launchpad.net/ubuntu/+source/ocrmypdf/+bug/1687308))
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v6
|
||||
|
||||
## v6.2.5
|
||||
|
||||
- Disable a failing test due to Tesseract 4.0rc1 behavior change.
|
||||
Previously, Tesseract would exit with an error message if its
|
||||
configuration was invalid, and OCRmyPDF would intercept this message.
|
||||
Now Tesseract issues a warning, which OCRmyPDF v6.2.5 may relay or
|
||||
ignore. (In v7.x, OCRmyPDF will respond to the warning.)
|
||||
- This release branch no longer supports using the optional PyMuPDF
|
||||
installation, since it was removed in v7.x.
|
||||
- This release branch no longer supports macOS. macOS users should
|
||||
upgrade to v7.x.
|
||||
|
||||
## v6.2.4
|
||||
|
||||
- Backport Ghostscript 9.25 compatibility fixes, which removes support
|
||||
for setting Unicode metadata
|
||||
- Backport blacklisting Ghostscript 9.24
|
||||
- Older versions of Ghostscript are still supported
|
||||
|
||||
## v6.2.3
|
||||
|
||||
- Fixed compatibility with img2pdf >= 0.3.0 by rejecting input images
|
||||
that have an alpha channel
|
||||
- This version will be included in Ubuntu 18.10
|
||||
|
||||
## v6.2.2
|
||||
|
||||
- Backport compatibility fixes for Python 3.7 and ruffus 2.7.0 from
|
||||
v7.0.0
|
||||
- Backport fix to ignore masks when deciding what colors are on a page
|
||||
- Backport some minor improvements from v7.0.0: better argument
|
||||
validation and warnings about the Tesseract 4.0.0 `--user-words`
|
||||
regression
|
||||
|
||||
## v6.2.1
|
||||
|
||||
- Fixed recent versions of Tesseract (after 4.0.0-beta1) not being
|
||||
detected as supporting the `sandwich` renderer ({issue}`271`).
|
||||
|
||||
## v6.2.0
|
||||
|
||||
- **Docker**: The Docker image `ocrmypdf-tess4` has been removed. The
|
||||
main Docker images, `ocrmypdf` and `ocrmypdf-polyglot` now use
|
||||
Ubuntu 18.04 as a base image, and as such Tesseract 4.0.0-beta1 is
|
||||
now the Tesseract version they use. There is no Docker image based on
|
||||
Tesseract 3.05 anymore.
|
||||
- Creation of PDF/A-3 is now supported. However, there is no ability to
|
||||
attach files to PDF/A-3.
|
||||
- Lists more reasons why the file size might grow.
|
||||
- Fixed {issue}`262`,
|
||||
`--remove-background` error on PDFs contained colormapped
|
||||
(paletted) images.
|
||||
- Fixed another XMP metadata validation issue, in cases where the input
|
||||
file's creation date has no timezone and the creation date is not
|
||||
overridden.
|
||||
|
||||
## v6.1.5
|
||||
|
||||
- Fixed {issue}`253`, a
|
||||
possible division by zero when using the `hocr` renderer.
|
||||
- Fixed incorrectly formatted `<xmp:ModifyDate>` field inside XMP
|
||||
metadata for PDF/As. veraPDF flags this as a PDF/A validation
|
||||
failure. The error is caused the timezone and final digit of the
|
||||
seconds of modified time to be omitted, so at worst the modification
|
||||
time stamp is rounded to the nearest 10 seconds.
|
||||
|
||||
## v6.1.4
|
||||
|
||||
- Fixed {issue}`248`
|
||||
`--clean` argument may remove OCR from left column of text on
|
||||
certain documents. We now set `--layout none` to suppress this.
|
||||
- The test cache was updated to reflect the change above.
|
||||
- Change test suite to accommodate Ghostscript 9.23's new ability to
|
||||
insert JPEGs into PDFs without transcoding.
|
||||
- XMP metadata in PDFs is now examined using `defusedxml` for safety.
|
||||
- If an external process exits with a signal when asked to report its
|
||||
version, we now print the system error message instead of suppressing
|
||||
it. This occurred when the required executable was found but was
|
||||
missing a shared library.
|
||||
- qpdf 7.0.0 or newer is now required as the test suite can no longer
|
||||
pass without it.
|
||||
|
||||
### Notes
|
||||
|
||||
- An apparent [regression in Ghostscript
|
||||
9.23](https://bugs.ghostscript.com/show_bug.cgi?id=699216) will
|
||||
cause some ocrmypdf output files to become invalid in rare cases; the
|
||||
workaround for the moment is to set `--force-ocr`.
|
||||
|
||||
## v6.1.3
|
||||
|
||||
- Fixed {issue}`247`,
|
||||
`/CreationDate` metadata not copied from input to output.
|
||||
- A warning is now issued when Python 3.5 is used on files with a large
|
||||
page count, as this case is known to regress to single core
|
||||
performance. The cause of this problem is unknown.
|
||||
|
||||
## v6.1.2
|
||||
|
||||
- Upgrade to PyMuPDF v1.12.5 which includes a more complete fix to
|
||||
{issue}`239`.
|
||||
- Add `defusedxml` dependency.
|
||||
|
||||
## v6.1.1
|
||||
|
||||
- Fixed text being reported as found on all pages if PyMuPDF is not
|
||||
installed.
|
||||
|
||||
## v6.1.0
|
||||
|
||||
- PyMuPDF is now an optional but recommended dependency, to alleviate
|
||||
installation difficulties on platforms that have less access to
|
||||
PyMuPDF than the author anticipated. (For version 6.x only) install
|
||||
OCRmyPDF with `pip install ocrmypdf[fitz]` to use it to its full
|
||||
potential.
|
||||
- Fixed `FileExistsError` that could occur if OCR timed out while it
|
||||
was generating the output file.
|
||||
({issue}`218`)
|
||||
- Fixed table of contents/bookmarks all being redirected to page 1 when
|
||||
generating a PDF/A (with PyMuPDF). (Without PyMuPDF the table of
|
||||
contents is removed in PDF/A mode.)
|
||||
- Fixed "RuntimeError: invalid key in dict" when table of
|
||||
contents/bookmarks titles contained the character `)`.
|
||||
({issue}`239`)
|
||||
- Added a new argument `--skip-repair` to skip the initial PDF repair
|
||||
step if the PDF is already well-formed (because another program
|
||||
repaired it).
|
||||
|
||||
## v6.0.0
|
||||
|
||||
- The software license has been changed to GPLv3 [it has since changed again].
|
||||
Test resource files and some individual sources may have other licenses.
|
||||
|
||||
- OCRmyPDF now depends on
|
||||
[PyMuPDF](https://pymupdf.readthedocs.io/en/latest/installation/).
|
||||
Including PyMuPDF is the primary reason for the change to GPLv3.
|
||||
|
||||
- Other backward incompatible changes
|
||||
|
||||
- The `OCRMYPDF_TESSERACT`, `OCRMYPDF_QPDF`, `OCRMYPDF_GS` and
|
||||
`OCRMYPDF_UNPAPER` environment variables are no longer used.
|
||||
Change `PATH` if you need to override the external programs
|
||||
OCRmyPDF uses.
|
||||
- The `ocrmypdf` package has been moved to `src/ocrmypdf` to
|
||||
avoid issues with accidental import.
|
||||
- The function `ocrmypdf.exec.get_program` was removed.
|
||||
- The deprecated module `ocrmypdf.pageinfo` was removed.
|
||||
- The `--pdf-renderer tess4` alias for `sandwich` was removed.
|
||||
|
||||
- Fixed an issue where OCRmyPDF failed to detect existing text on
|
||||
pages, depending on how the text and fonts were encoded within the
|
||||
PDF. ({issue}`233,232`)
|
||||
|
||||
- Fixed an issue that caused dramatic inflation of file sizes when
|
||||
`--skip-text --output-type pdf` was used. OCRmyPDF now removes
|
||||
duplicate resources such as fonts, images and other objects that it
|
||||
generates. ({issue}`237`)
|
||||
|
||||
- Improved performance of the initial page splitting step. Originally
|
||||
this step was not believed to be expensive and ran in a process.
|
||||
Large file testing revealed it to be a bottleneck, so it is now
|
||||
parallelized. On a 700 page file with quad core machine, this change
|
||||
saves about 2 minutes. ({issue}`234`)
|
||||
|
||||
- The test suite now includes a cache that can be used to speed up test
|
||||
runs across platforms. This also does not require computing
|
||||
checksums, so it's faster. ({issue}`217`)
|
||||
|
||||
@@ -0,0 +1,288 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v7
|
||||
|
||||
## v7.4.0
|
||||
|
||||
- `--force-ocr` may now be used with the new `--threshold` and
|
||||
`--mask-barcodes` features
|
||||
- pikepdf >= 0.9.1 is now required.
|
||||
- Changed metadata handling to pikepdf 0.9.1. As a result, metadata
|
||||
handling of non-ASCII characters in Ghostscript 9.25 or later is
|
||||
fixed.
|
||||
- chardet >= 3.0.4 is temporarily listed as required. pdfminer.six
|
||||
depends on it, but the most recent release does not specify this
|
||||
requirement.
|
||||
({issue}`326`)
|
||||
- python-xmp-toolkit and libexempi are no longer required.
|
||||
- A new Docker image is now being provided for users who wish to access
|
||||
OCRmyPDF over a simple HTTP interface, instead of the command line.
|
||||
- Increase tolerance of PDFs that overflow or underflow the PDF
|
||||
graphics stack.
|
||||
({issue}`325`)
|
||||
|
||||
## v7.3.1
|
||||
|
||||
- Fixed performance regression from v7.3.0; fast page analysis was not
|
||||
selected when it should be.
|
||||
- Fixed a few exceptions related to the new `--mask-barcodes` feature
|
||||
and improved argument checking
|
||||
- Added missing detection of TrueType fonts that lack a Unicode mapping
|
||||
|
||||
## v7.3.0
|
||||
|
||||
- Added a new feature `--redo-ocr` to detect existing OCR in a file,
|
||||
remove it, and redo the OCR. This may be particularly helpful for
|
||||
anyone who wants to take advantage of OCR quality improvements in
|
||||
Tesseract 4.0. Note that OCR added by OCRmyPDF before version 3.0
|
||||
cannot be detected since it was not properly marked as invisible text
|
||||
in the earliest versions. OCR that constructs a font from visible
|
||||
text, such as Adobe Acrobat's ClearScan.
|
||||
|
||||
- OCRmyPDF's content detection is generally more sophisticated. It
|
||||
learns more about the contents of each PDF and makes better
|
||||
recommendations:
|
||||
|
||||
- OCRmyPDF can now detect when a PDF contains text that cannot be
|
||||
mapped to Unicode (meaning it is readable to human eyes but
|
||||
copy-pastes as gibberish). In these cases it recommends
|
||||
`--force-ocr` to make the text searchable.
|
||||
- PDFs containing vector objects are now rendered at more
|
||||
appropriate resolution for OCR.
|
||||
- We now exit with an error for PDFs that contain Adobe LiveCycle
|
||||
Designer's dynamic XFA forms. Currently the open source community
|
||||
does not have tools to work with these files.
|
||||
- OCRmyPDF now warns when a PDF that contains Adobe AcroForms, since
|
||||
such files probably do not need OCR. It can work with these files.
|
||||
|
||||
- Added three new **experimental** features to improve OCR quality in
|
||||
certain conditions. The name, syntax and behavior of these arguments
|
||||
is subject to change. They may also be incompatible with some other
|
||||
features.
|
||||
|
||||
- `--remove-vectors` which strips out vector graphics. This can
|
||||
improve OCR quality since OCR will not search artwork for readable
|
||||
text; however, it currently removes "text as curves" as well.
|
||||
- `--mask-barcodes` to detect and suppress barcodes in files. We
|
||||
have observed that barcodes can interfere with OCR because they
|
||||
are "text-like" but not actually textual.
|
||||
- `--threshold` which uses a more sophisticated thresholding
|
||||
algorithm than is currently in use in Tesseract OCR. This works
|
||||
around a [known issue in Tesseract
|
||||
4.0](https://github.com/tesseract-ocr/tesseract/issues/1990)
|
||||
with dark text on bright backgrounds.
|
||||
|
||||
- Fixed an issue where an error message was not reported when the
|
||||
installed Ghostscript was very old.
|
||||
|
||||
- The PDF optimizer now saves files with object streams enabled when
|
||||
the optimization level is `--optimize 1` or higher (the default).
|
||||
This makes files a little bit smaller, but requires PDF 1.5. PDF 1.5
|
||||
was first released in 2003 and is broadly supported by PDF viewers,
|
||||
but some rudimentary PDF parsers such as PyPDF2 do not understand
|
||||
object streams. You can use the command line tool
|
||||
`qpdf --object-streams=disable` or
|
||||
[pikepdf](https://github.com/pikepdf/pikepdf) library to remove
|
||||
them.
|
||||
|
||||
- New dependency: pdfminer.six 20181108. Note this is a fork of the
|
||||
Python 2-only pdfminer.
|
||||
|
||||
- Deprecation notice: At the end of 2018, we will be ending support for
|
||||
Python 3.5 and Tesseract 3.x. OCRmyPDF v7 will continue to work with
|
||||
older versions.
|
||||
|
||||
## v7.2.1
|
||||
|
||||
- Fixed compatibility with an API change in pikepdf 0.3.5.
|
||||
- A kludge to support Leptonica versions older than 1.72 in the test
|
||||
suite was dropped. Older versions of Leptonica are likely still
|
||||
compatible. The only impact is that a portion of the test suite will
|
||||
be skipped.
|
||||
|
||||
## v7.2.0
|
||||
|
||||
**Lossy JBIG2 behavior change**
|
||||
|
||||
A user reported that ocrmypdf was in fact using JBIG2 in **lossy**
|
||||
compression mode. This was not the intended behavior. Users should
|
||||
[review the technical concerns with JBIG2 in lossy
|
||||
mode](https://abbyy.technology/en:kb:tip:jbig2_compression_and_ocr)
|
||||
and decide if this is a concern for their use case.
|
||||
|
||||
JBIG2 lossy mode does achieve higher compression ratios than any other
|
||||
monochrome compression technology; for large text documents the savings
|
||||
are considerable. JBIG2 lossless still gives great compression ratios
|
||||
and is a major improvement over the older CCITT G4 standard.
|
||||
|
||||
Only users who have reviewed the concerns with JBIG2 in lossy mode
|
||||
should opt-in. As such, lossy mode JBIG2 is only turned on when the new
|
||||
argument `--jbig2-lossy` is issued. This is independent of the setting
|
||||
for `--optimize`.
|
||||
|
||||
Users who did not install an optional JBIG2 encoder are unaffected.
|
||||
|
||||
(Thanks to user 'bsdice' for reporting this issue.)
|
||||
|
||||
**Other issues**
|
||||
|
||||
- When the image optimizer quantizes an image to 1 bit per pixel, it
|
||||
will now attempt to further optimize that image as CCITT or JBIG2,
|
||||
instead of keeping it in the "flate" encoding which is not efficient
|
||||
for 1 bpp images.
|
||||
({issue}`297`)
|
||||
- Images in PDFs that are used as soft masks (i.e. transparency masks
|
||||
or alpha channels) are now excluded from optimization.
|
||||
- Fixed handling of Tesseract 4.0-rc1 which now accepts invalid
|
||||
Tesseract configuration files, which broke the test suite.
|
||||
|
||||
## v7.1.0
|
||||
|
||||
- Improve the performance of initial text extraction, which is done to
|
||||
determine if a file contains existing text of some kind or not. On
|
||||
large files, this initial processing is now about 20x times faster.
|
||||
({issue}`299`)
|
||||
- pikepdf 0.3.3 is now required.
|
||||
- Fixed {issue}`231`, a
|
||||
problem with JPEG2000 images where image metadata was only available
|
||||
inside the JPEG2000 file.
|
||||
- Fixed some additional Ghostscript 9.25 compatibility issues.
|
||||
- Improved handling of KeyboardInterrupt error messages.
|
||||
({issue}`301`)
|
||||
- README.md is now served in GitHub markdown instead of
|
||||
reStructuredText.
|
||||
|
||||
## v7.0.6
|
||||
|
||||
- Blacklist Ghostscript 9.24, now that 9.25 is available and fixes many
|
||||
regressions in 9.24.
|
||||
|
||||
## v7.0.5
|
||||
|
||||
- Improve capability with Ghostscript 9.24, and enable the JPEG
|
||||
passthrough feature when this version in installed.
|
||||
- Ghostscript 9.24 lost the ability to set PDF title, author, subject
|
||||
and keyword metadata to Unicode strings. OCRmyPDF will set ASCII
|
||||
strings and warn when Unicode is suppressed. Other software may be
|
||||
used to update metadata. This is a short term work around.
|
||||
- PDFs generated by Kodak Capture Desktop, or generally PDFs that
|
||||
contain indirect references to null objects in their table of
|
||||
contents, would have an invalid table of contents after processing by
|
||||
OCRmyPDF that might interfere with other viewers. This has been
|
||||
fixed.
|
||||
- Detect PDFs generated by Adobe LiveCycle, which can only be displayed
|
||||
in Adobe Acrobat and Reader currently. When these are encountered,
|
||||
exit with an error instead of performing OCR on the "Please wait"
|
||||
error message page.
|
||||
|
||||
## v7.0.4
|
||||
|
||||
- Fixed exception thrown when trying to optimize a certain type of PNG
|
||||
embedded in a PDF with the `-O2`
|
||||
- Update to pikepdf 0.3.2, to gain support for optimizing some
|
||||
additional image types that were previously excluded from
|
||||
optimization (CMYK and grayscale). Fixes
|
||||
{issue}`285`.
|
||||
|
||||
## v7.0.3
|
||||
|
||||
- Fixed {issue}`284`, an error
|
||||
when parsing inline images that have are also image masks, by
|
||||
upgrading pikepdf to 0.3.1
|
||||
|
||||
## v7.0.2
|
||||
|
||||
- Fixed a regression with `--rotate-pages` on pages that already had
|
||||
rotations applied.
|
||||
({issue}`279`)
|
||||
- Improve quality of page rotation in some cases by rasterizing a
|
||||
higher quality preview image.
|
||||
({issue}`281`)
|
||||
|
||||
## v7.0.1
|
||||
|
||||
- Fixed compatibility with img2pdf >= 0.3.0 by rejecting input images
|
||||
that have an alpha channel
|
||||
- Add forward compatibility for pikepdf 0.3.0 (unrelated to img2pdf)
|
||||
- Various documentation updates for v7.0.0 changes
|
||||
|
||||
## v7.0.0
|
||||
|
||||
- The core algorithm for combining OCR layers with existing PDF pages
|
||||
has been rewritten and improved considerably. PDFs are no longer
|
||||
split into single page PDFs for processing; instead, images are
|
||||
rendered and the OCR results are grafted onto the input PDF. The new
|
||||
algorithm uses less temporary disk space and is much more performant
|
||||
especially for large files.
|
||||
|
||||
- New dependency: [pikepdf](https://github.com/pikepdf/pikepdf).
|
||||
pikepdf is a powerful new Python PDF library driving the latest
|
||||
OCRmyPDF features, built on the QPDF C++ library (libqpdf).
|
||||
|
||||
- New feature: PDF optimization with `-O` or `--optimize`. After
|
||||
OCR, OCRmyPDF will perform image optimizations relevant to OCR PDFs.
|
||||
|
||||
- If a JBIG2 encoder is available, then monochrome images will be
|
||||
converted, with the potential for huge savings on large black and
|
||||
white images, since JBIG2 is far more efficient than any other
|
||||
monochrome (bi-level) compression. (All known US patents related
|
||||
to JBIG2 have probably expired, but it remains the responsibility
|
||||
of the user to supply a JBIG2 encoder such as
|
||||
[jbig2enc](https://github.com/agl/jbig2enc). OCRmyPDF does not
|
||||
implement JBIG2 encoding.)
|
||||
- If `pngquant` is installed, OCRmyPDF will optionally use it to
|
||||
perform lossy quantization and compression of PNG images.
|
||||
- The quality of JPEGs can also be lowered, on the assumption that a
|
||||
lower quality image may be suitable for storage after OCR.
|
||||
- This image optimization component will eventually be offered as an
|
||||
independent command line utility.
|
||||
- Optimization ranges from `-O0` through `-O3`, where `0`
|
||||
disables optimization and `3` implements all options. `1`, the
|
||||
default, performs only safe and lossless optimizations. (This is
|
||||
similar to GCC's optimization parameter.) The exact type of
|
||||
optimizations performed will vary over time.
|
||||
|
||||
- Small amounts of text in the margins of a page, such as watermarks,
|
||||
page numbers, or digital stamps, will no longer prevent the rest of a
|
||||
page from being OCRed when `--skip-text` is issued. This behavior
|
||||
is based on a heuristic.
|
||||
|
||||
- Removed features
|
||||
|
||||
- The deprecated `--pdf-renderer tesseract` PDF renderer was
|
||||
removed.
|
||||
- `-g`, the option to generate debug text pages, was removed
|
||||
because it was a maintenance burden and only worked in isolated
|
||||
cases. HOCR pages can still be previewed by running the
|
||||
hocrtransform.py with appropriate settings.
|
||||
|
||||
- Removed dependencies
|
||||
|
||||
- `PyPDF2`
|
||||
- `defusedxml`
|
||||
- `PyMuPDF`
|
||||
|
||||
- The `sandwich` PDF renderer can be used with all supported versions
|
||||
of Tesseract, including that those prior to v3.05 which don't support
|
||||
`-c textonly`. (Tesseract v4.0.0 is recommended and more
|
||||
efficient.)
|
||||
|
||||
- `--pdf-renderer auto` option and the diagnostics used to select a
|
||||
PDF renderer now work better with old versions, but may make
|
||||
different decisions than past versions.
|
||||
|
||||
- If everything succeeds but PDF/A conversion fails, a distinct return
|
||||
code is now returned (`ExitCode.pdfa_conversion_failed (10)`) where
|
||||
this situation previously returned
|
||||
`ExitCode.invalid_output_pdf (4)`. The latter is now returned only
|
||||
if there is some indication that the output file is invalid.
|
||||
|
||||
- Notes for downstream packagers
|
||||
|
||||
- There is also a new dependency on `python-xmp-toolkit` which in
|
||||
turn depends on `libexempi3`.
|
||||
- It may be necessary to separately `pip install pycparser` to
|
||||
avoid [another Python 3.7
|
||||
issue](https://github.com/eliben/pycparser/pull/135).
|
||||
|
||||
@@ -0,0 +1,153 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v8
|
||||
|
||||
## v8.3.2
|
||||
|
||||
- Dropped workaround for macOS that allowed it work without pdfminer.six,
|
||||
now a proper sdist release of pdfminer.six is available.
|
||||
- pikepdf 1.5.0 is now required.
|
||||
|
||||
## v8.3.1
|
||||
|
||||
- Fixed an issue where PDFs with malformed metadata would be rendered as
|
||||
blank pages. {issue}`398`.
|
||||
|
||||
## v8.3.0
|
||||
|
||||
- Improved the strategy for updating pages when a new image of the page
|
||||
was produced. We now attempt to preserve more content from the
|
||||
original file, for annotations in particular.
|
||||
- For PDFs with more than 100 pages and a sequence where one PDF page
|
||||
was replaced and one or more subsequent ones were skipped, an
|
||||
intermediate file would be corrupted while grafting OCR text, causing
|
||||
processing to fail. This is a regression, likely introduced in
|
||||
v8.2.4.
|
||||
- Previously, we resized the images produced by Ghostscript by a small
|
||||
number of pixels to ensure the output image size was an exactly what
|
||||
we wanted. Having discovered a way to get Ghostscript to produce the
|
||||
exact image sizes we require, we eliminated the resizing step.
|
||||
- Command line completions for `bash` are now available, in addition
|
||||
to `fish`, both in `misc/completion`. Package maintainers, please
|
||||
install these so users can take advantage.
|
||||
- Updated requirements.
|
||||
- pikepdf 1.3.0 is now required.
|
||||
|
||||
## v8.2.4
|
||||
|
||||
- Fixed a false positive while checking for a certain type of PDF that
|
||||
only Acrobat can read. We now more accurately detect Acrobat-only
|
||||
PDFs.
|
||||
- OCRmyPDF holds fewer open file handles and is more prompt about
|
||||
releasing those it no longer needs.
|
||||
- Minor optimization: we no longer traverse the table of contents to
|
||||
ensure all references in it are resolved, as changes to libqpdf have
|
||||
made this unnecessary.
|
||||
- pikepdf 1.2.0 is now required.
|
||||
|
||||
## v8.2.3
|
||||
|
||||
- Fixed that `--mask-barcodes` would occasionally leave a unwanted
|
||||
temporary file named `junkpixt` in the current working folder.
|
||||
- Fixed (hopefully) handling of Leptonica errors in an environment
|
||||
where a non-standard `sys.stderr` is present.
|
||||
- Improved help text for `--verbose`.
|
||||
|
||||
## v8.2.2
|
||||
|
||||
- Fixed a regression from v8.2.0, an exception that occurred while
|
||||
attempting to report that `unpaper` or another optional dependency
|
||||
was unavailable.
|
||||
- In some cases, `ocrmypdf [-c|--clean]` failed to exit with an error
|
||||
when `unpaper` is not installed.
|
||||
|
||||
## v8.2.1
|
||||
|
||||
- This release was canceled.
|
||||
|
||||
## v8.2.0
|
||||
|
||||
- A major improvement to our Docker image is now available thanks to
|
||||
hard work contributed by @mawi12345. The new Docker image,
|
||||
ocrmypdf-alpine, is based on Alpine Linux, and includes most of the
|
||||
functionality of three existed images in a smaller package. This
|
||||
image will replace the main Docker image eventually but for now all
|
||||
are being built. [See documentation for
|
||||
details](https://ocrmypdf.readthedocs.io/en/latest/docker.html).
|
||||
- Documentation reorganized especially around the use of Docker images.
|
||||
- Fixed a problem with PDF image optimization, where the optimizer
|
||||
would unnecessarily decompress and recompress PNG images, in some
|
||||
cases losing the benefits of the quantization it just had just
|
||||
performed. The optimizer is now capable of embedding PNG images into
|
||||
PDFs without transcoding them.
|
||||
- Fixed a minor regression with lossy JBIG2 image optimization. All
|
||||
JBIG2 candidates images were incorrectly placed into a single
|
||||
optimization group for the whole file, instead of grouping pages
|
||||
together. This usually makes a larger JBIG2Globals dictionary and
|
||||
results in inferior compression, so it worked less well than
|
||||
designed. However, quality would not be impacted. Lossless JBIG2 was
|
||||
entirely unaffected.
|
||||
- Updated dependencies, including pikepdf to 1.1.0. This fixes
|
||||
{issue}`358`.
|
||||
- The install-time version checks for certain external programs have
|
||||
been removed from setup.py. These tests are now performed at
|
||||
run-time.
|
||||
- The non-standard option to override install-time checks
|
||||
(`setup.py install --force`) is now deprecated and prints a
|
||||
warning. It will be removed in a future release.
|
||||
|
||||
## v8.1.0
|
||||
|
||||
- Added a feature, `--unpaper-args`, which allows passing arbitrary
|
||||
arguments to `unpaper` when using `--clean` or `--clean-final`.
|
||||
The default, very conservative unpaper settings are suppressed.
|
||||
- The argument `--clean-final` now implies `--clean`. It was
|
||||
possible to issue `--clean-final` on its before this, but it would
|
||||
have no useful effect.
|
||||
- Fixed an exception on traversing corrupt table of contents entries
|
||||
(specifically, those with invalid destination objects)
|
||||
- Fixed an issue when using `--tesseract-timeout` and image
|
||||
processing features on a file with more than 100 pages.
|
||||
{issue}`347`
|
||||
- OCRmyPDF now always calls `os.nice(5)` to signal to operating
|
||||
systems that it is a background process.
|
||||
|
||||
## v8.0.1
|
||||
|
||||
- Fixed an exception when parsing PDFs that are missing a required
|
||||
field. {issue}`325`
|
||||
- pikepdf 1.0.5 is now required, to address some other PDF parsing
|
||||
issues.
|
||||
|
||||
## v8.0.0
|
||||
|
||||
No major features. The intent of this release is to sever support for
|
||||
older versions of certain dependencies.
|
||||
|
||||
**Breaking changes**
|
||||
|
||||
- Dropped support for Tesseract 3.x. Tesseract 4.0 or newer is now
|
||||
required.
|
||||
- Dropped support for Python 3.5.
|
||||
- Some `ocrmypdf.pdfa` APIs that were deprecated in v7.x were
|
||||
removed. This functionality has been moved to pikepdf.
|
||||
|
||||
**Other changes**
|
||||
|
||||
- Fixed an unhandled exception when attempting to mask barcodes.
|
||||
{issue}`322`
|
||||
- It is now possible to use ocrmypdf without pdfminer.six, to support
|
||||
distributions that do not have it or cannot currently use it (e.g.
|
||||
Homebrew). Downstream maintainers should include pdfminer.six if
|
||||
possible.
|
||||
- A warning is now issue when PDF/A conversion removes some XMP
|
||||
metadata from the input PDF. (Only a "whitelist" of certain XMP
|
||||
metadata types are allowed in PDF/A.)
|
||||
- Fixed several issues that caused PDF/As to be produced with
|
||||
nonconforming XMP metadata (would fail validation with veraPDF).
|
||||
- Fixed some instances where invalid DocumentInfo from a PDF cause XMP
|
||||
metadata creation to fail.
|
||||
- Fixed a few documentation problems.
|
||||
- pikepdf 1.0.2 is now required.
|
||||
|
||||
@@ -0,0 +1,252 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v9
|
||||
|
||||
## v9.8.2
|
||||
|
||||
- Fixed an issue where OCRmyPDF would ignore text inside Form XObject when
|
||||
making certain decisions about whether a document already had text.
|
||||
- Fixed file size increase warning to take overhead of small files into account.
|
||||
- Added instructions for installing on Cygwin.
|
||||
|
||||
## v9.8.1
|
||||
|
||||
- Fixed an issue where unexpected files in the `%PROGRAMFILES%\gs` directory
|
||||
(Windows) caused an exception.
|
||||
- Mark pdfminer.six 20200517 as supported.
|
||||
- If jbig2enc is missing and optimization is requested, a warning is issued
|
||||
instead of an error, which was the intended behavior.
|
||||
- Documentation updates.
|
||||
|
||||
## v9.8.0
|
||||
|
||||
- Fixed issue where only the first PNG (FlateDecode) image in a file would be
|
||||
considered for optimization. File sizes should be improved from here on.
|
||||
- Fixed a startup crash when the chosen language was Japanese ({issue}`543`).
|
||||
- Added options to configure polling and log level to watcher.py.
|
||||
|
||||
## v9.7.2
|
||||
|
||||
- Fixed an issue with `ocrmypdf.ocr(...language=)` not accepting a list of
|
||||
languages as documented.
|
||||
- Updated setup.py to confirm that pdfminer.six version 20200402 is supported.
|
||||
|
||||
## v9.7.1
|
||||
|
||||
- Fixed version check failing when used with qpdf 10.0.0.
|
||||
- Added some missing type annotations.
|
||||
- Updated documentation to warn about need for "ifmain" guard and Windows.
|
||||
|
||||
## v9.7.0
|
||||
|
||||
- Fixed an error in watcher.py if `OCR_JSON_SETTINGS` was not defined.
|
||||
- Ghostscript 9.51 is now blacklisted, due to numerous problems with this version.
|
||||
- Added a workaround for a problem with "txtwrite" in Ghostscript 9.52.
|
||||
- Fixed an issue where the incorrect number of threads used was shown when
|
||||
`OMP_THREAD_LIMIT` was manipulated.
|
||||
- Removed a possible performance bottlenecks for files that use hundreds to
|
||||
thousands of images on the same page.
|
||||
- Documentation improvements.
|
||||
- Optimization will now be applied to some monochrome images that have a color
|
||||
profile defined instead of only black and white.
|
||||
- ICC profiles are consulted when determining the simplified colorspace of an
|
||||
image.
|
||||
|
||||
## v9.6.1
|
||||
|
||||
- Documentation improvements - thanks to many users for their contributions!
|
||||
|
||||
> - Fixed installation instructions for ArchLinux (@pigmonkey)
|
||||
> - Updated installation instructions for FreeBSD and other OSes (@knobix)
|
||||
> - Added instructions for using Docker Compose with watchdog (@ianalexander,
|
||||
> @deisi)
|
||||
> - Other miscellany (@mb720, @toy, @caiofacchinato)
|
||||
> - Some scripts provided in the documentation have been migrated out so that
|
||||
> they can be copied out as whole files, and to ensure syntax checking
|
||||
> is maintained.
|
||||
|
||||
- Fixed an error that caused bash completions to fail on macOS. ({issue}`502,504`;
|
||||
@AlexanderWillner)
|
||||
|
||||
- Fixed a rare case where OCRmyPDF threw an exception while processing a PDF
|
||||
with the wrong object type in its `/Trailer /Info`. The error is now logged
|
||||
and incorrect object is ignored. ({issue}`497`)
|
||||
|
||||
- Removed potentially non-free file `enron1.pdf` and simplified the test that
|
||||
used it.
|
||||
|
||||
- Removed potentially non-free file `misc/media/logo.afdesign`.
|
||||
|
||||
## v9.6.0
|
||||
|
||||
- Fixed a regression with transferring metadata from the input PDF to the output
|
||||
PDF in certain situations.
|
||||
- pdfminer.six is now supported up to version 2020-01-24.
|
||||
- Messages are explaining page rotation decisions are now shown at the standard
|
||||
verbosity level again when `--rotate-pages`. In some previous version they
|
||||
were set to debug level messages that only appeared with the parameter `-v1`.
|
||||
- Improvements to `misc/watcher.py`. Thanks to @ianalexander and @svenihoney.
|
||||
- Documentation improvements.
|
||||
|
||||
## v9.5.0
|
||||
|
||||
- Added API functions to measure OCR quality.
|
||||
- Modest improvements to handling PDFs with difficult/non compliant metadata.
|
||||
|
||||
## v9.4.0
|
||||
|
||||
- Updated recommended dependency versions.
|
||||
- Improvements to test coverage and changes to facilitate better measurement of
|
||||
test coverage, such as when tests run in subprocesses.
|
||||
- Improvements to error messages when Leptonica is not installed correctly.
|
||||
- Fixed use of pytest "session scope" that may have caused some intermittent
|
||||
CI failures.
|
||||
- When the argument `--keep-temporary-files` or verbosity is set to `-v1`,
|
||||
a debug log file is generated in the working temporary folder.
|
||||
|
||||
## v9.3.0
|
||||
|
||||
- Improved native Windows support: we now check in the obvious places in
|
||||
the "Program Files" folders installations of Tesseract and Ghostscript,
|
||||
rather than relying on the user to edit `PATH` to specify their location.
|
||||
The `PATH` environment variable can still be used to differentiate when
|
||||
multiple installations are present or the programs are installed to non-
|
||||
standard locations.
|
||||
- Fixed an exception on parsing Ghostscript error messages.
|
||||
- Added an improved example demonstrating how to set up a watched folder
|
||||
for automated OCR processing (thanks to @ianalexander for the contribution).
|
||||
|
||||
## v9.2.0
|
||||
|
||||
- Native Windows is now supported.
|
||||
- Continuous integration moved to Azure Pipelines.
|
||||
- Improved test coverage and speed of tests.
|
||||
- Fixed an issue where a page that was originally a JPEG would be saved as a
|
||||
PNG, increasing file size. This occurred only when a preprocessing option
|
||||
was selected along with `--output-type=pdf` and all images on the original
|
||||
page were JPEGs. Regression since v7.0.0.
|
||||
- OCRmyPDF no longer depends on the QPDF executable `qpdf` or `libqpdf`.
|
||||
It uses pikepdf (which in turn depends on `libqpdf`). Package maintainers
|
||||
should adjust dependencies so that OCRmyPDF no longer calls for libqpdf on
|
||||
its own. For users of Python binary wheels, this change means a separate
|
||||
installation of QPDF is no longer necessary. This change is mainly to
|
||||
simplify installation on Windows.
|
||||
- Fixed a rare case where log messages from Tesseract would be discarded.
|
||||
- Fixed incorrect function signature for pixFindPageForeground, causing
|
||||
exceptions on certain platforms/Leptonica versions.
|
||||
|
||||
## v9.1.1
|
||||
|
||||
- Expand the range of pdfminer.six versions that are supported.
|
||||
- Fixed Docker build when using pikepdf 1.7.0.
|
||||
- Fixed documentation to recommend using pip from get-pip.py.
|
||||
|
||||
## v9.1.0
|
||||
|
||||
- Improved diagnostics when file size increases at output. Now warns if JBIG2
|
||||
or pngquant were not available.
|
||||
- pikepdf 1.7.0 is now required, to pick up changes that remove the need for
|
||||
a source install on Linux systems running Python 3.8.
|
||||
|
||||
## v9.0.5
|
||||
|
||||
- The Alpine Docker image (jbarlow83/ocrmypdf-alpine) has been dropped due to
|
||||
the difficulties of supporting Alpine Linux.
|
||||
- The primary Docker image (jbarlow83/ocrmypdf) has been improved to take on
|
||||
the extra features that used to be exclusive to the Alpine image.
|
||||
- No changes to application code.
|
||||
- pdfminer.six version 20191020 is now supported.
|
||||
|
||||
## v9.0.4
|
||||
|
||||
- Fixed compatibility with Python 3.8 (but requires source install for the moment).
|
||||
- Fixed Tesseract settings for `--user-words` and `--user-patterns`.
|
||||
- Changed to pikepdf 1.6.5 (for Python 3.8).
|
||||
- Changed to Pillow 6.2.0 (to mitigate a security vulnerability in earlier Pillow).
|
||||
- A debug message now mentions when English is automatically selected if the locale
|
||||
is not English.
|
||||
|
||||
## v9.0.3
|
||||
|
||||
- Embed an encoded version of the sRGB ICC profile in the intermediate
|
||||
Postscript file (used for PDF/A conversion). Previously we included the
|
||||
filename, which required Postscript to run with file access enabled. For
|
||||
security, Ghostscript 9.28 enables `-dSAFER` and as such, no longer
|
||||
permits access to any file by default. This fix is necessary for
|
||||
compatibility with Ghostscript 9.28.
|
||||
- Exclude a test that sometimes times out and fails in continuous integration
|
||||
from the standard test suite.
|
||||
|
||||
## v9.0.2
|
||||
|
||||
- The image optimizer now skips optimizing flate (PNG) encoded images in some
|
||||
situations where the optimization effort was likely wasted.
|
||||
- The image optimizer now ignores images that specify arbitrary decode arrays,
|
||||
since these are rare.
|
||||
- Fixed an issue that caused inversion of black and white in monochrome images.
|
||||
We are not certain but the problem seems to be linked to Leptonica 1.76.0 and
|
||||
older.
|
||||
- Fixed some cases where the test suite failed if
|
||||
English or German Tesseract language packs were not installed.
|
||||
- Fixed a runtime error if the Tesseract English language is not installed.
|
||||
- Improved explicit closing of Pillow images after use.
|
||||
- Actually fixed of Alpine Docker image build.
|
||||
- Changed to pikepdf 1.6.3.
|
||||
|
||||
## v9.0.1
|
||||
|
||||
- Fixed test suite failing when either of optional dependencies unpaper and
|
||||
pngquant were missing.
|
||||
- Attempted fix of Alpine Docker image build.
|
||||
- Documented that FreeBSD ports are now available.
|
||||
- Changed to pikepdf 1.6.1.
|
||||
|
||||
## v9.0.0
|
||||
|
||||
**Breaking changes**
|
||||
|
||||
- The `--mask-barcodes` experimental feature has been dropped due to poor
|
||||
reliability and occasional crashes, both due to the underlying library that
|
||||
implements this feature (Leptonica).
|
||||
- The `-v` (verbosity level) parameter now accepts only `0`, `1`, and
|
||||
`2`.
|
||||
- Dropped support for Tesseract 4.00.00-alpha releases. Tesseract 4.0 beta and
|
||||
later remain supported.
|
||||
- Dropped the `ocrmypdf-polyglot` and `ocrmypdf-webservice` images.
|
||||
|
||||
**New features**
|
||||
|
||||
- Added a high level API for applications that want to integrate OCRmyPDF.
|
||||
Special thanks to Martin Wind (@mawi1988) whose made significant contributions
|
||||
to this effort.
|
||||
- Added progress bars for long-running steps. ■■■■■■■□□
|
||||
- We now create linearized ("fast web view") PDFs by default. The new parameter
|
||||
`--fast-web-view` provides control over when this feature is applied.
|
||||
- Added a new `--pages` feature to limit OCR to only a specific page range.
|
||||
The list may contain commas or single pages, such as `1, 3, 5-11`.
|
||||
- When the number of pages is small compared to the number of allowed jobs, we
|
||||
run Tesseract in multithreaded (OpenMP) mode when available. This should
|
||||
improve performance on files with low page counts.
|
||||
- Removed dependency on `ruffus`, and with that, the non-reentrancy
|
||||
restrictions that previous made an API impossible.
|
||||
- Output and logging messages overhauled so that ocrmypdf may be integrated
|
||||
into applications that use the logging module.
|
||||
- pikepdf 1.6.0 is required.
|
||||
- Added a logo. 😊
|
||||
|
||||
**Bug fixes**
|
||||
|
||||
- Pages with vector artwork are treated as full color. Previously, vectors
|
||||
were ignored when considering the colorspace needed to cover a page, which
|
||||
could cause loss of color under certain settings.
|
||||
- Test suite now spawns processes less frequently, allowing more accurate
|
||||
measurement of code coverage.
|
||||
- Improved test coverage.
|
||||
- Fixed a rare division by zero (if optimization produced an invalid file).
|
||||
- Updated Docker images to use newer versions.
|
||||
- Fixed images encoded as JBIG2 with a colorspace other than `/DeviceGray`
|
||||
were not interpreted correctly.
|
||||
- Fixed a OCR text-image registration (i.e. alignment) problem when the page
|
||||
when MediaBox had a nonzero corner.
|
||||
|
||||
@@ -0,0 +1,121 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v10
|
||||
|
||||
## v10.3.3
|
||||
|
||||
- Fixed a "KeyError: 'dpi'" error message when using `--threshold` on an image.
|
||||
({issue}`607`)
|
||||
|
||||
## v10.3.2
|
||||
|
||||
- Fixed a case where we reported "no reason" for a file size increase, when we
|
||||
could determine the reason.
|
||||
- Enabled support for pdfminer.six 20200726.
|
||||
|
||||
## v10.3.1
|
||||
|
||||
- Fixed a number of test suite failures with pdfminer.six older than version 20200402.
|
||||
- Enabled support for pdfminer.six 20200720.
|
||||
|
||||
## v10.3.0
|
||||
|
||||
- Fixed an issue where we would consider images that were already JBIG2-encoded
|
||||
for optimization, potentially producing a less optimized image than the original.
|
||||
We do not believe this issue would ever cause an image to loss fidelity.
|
||||
- Where available, pikepdf memory mapping is now used. This improves performance.
|
||||
- When Leptonica 1.79+ is installed, use its new error handling API to avoid
|
||||
a "messy" redirection of stderr which was necessary to capture its error
|
||||
messages.
|
||||
- For older versions of Leptonica, added a new thread level lock. This fixes a
|
||||
possible race condition in handling error conditions in Leptonica (although
|
||||
there is no evidence it ever caused issues in practice).
|
||||
- Documentation improvements and more type hinting.
|
||||
|
||||
## v10.2.1
|
||||
|
||||
- Disabled calculation of text box order with pdfminer. We never needed this result
|
||||
and it is expensive to calculate on files with complex pre-existing text.
|
||||
- Fixed plugin manager to accept `Path(plugin)` as a path to a plugin.
|
||||
- Fixed some typing errors.
|
||||
- Documentation improvements.
|
||||
|
||||
## v10.2.0
|
||||
|
||||
- Update Docker image to use Ubuntu 20.04.
|
||||
- Fixed issue PDF/A acquires title "Untitled" after conversion. ({issue}`582`)
|
||||
- Fixed a problem where, when using `--pdf-renderer hocr`, some text would
|
||||
be missing from the output when using a more recent version of Tesseract.
|
||||
Tesseract began adding more detailed markup about the semantics of text
|
||||
that our HOCR transform did not recognize, so it ignored them. This option is
|
||||
not the default. If necessary `--redo-ocr` also redoing OCR to fix such issues.
|
||||
- Fixed an error in Python 3.9 beta, due to removal of deprecated
|
||||
`Element.getchildren()`. ({issue}`584`)
|
||||
- Implemented support using the API with `BytesIO` and other file stream objects.
|
||||
({issue}`545`)
|
||||
|
||||
## v10.1.1
|
||||
|
||||
- Fixed `OMP_THREAD_LIMIT` set to invalid value error messages on some input
|
||||
files. (The error was harmless, apart from less than optimal performance in
|
||||
some cases.)
|
||||
|
||||
## v10.1.0
|
||||
|
||||
- Previously, we `--clean-final` would cause an unpaper-cleaned page image to
|
||||
be produced twice, which was necessary in some cases but not in general. We
|
||||
now take this optimization opportunity and reuse the image if possible.
|
||||
- We now provide PNG files as input to unpaper, since it accepts them, instead
|
||||
of generating PPM files which can be very large. This can improve performance
|
||||
and temporary disk usage.
|
||||
- Documentation updated for plugins.
|
||||
|
||||
## v10.0.1
|
||||
|
||||
- Fixed regression when `-l lang1+lang2` is used from command line.
|
||||
|
||||
## v10.0.0
|
||||
|
||||
**Breaking changes**
|
||||
|
||||
- Support for pdfminer.six version 20181108 has been dropped, along with a
|
||||
monkeypatch that made this version work.
|
||||
- Output messages are now displayed in color (when supported by the terminal)
|
||||
and prefixes describing the severity of the message are removed. As such
|
||||
programs that parse OCRmyPDF's log message will need to be revised. (Please
|
||||
consider using OCRmyPDF as a library instead.)
|
||||
- The minimum version for certain dependencies has increased.
|
||||
- Many API changes; see developer changes.
|
||||
- The Python libraries pluggy and coloredlogs are now required.
|
||||
|
||||
**New features and improvements**
|
||||
|
||||
- PDF page scanning is now parallelized across CPUs, speeding up this phase
|
||||
dramatically for files with a high page counts.
|
||||
- PDF page scanning is optimized, addressing some performance regressions.
|
||||
- PDF page scanning is no longer run on pages that are not selected when the
|
||||
`--pages` argument is used.
|
||||
- PDF page scanning is now independent of Ghostscript, ending our past reliance
|
||||
on this occasionally unstable feature in Ghostscript.
|
||||
- A plugin architecture has been added, currently allowing one to more easily
|
||||
use a different OCR engine or PDF renderer from Tesseract and Ghostscript,
|
||||
respectively. A plugin can also override some decisions, such changing
|
||||
the OCR settings after initial scanning.
|
||||
- Colored log messages.
|
||||
|
||||
**Developer changes**
|
||||
|
||||
- The test spoofing mechanism, used to test correct handling of failures in
|
||||
Tesseract and Ghostscript, has been removed in favor of using plugins for
|
||||
testing. The spoofing mechanism was fairly complex and required many special
|
||||
hacks for Windows.
|
||||
- Code describing the resolution in DPI of images was refactored into a
|
||||
`ocrmypdf.helpers.Resolution` class.
|
||||
- The module `ocrmypdf._exec` is now private to OCRmyPDF.
|
||||
- The `ocrmypdf.hocrtransform` module has been updated to follow PEP8 naming
|
||||
conventions.
|
||||
- Ghostscript is no longer used for finding the location of text in PDFs, and
|
||||
APIs related to this feature have been removed.
|
||||
- Lots of internal reorganization to support plugins.
|
||||
|
||||
@@ -0,0 +1,235 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v11
|
||||
|
||||
## v11.7.3
|
||||
|
||||
- Exclude CCITT Group 3 images from being optimized. Some libraries
|
||||
OCRmyPDF uses do not seem to handle this obscure compression format properly.
|
||||
You may get errors or possible corrupted output images without this fix.
|
||||
|
||||
## v11.7.2
|
||||
|
||||
- Updated pinned versions in main.txt, primarily to upgrade Pillow to 8.1.2, due
|
||||
to recently disclosed security vulnerabilities in that software.
|
||||
- The `--sidecar` parameter now causes an exception if set to the same file as
|
||||
the input or output PDF.
|
||||
|
||||
## v11.7.1
|
||||
|
||||
- Some exceptions while attempting image optimization were only logged at the debug
|
||||
level, causing them to be suppressed. These errors are now logged appropriately.
|
||||
- Improved the error message related to `--unpaper-args`.
|
||||
- Updated documentation to mention the new conda distribution.
|
||||
|
||||
## v11.7.0
|
||||
|
||||
- We now support using `--sidecar` in conjunction with `--pages`; these arguments
|
||||
used to be mutually exclusive. ({issue}`735`)
|
||||
- Fixed a possible issue with PDF/A-1b generation. Acrobat complained that our PDFs use
|
||||
object streams. More robust PDF/A validators like veraPDF don't consider this a
|
||||
problem, but we'll honor Acrobat's objection from here on. This may increase file
|
||||
size of PDF/A-1b files. PDF/A-2b files will not be affected.
|
||||
|
||||
## v11.6.2
|
||||
|
||||
- Fixed a regression where the wrong page orientation would be produced when using
|
||||
arguments such as `--deskew --rotate-pages` ({issue}`730`).
|
||||
|
||||
## v11.6.1
|
||||
|
||||
- Fixed an issue with attempting optimize unusually narrow-width images by excluding
|
||||
these images from optimization ({issue}`732`).
|
||||
- Remove an obsolete compatibility shim for a version of pikepdf that is no longer
|
||||
supported.
|
||||
|
||||
## v11.6.0
|
||||
|
||||
- OCRmyPDF will now automatically register plugins from the same virtual environment
|
||||
with an appropriate setuptools entrypoint.
|
||||
- Refactor the plugin manager to remove unnecessary complications and make plugin
|
||||
registration more automatic.
|
||||
- `PageContext` and `PdfContext` are now formally part of the API, as they
|
||||
should have been, since they were part of `ocrmypdf.pluginspec`.
|
||||
|
||||
## v11.5.0
|
||||
|
||||
- Fixed an issue where the output page size might differ by a fractional amount
|
||||
due to rounding, when `--force-ocr` was used and the page contained objects
|
||||
with multiple resolutions.
|
||||
- When determining the resolution at which to rasterize a page, we now consider
|
||||
printed text on the page as requiring a higher resolution. This fixes issues
|
||||
with certain pages being rendered with unacceptably low resolution text, but
|
||||
may increase output file sizes in some workflows where low resolution text
|
||||
is acceptable.
|
||||
- Added a workaround to fix an exception that occurs when trying to
|
||||
`import ocrmypdf.leptonica` on Apple ARM silicon (or potentially, other
|
||||
platforms that do not permit write+executable memory).
|
||||
|
||||
## v11.4.5
|
||||
|
||||
- Fixed an issue where files may not be closed when the API is used.
|
||||
- Improved `setup.cfg` with better settings for test coverage.
|
||||
|
||||
## v11.4.4
|
||||
|
||||
- Fixed `AttributeError: 'NoneType' object has no attribute 'userunit'` ({issue}`700`),
|
||||
related to OCRmyPDF not properly forwarded an error message from pdfminer.six.
|
||||
- Adjusted typing of some arguments.
|
||||
- `ocrmypdf.ocr` now takes a `threading.Lock` for reasons outlined in the
|
||||
documentation.
|
||||
|
||||
## v11.4.3
|
||||
|
||||
- Removed a redundant debug message.
|
||||
- Test suite now asserts that most patched functions are called when they should be.
|
||||
- Test suite now skips a test that fails on two particular versions of piekpdf.
|
||||
|
||||
## v11.4.2
|
||||
|
||||
- Fixed support for Cygwin, hopefully.
|
||||
- watcher.py: Fixed an issue with the OCR_LOGLEVEL not being interpreted.
|
||||
|
||||
## v11.4.1
|
||||
|
||||
- Fixed an issue where invalid pages ranges passed using the `pages` argument,
|
||||
such as "1-0" would cause unhandled exceptions.
|
||||
- Accepted a user-contributed to the Synology demo script in misc/synology.py.
|
||||
- Clarified documentation about change of temporary file location `ocrmypdf.io`.
|
||||
- Fixed Python wheel tag which was incorrectly set to py35 even though we long
|
||||
since dropped support for Python 3.5.
|
||||
|
||||
## v11.4.0
|
||||
|
||||
- When looking for Tesseract and Ghostscript, we now check the Windows Registry to
|
||||
see if their installers registered the location of their executables. This should
|
||||
help Windows users who have installed these programs to non-standard
|
||||
locations.
|
||||
- We now report on the progress of PDF/A conversion, since this operation is
|
||||
sometimes slow.
|
||||
- Improved command line completions.
|
||||
- The prefix of the temporary folder OCRmyPDF creates has been changed from
|
||||
`com.github.ocrmypdf` to `ocrmypdf.io`. Scripts that chose to depend on this
|
||||
prefix may need to be adjusted. (This has always been an implementation detail so is
|
||||
not considered part of the semantic versioning "contract".)
|
||||
- Fixed {issue}`692`, where a particular file with malformed fonts would flood an
|
||||
internal message cue by generating so many debug messages.
|
||||
- Fixed an exception on processing hOCR files with no page record. Tesseract
|
||||
is not known to generate such files.
|
||||
|
||||
## v11.3.4
|
||||
|
||||
- Fixed an error message 'called readLinearizationData for file that is not
|
||||
linearized' that may occur when pikepdf 2.1.0 is used. (Upgrading to pikepdf
|
||||
2.1.1 also fixes the issue.)
|
||||
- File watcher now automatically includes `.PDF` in addition to `.pdf` to
|
||||
better support case sensitive file systems.
|
||||
- Some documentation and comment improvements.
|
||||
|
||||
## v11.3.3
|
||||
|
||||
- If unpaper outputs non-UTF-8 data, quietly fix this rather than choke on the
|
||||
conversion. (Possibly addresses {issue}`671`.)
|
||||
|
||||
## v11.3.2
|
||||
|
||||
- Explicitly require pikepdf 2.0.0 or newer when running on Python 3.9. (There are
|
||||
concerns about the stability of pybind11 2.5.x with Python 3.9, which is used in
|
||||
pikepdf 1.x.)
|
||||
- Fixed another issue related to page rotation.
|
||||
- Fixed an issue where image marked as image masks were not properly considered
|
||||
as optimization candidates.
|
||||
- On some systems, unpaper seems to be unable to process the PNGs we offer it
|
||||
as input. We now convert the input to PNM format, which unpaper always accepts.
|
||||
Fixes {issue}`665` and {issue}`667`.
|
||||
- DPI sent to unpaper is now rounded to a more reasonable number of decimal digits.
|
||||
- Debug and error messages from unpaper were being suppressed.
|
||||
- Some documentation tweaks.
|
||||
|
||||
## v11.3.1
|
||||
|
||||
- Declare support for new versions: pdfminer.six 20201018 and pikepdf 2.x
|
||||
- Fixed warning related to `--pdfa-image-compression` that appears at the wrong
|
||||
time.
|
||||
|
||||
## v11.3.0
|
||||
|
||||
- The "OCR" step is describing as "Image processing" in the output messages when
|
||||
OCR is disabled, to better explain the application's behavior.
|
||||
- Debug logs are now only created when run as a command line, and not when OCR
|
||||
is performed for an API call. It is the calling application's responsibility
|
||||
to set up logging.
|
||||
- For PDFs with a low number of pages, we gathered information about the input PDF
|
||||
in a thread rather than process (when there are more pages). When run as a
|
||||
thread, we did not close the file handle to the working PDF, leaking one file
|
||||
handle per call of `ocrmypdf.ocr`.
|
||||
- Fixed an issue where debug messages send by child worker processes did not match
|
||||
the log settings of parent process, causing messages to be dropped. This affected
|
||||
macOS and Windows only where the parent process is not forked.
|
||||
- Fixed the hookspec of rasterize_pdf_page to remove default parameters that
|
||||
were not handled in an expected way by pluggy.
|
||||
- Fixed another issue with automatic page rotation ({issue}`658`) due to the issue above.
|
||||
|
||||
## v11.2.1
|
||||
|
||||
- Fixed an issue where optimization of a 1-bit image with a color palette or
|
||||
associated ICC that was optimized to JBIG2 could have its colors inverted.
|
||||
|
||||
## v11.2.0
|
||||
|
||||
- Fixed an issue with optimizing PNG-type images that had soft masks or image masks.
|
||||
This is a regression introduced in (or about) v11.1.0.
|
||||
- Improved type checking of the `plugins` parameter for the `ocrmypdf.ocr`
|
||||
API call.
|
||||
|
||||
## v11.1.2
|
||||
|
||||
- Fixed hOCR renderer writing the text in roughly reverse order. This should not
|
||||
affect reasonably smart PDF readers that properly locate the position of all
|
||||
text, but may confuse those that rely on the order of objects in the content
|
||||
stream. ({issue}`642`)
|
||||
|
||||
## v11.1.1
|
||||
|
||||
- We now avoid using named temporary files when using pngquant allowing containerized
|
||||
pngquant installs to be used.
|
||||
- Clarified an error message.
|
||||
- Highest number of 1's in a release ever!
|
||||
|
||||
## v11.1.0
|
||||
|
||||
- Fixed page rotation issues: {issue}`634,589`.
|
||||
- Fixed some cases where optimization created an invalid image such as a
|
||||
1-bit "RGB" image: {issue}`629,620`.
|
||||
- Page numbers are now displayed in debug logs when pages are being grafted.
|
||||
- ocrmypdf.optimize.rewrite_png and ocrmypdf.optimize.rewrite_png_as_g4 were
|
||||
marked deprecated. Strictly speaking these should have been internal APIs,
|
||||
but they were never hidden.
|
||||
- As a precaution, pikepdf mmap-based file access has been disabled due to a
|
||||
rare race condition that causes a crash when certain objects are deallocated.
|
||||
The problem is likely in pikepdf's dependency pybind11.
|
||||
- Extended the example plugin to demonstrate conversion to mono.
|
||||
|
||||
## v11.0.2
|
||||
|
||||
- Fixed {issue}`612`, TypeError exception. Fixed by eliminating unnecessary repair of
|
||||
input PDF metadata in memory.
|
||||
|
||||
## v11.0.1
|
||||
|
||||
- Blacklist pdfminer.six 20200720, which has a regression fixed in 20200726.
|
||||
- Approve img2pdf 0.4 as it passes tests.
|
||||
- Clarify that the GPL-3 portion of pdfa.py was removed with the changes in v11.0.0;
|
||||
the debian/copyright file did not properly annotate this change.
|
||||
|
||||
## v11.0.0
|
||||
|
||||
- Project license changed to Mozilla Public License 2.0. Some miscellaneous
|
||||
code is now under MIT license and non-code content/media remains under
|
||||
CC-BY-SA 4.0. License changed with approval of all people who were found
|
||||
to have contributed to GPLv3 licensed sections of the project. ({issue}`600`)
|
||||
- Because the license changed, this is being treated as a major version number
|
||||
change; however, there are no known breaking changes in functional behavior
|
||||
or API compared to v10.x.
|
||||
|
||||
@@ -0,0 +1,181 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v12
|
||||
|
||||
## v12.7.2
|
||||
|
||||
- Fixed "invalid version number" error for Tesseract packaging with nonstandard
|
||||
version "5.0.0-rc1.20211030".
|
||||
- Fixed use of deprecated `importlib.resources.read_binary`.
|
||||
- Replace some uses of string paths with `pathlib.Path`.
|
||||
- Fixed a leaked file handle when using `--output-type none`.
|
||||
- Removed shims to support versions of pikepdf that are no longer supported.
|
||||
|
||||
## v12.7.1
|
||||
|
||||
- Declare support for pdfminer.six v20211012.
|
||||
|
||||
## v12.7.0
|
||||
|
||||
- Fixed test suite failure when using pikepdf 3.2.0 that was compiled with pybind11
|
||||
2.8.0. {issue}`843`
|
||||
- Improve advice to user about using `--max-image-mpixels` if OCR fails for this
|
||||
reason.
|
||||
- Minor documentation fixes. (Thanks to @mara004.)
|
||||
- Don't require importlib-metadata and importlib-resources backports on versions of
|
||||
Python where the standard library implementation is sufficient.
|
||||
(Thanks to Marco Genasci.)
|
||||
|
||||
## v12.6.0
|
||||
|
||||
- Implemented `--output-type=none` to skip producing PDFs for applications that
|
||||
only want sidecar files ({issue}`787`).
|
||||
- Fixed ambiguities in descriptions of behavior of `--jbig2-lossy`.
|
||||
- Various improvements to documentation.
|
||||
|
||||
## v12.5.0
|
||||
|
||||
- Fixed build failure for the combination of PyPy 3.6 and pikepdf 3.0. This
|
||||
combination can work in a source build but does not work with wheels.
|
||||
- Accepted bot that wanted to upgrade our deprecated requirements.txt.
|
||||
- Documentation updates.
|
||||
- Replace pkg_resources and install dependency on setuptools with
|
||||
importlib-metadata and importlib-resources.
|
||||
- Fixed regression in hocrtransform causing text to be omitted when this
|
||||
renderer was used.
|
||||
- Fixed some typing errors.
|
||||
|
||||
## v12.4.0
|
||||
|
||||
- When grafting text layers, use pikepdf's `unparse_content_stream` if available.
|
||||
- Confirmed support for pluggy 1.0. (Thanks @QuLogic.)
|
||||
- Fixed some typing issues, improved pre-commit settings, and fixed issues
|
||||
flagged by linters.
|
||||
- PyPy 7.3.3 (=Python 3.6) is now supported. Note that PyPy does not necessarily
|
||||
run faster, because the vast majority of OCRmyPDF's execution time is spent
|
||||
running OCR or generally executing native code. However, PyPy may bring speed
|
||||
improvements in some areas.
|
||||
|
||||
## v12.3.3
|
||||
|
||||
- watcher.py: fixed interpretation of boolean env vars ({issue}`821`).
|
||||
- Adjust CI scripts to test Tesseract 5 betas.
|
||||
- Document our support for the Tesseract 5 betas.
|
||||
|
||||
## v12.3.2
|
||||
|
||||
- Indicate support for flask 2.x, watcher 2.x ({issue}`815, 816`).
|
||||
|
||||
## v12.3.1
|
||||
|
||||
- Fixed issue with selection of text when using the hOCR renderer ({issue}`813`).
|
||||
- Fixed build errors with the Docker image by upgrading to a newer Ubuntu.
|
||||
Also set the timezone of this image to UTC.
|
||||
|
||||
## v12.3.0
|
||||
|
||||
- Fixed a regression introduced in Pillow 8.3.0. Pillow no longer rounds DPI
|
||||
for image resolutions. We now account for this ({issue}`802`).
|
||||
- We no longer use some API calls that are deprecated in the latest versions of
|
||||
pikepdf.
|
||||
- Improved error message when a language is requested that doesn't look like a
|
||||
typical ISO 639-2 code.
|
||||
- Fixed some tests that attempted to symlink on Windows, breaking tests on a
|
||||
Windows desktop but not usually on CI.
|
||||
- Documentation fixes (thanks to @mara004)
|
||||
|
||||
## v12.2.0
|
||||
|
||||
- Fixed invalid Tesseract version number on Windows ({issue}`795`).
|
||||
- Documentation tweaks. Documentation build now depends on sphinx-issues package.
|
||||
|
||||
## v12.1.0
|
||||
|
||||
- For security reasons we now require Pillow >= 8.2.x. (Older versions will continue
|
||||
to work if upgrading is not an option.)
|
||||
- The build system was reorganized to rely on `setup.cfg` instead of `setup.py`.
|
||||
All changes should work with previously supported versions of setuptools.
|
||||
- The files in `requirements/*` are now considered deprecated but will be retained for v12.
|
||||
Instead use `pip install ocrmypdf[test]` instead of `requirements/test.txt`, etc.
|
||||
These files will be removed in v13.
|
||||
|
||||
## v12.0.3
|
||||
|
||||
- Expand the list of languages supported by the hocr PDF renderer.
|
||||
Several languages were previously considered not supported, particularly those
|
||||
non-European languages that use the Latin alphabet.
|
||||
- Fixed a case where the exception stack trace was suppressed in verbose mode.
|
||||
- Improved documentation around commercial OCR.
|
||||
|
||||
## v12.0.2
|
||||
|
||||
- Fixed exception thrown when using `--remove-background` on files containing small
|
||||
images ({issue}`769`).
|
||||
- Improve documentation for description of adding language packs to the Docker image
|
||||
and corrected name of French language pack.
|
||||
|
||||
## v12.0.1
|
||||
|
||||
- Fixed "invalid version number" for untagged tesseract versions ({issue}`770`).
|
||||
|
||||
## v12.0.0
|
||||
|
||||
**Breaking changes**
|
||||
|
||||
- Due to recent security issues in pikepdf, Pillow and reportlab, we now require
|
||||
newer versions of these libraries and some of their dependencies. (If necessary,
|
||||
package maintainers may override these versions at their discretion; lower
|
||||
versions will often work.)
|
||||
- We now use the "LeaveColorUnchanged" color conversion strategy when directing
|
||||
Ghostscript to create a PDF/A. Generally this is faster than performing a
|
||||
color conversion, which is not always necessary.
|
||||
- OCR text is now packaged in a Form XObject. This makes it easier to isolate
|
||||
OCR from other document content. However, some poorly implemented PDF text
|
||||
extraction algorithms may fail to detect the text.
|
||||
- Many API functions have stricter parameter checking or expect keyword arguments
|
||||
were they previously did not.
|
||||
- Some deprecated functions in `ocrmypdf.optimize` were removed.
|
||||
- The `ocrmypdf.leptonica` module is now deprecated, due to difficulties with
|
||||
the current strategy of ABI binding on newer platforms like Apple Silicon.
|
||||
It will be removed and replaced, either by repackaging Leptonica as an
|
||||
independent library using or using a different image processing library.
|
||||
- Continuous integration moved to GitHub Actions.
|
||||
- We no longer depend on `pytest_helpers_namespace` for testing.
|
||||
|
||||
**New features**
|
||||
|
||||
- New plugin hook: `get_progressbar_class`, for progress reporting,
|
||||
allowing developers to replace the standard console progress bar with some
|
||||
other mechanism, such as updating a GUI progress bar.
|
||||
- New plugin hook: `get_executor`, for replacing the concurrency model.
|
||||
This is primarily to support execution on AWS Lambda, which does not support
|
||||
standard Python `multiprocessing` due to its lack of shared memory.
|
||||
- New plugin hook: `get_logging_console`, for replacing the standard
|
||||
way OCRmyPDF outputs its messages.
|
||||
- New plugin hook: `filter_pdf_page`, for modifying individual PDF
|
||||
pages produced by OCRmyPDF.
|
||||
- OCRmyPDF now runs on nonstandard execution environments that do not have
|
||||
interprocess semaphores, such as AWS Lambda and Android Termux. If the environment
|
||||
does not have semaphores, OCRmyPDF will automatically select an alternate
|
||||
process executor that does not use semaphores.
|
||||
- Continuous integration moved to GitHub Actions.
|
||||
- We now generate an ARM64-compatible Docker image alongside the x64 image.
|
||||
Thanks to @andkrause for doing most of the work in a pull request several months
|
||||
ago, which we were finally able to integrate now. Also thanks to @0x326 for
|
||||
review comments.
|
||||
|
||||
**Fixes**
|
||||
|
||||
- Fixed a possible deadlock on attempting to flush `sys.stderr` when older
|
||||
versions of Leptonica are in use.
|
||||
- Some worker processes inherited resources from their parents such as log
|
||||
handlers that may have also lead to deadlocks. These resources are now released.
|
||||
- Improvements to test coverage.
|
||||
- Removed vestiges of support for Tesseract versions older than 4.0.0-beta1 (
|
||||
which ships with Ubuntu 18.04).
|
||||
- OCRmyPDF can now parse all of Tesseract version numbers, since several
|
||||
schemes have been in use.
|
||||
- Fixed an issue with parsing PDFs that contain images drawn at a scale of 0. ({issue}`761`)
|
||||
- Removed a frequently repeated message about disabling mmap.
|
||||
|
||||
@@ -0,0 +1,175 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v13
|
||||
|
||||
## v13.7.0
|
||||
|
||||
- Fixed an exception when attempting to run and Tesseract is not installed.
|
||||
- Changed to SPDX license tracking and information files.
|
||||
|
||||
## v13.6.2
|
||||
|
||||
- Added a shim to prevent an "error during error handling" for Python 3.7 and 3.8.
|
||||
- Modernized some type annotations.
|
||||
- Improved annotations on our \_windows module to help IDEs and mypy figure out what
|
||||
we're doing.
|
||||
|
||||
## v13.6.1
|
||||
|
||||
- Require setuptools-scm 7.0.5 to avoid possible issues with source distributions in
|
||||
earlier versions of setuptools-scm.
|
||||
- Suppress a spurious warning, improve tests, improve typing and other miscellany.
|
||||
|
||||
## v13.6.0
|
||||
|
||||
- Added a new `initialize` plugin hook, making it possible to suppress built-in
|
||||
plugins more easily, among other possibilities.
|
||||
- Fixed an issue where unpaper would exit with a "wrong stream" error, probably
|
||||
related to images with an odd integer width. {issue}`887, 665`
|
||||
|
||||
## v13.5.0
|
||||
|
||||
- Added a new `optimize_pdf` plugin hook, making it possible to create plugins that
|
||||
replace or enhance OCRmyPDF's PDF optimizer.
|
||||
- Removed all max version restrictions. Our new policy is to blacklist known-bad releases
|
||||
and only block known-bad versions of dependencies.
|
||||
- The naming schema for object that holds all OCR text that OCRmyPDF inserts has
|
||||
changed. This has always been an implementation detail (and remains so), but possibly,
|
||||
someone was relying on it and would appreciate the heads-up.
|
||||
- Cleanup.
|
||||
|
||||
## v13.4.7
|
||||
|
||||
- Fixed PermissionError when cleaning up temporary files in rare cases. {issue}`974`
|
||||
- Fixed PermissionError when calling `os.nice` on platforms that lack it. {issue}`973`
|
||||
- Suppressed some warnings from libxmp during tests.
|
||||
|
||||
## v13.4.6
|
||||
|
||||
- Convert error on corrupt ICC profiles into a warning. Thanks to @oscherler.
|
||||
|
||||
## v13.4.5
|
||||
|
||||
- Remove upper bound on pdfminer.six version.
|
||||
- Documentation.
|
||||
|
||||
## v13.4.4
|
||||
|
||||
- Updated pdfminer.six version.
|
||||
- Docker image changed to Ubuntu 22.04 now that it is released and provides the
|
||||
dependencies we need. This seems more consistent than our recent change to
|
||||
Debian.
|
||||
|
||||
## v13.4.3
|
||||
|
||||
- Fix error on pytest.skip() with older versions of pytest.
|
||||
- Documentation updates.
|
||||
|
||||
## v13.4.2
|
||||
|
||||
- Worked around a
|
||||
[major regression in Ghostscript 9.56.0](https://bugs.ghostscript.com/show_bug.cgi?id=705187)
|
||||
where **all OCR text is stripped out of the PDF**. It simply removes all text,
|
||||
even generated by software other than OCRmyPDF. Fortunately, we can ask
|
||||
Ghostscript 9.56.0 to use its old behavior that worked correctly for our purposes.
|
||||
Users must avoid the combination (Ghostscript 9.56.0, ocrmypdf \<13.4.2) since
|
||||
older versions of OCRmyPDF have no way of detecting that this particular
|
||||
version of Ghostscript removes all OCR text.
|
||||
- Marked pdfminer 20220319 as supported.
|
||||
- Fixed some deprecation warnings from recent versions of Pillow and pytest.
|
||||
- Test suite now covers Python 3.10 (Python 3.10 worked fine before, but was not
|
||||
being tested).
|
||||
- Docker image now uses debian:bookworm-slim as the base image to fix the Docker
|
||||
image build.
|
||||
|
||||
## v13.4.1
|
||||
|
||||
- Temporarily make threads rather than processes the default executor worker, due
|
||||
to a persistent deadlock issue when processes are used. Add a new command line
|
||||
argument `--no-use-threads` to disable this.
|
||||
|
||||
## v13.4.0
|
||||
|
||||
- Fixed test failures when using pikepdf 5.0.0.
|
||||
- Various improvements to the optimizer. In particular, we now recognize PDF images
|
||||
that are encoded with both deflate (PNG) and DCT (JPEG), and also produce PDF
|
||||
with images compressed with deflate and DCT, since this often yields file size
|
||||
improvements compared to plain DCT.
|
||||
|
||||
## v13.3.0
|
||||
|
||||
- Made a harmless but "scary" exception after failing to optimize an image less scary.
|
||||
- Added a warning if a page image is too large for unpaper to clean. The image is
|
||||
passed through without cleaning. This is due to a hard-coded limitation in a
|
||||
C library used by unpaper so it cannot be rectified easily.
|
||||
- We now use better default settings when calling img2pdf.
|
||||
- We no longer try to optimize images that we failed to save in certain situations.
|
||||
- We now account for some differences in text output from Tesseract 5 compared to
|
||||
Tesseract 4.
|
||||
- Better handling of Ghostscript producing empty images when attempting to rasterize
|
||||
page images.
|
||||
|
||||
## v13.2.0
|
||||
|
||||
- Removed all runtime uses of distutils since it is deprecated in standard library. We
|
||||
previous used `distutils.version` to examine version numbers of dependencies
|
||||
at run time, and now use `packaging.version` for this. This is a new
|
||||
dependency.
|
||||
- Fixed an error message advising the user that Ghostscript was not installed being
|
||||
suppressed when this condition actually happens.
|
||||
- Fixed an issue with incorrect page number and totals being displayed in the progress
|
||||
bar. This was purely a display/presentation issue. {issue}`876`.
|
||||
|
||||
## v13.1.1
|
||||
|
||||
- Fixed issue with attempting to deskew a blank page on Tesseract 5. {issue}`868`.
|
||||
|
||||
## v13.1.0
|
||||
|
||||
- Changed to using Python concurrent.futures-based parallel execution instead of
|
||||
pools, since futures have now exceed pools in features.
|
||||
- If a child worker is terminated (perhaps by the operating system or the user
|
||||
killing it in a task manager), the parallel task will fail an error message.
|
||||
Previously, the main ocrmypdf process would "hang" indefinitely, waiting for the
|
||||
child to report.
|
||||
- Added new argument `--tesseract-thresholding` to provide control over Tesseract 5's
|
||||
threshold parameter.
|
||||
- Documentation updates and changes. Better documentation for `--output-type none`,
|
||||
added a few releases ago. Removed some obsolete documentation.
|
||||
- Improved bash completions - thanks to @FPille.
|
||||
|
||||
## v13.0.0
|
||||
|
||||
**Breaking changes**
|
||||
|
||||
- The deprecated module `ocrmypdf.leptonica` has been removed.
|
||||
- We no longer depend on Leptonica (`liblept`) or CFFI (`libffi`,
|
||||
`python3-cffi`). (Note that Tesseract still requires Leptonica; OCRmyPDF no longer
|
||||
directly uses this library.)
|
||||
- The argument `--remove-background` is temporarily disabled while we search for an
|
||||
alternative to the Leptonica implementation of this feature.
|
||||
- The `--threshold` argument has been removed, since this also depended on Leptonica.
|
||||
Tesseract 5.x has implemented improvements to thresholding, so this feature will be
|
||||
redundant anyway.
|
||||
- `--deskew` was previous calculated by a Leptonica algorithm. We now use a feature
|
||||
of Tesseract to find the appropriate the angle to deskew a page. The deskew angle
|
||||
according to Tesseract may differ from Leptonica's algorithm. At least in theory,
|
||||
Tesseract's deskew angle is informed by a more complex analysis than Leptonica,
|
||||
so this should improve results in general. We also use Pillow to perform the
|
||||
deskewing, which may affect the appearance of the image compared to Leptonica.
|
||||
- Support for Python 3.6 was dropped, since this release is approaching end of life.
|
||||
- We now require pikepdf 4.0 or newer. This, in turn, means that OCRmyPDF requires
|
||||
a system compatible with the manylinux2014 specification. This change was "forced"
|
||||
by Pillow not releasing manylinux2010 wheels anymore.
|
||||
- We no longer provide requirements.txt-style files. Use `pip install ocrmypdf[...]`
|
||||
instead.
|
||||
- Bumped required versions of several libraries.
|
||||
|
||||
**Fixes**
|
||||
|
||||
- Fixed an issue where OCRmyPDF failed to find Ghostscript on Windows even when
|
||||
installed, and would exit with an error.
|
||||
- By removing Leptonica, we fixed all issues related to Leptonica on Apple
|
||||
Silicon or Leptonica failing to import on Windows.
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v14
|
||||
|
||||
## v14.4.0
|
||||
|
||||
- Digitally signed PDFs are now detected. If the PDF is signed, OCRmyPDF will
|
||||
refuse to modify it. Previously, only encrypted PDFs were detected, not
|
||||
those that were signed but not encrypted. {issue}`1040`
|
||||
- In addition, `--invalidate-digital-signatures` can be used to override the
|
||||
above behavior and modify the PDF anyway. {issue}`1040`
|
||||
- tqdm progress bars replaced with "rich" progress bars. The rich library is
|
||||
a new dependency. Certain APIs that used tqdm are now deprecated and will
|
||||
be removed in the next major release.
|
||||
- Improved integration with GitHub Releases. Thanks to @stumpylog.
|
||||
|
||||
## v14.3.0
|
||||
|
||||
- Renamed master branch to main.
|
||||
- Improve PDF rasterization accuracy by using the `-dPDFSTOPONERROR` option
|
||||
to Ghostscript. Use `--continue-on-soft-render-error` if you want to render
|
||||
the PDF anyway. The plugin specification was adjusted to support this feature;
|
||||
plugin authors may want to adapt PDF rasterizing and rendering
|
||||
plugins. {issue}`1083`
|
||||
- The calculated deskew angle is now recorded in the logged output. {issue}`1101`
|
||||
- Metadata can now be unset by setting a metadata type such as `--title` to an
|
||||
empty string. {issue}`1117,1059`
|
||||
- Fixed random order of languages due to use of a set. This may have caused output
|
||||
to vary when multiple languages were set for OCR. {issue}`1113`
|
||||
- Clarified the optimization ratio reported in the log output.
|
||||
- Documentation improvements.
|
||||
|
||||
## v14.2.1
|
||||
|
||||
- Fixed {issue}`977`, where images inside Form XObjects were always excluded
|
||||
from image optimization.
|
||||
|
||||
## v14.2.0
|
||||
|
||||
- Added `--tesseract-downsample-above` to downsample larger images even when
|
||||
they do not exceed Tesseract's internal limits. This can be used to speed
|
||||
up OCR, possibly sacrificing accuracy.
|
||||
- Fixed resampling AttributeError on older Pillow. {issue}`1096`
|
||||
- Removed an error about using Ghostscript on PDFs with that have the /UserUnit
|
||||
feature in use. Previously, Ghostscript would fail to process these PDFs,
|
||||
but in all supported versions it is now supported, so the error is no longer
|
||||
needed.
|
||||
- Improved documentation around installing other language packs for Tesseract.
|
||||
|
||||
## v14.1.0
|
||||
|
||||
- Added `--tesseract-non-ocr-timeout`. This allows using Tesseract's deskew
|
||||
and other non-OCR features while disabling OCR using `--tesseract-timeout 0`.
|
||||
- Added `--tesseract-downsample-large-images`. This downsamples larges images
|
||||
that exceed the maximum image size Tesseract can handle. Large images may still
|
||||
take a long time to process, but this allows them to be processed if that
|
||||
is desired.
|
||||
- Fixed {issue}`1082`, an issue with snap packaged building.
|
||||
- Change linter to ruff, fix lint errors, update documentation.
|
||||
|
||||
## v14.0.4
|
||||
|
||||
- Fixed {issue}`1066, 1075`, an exception when processing certain malformed PDFs.
|
||||
|
||||
## v14.0.3
|
||||
|
||||
- Fixed {issue}`1068`, avoid deleting /dev/null when running as root.
|
||||
- Other documentation fixes.
|
||||
|
||||
## v14.0.2
|
||||
|
||||
- Fixed {issue}`1052`, an exception on attempting to process certain nonconforming PDFs.
|
||||
- Explicitly documented that Windows 32-bit is no longer supported.
|
||||
- Fixed source installation instructions.
|
||||
- Other documentation fixes.
|
||||
|
||||
## v14.0.1
|
||||
|
||||
- Fixed some version checks done with smart version comparison.
|
||||
- Added missing jbig2dec to Docker image.
|
||||
|
||||
## v14.0.0
|
||||
|
||||
- Dropped support for Python 3.7.
|
||||
- Dropped support generally speaking, all dependencies older than what Ubuntu 20.04
|
||||
provides.
|
||||
- Ghostscript 9.50 or newer is now required. Shims to support old versions were
|
||||
removed.
|
||||
- Tesseract 4.1.1 or newer is now required. Shims to support old versions were
|
||||
removed.
|
||||
- Docker image now uses Tesseract 5.
|
||||
- Dropped setup.cfg configuration for pyproject.toml.
|
||||
- Removed deprecation exception PdfMergeFailedError.
|
||||
- A few more public domain test files were removed or replaced. We are aiming for
|
||||
100% compliance with SPDX and generally towards simplifying copyright.
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v15
|
||||
|
||||
## v15.4.4
|
||||
|
||||
- Fixed documentation for installing Ghostscript on Windows. {issue}`1198`
|
||||
- Added warning message about security issue in older versions of Ghostscript.
|
||||
|
||||
## v15.4.3
|
||||
|
||||
- Fixed deprecation warning in pikepdf older than 8.7.1; pikepdf >= 8.7.1 is
|
||||
now required.
|
||||
|
||||
## v15.4.2
|
||||
|
||||
- We now raise an exception on a certain class of PDFs that likely need an
|
||||
explicit color conversion strategy selected to display correctly
|
||||
for PDF/A conversion.
|
||||
- Fixed an error that occurred while trying to write a log message after the
|
||||
debug log handler was removed.
|
||||
|
||||
## v15.4.1
|
||||
|
||||
- Fixed misc/watcher.py regressions: accept `--ocr-json-settings` as either
|
||||
filename or JSON string, as previously; and argument count mismatch.
|
||||
{issue}`1183,1185`
|
||||
- We no longer attempt to set /ProcSet in the PDF output, since this is an
|
||||
obsolete PDF feature.
|
||||
- Documentation improvements.
|
||||
|
||||
## v15.4.0
|
||||
|
||||
- Added new experimental APIs to support offline editing of the final text.
|
||||
Specifically, one can now generate hOCR files with OCRmyPDF, edit them with
|
||||
some other tool, and then finalize the PDF. They are experimental and
|
||||
subject to change, including details of how the working folder is used.
|
||||
There is no command line interface.
|
||||
- Code reorganization: executors, progress bars, initialization and setup.
|
||||
- Fixed test coverage in cases where the coverage tool did not properly trace
|
||||
into threads or subprocesses. This code was still being tested but appeared
|
||||
as not covered.
|
||||
- In the test suite, reduced use of subprocesses and other techniques that
|
||||
interfere with coverage measurement.
|
||||
- Improved error check for when we appear to be running inside a snap container
|
||||
and files are not available.
|
||||
- Plugin specification now properly defines progress bars as a protocol rather
|
||||
than defining them as "tqdm-like".
|
||||
- We now default to using "forkserver" process creation on POSIX platforms
|
||||
rather than fork, since this is method is more robust and avoids some
|
||||
issues when threads are present.
|
||||
- Fixed an instance where the user's request to `--no-use-threads` was ignored.
|
||||
- If a PDF does not have language metadata on its top level object, we add
|
||||
the OCR language.
|
||||
- Replace some cryptic test error messages with more helpful ones.
|
||||
- Debug messages for how OCRmyPDF picks the colorspace for a page are now
|
||||
more descriptive.
|
||||
|
||||
## v15.3.1
|
||||
|
||||
- Fixed an issue with logging settings for misc/watcher.py introduced in the
|
||||
previous release. {issue}`1180`
|
||||
- We now attempt to preserve the input's extended attributes when creating
|
||||
the output file.
|
||||
- For some reason, the macOS build now needs OpenSSL explicitly installed.
|
||||
- Updated documentation on Docker performance concerns.
|
||||
|
||||
## v15.3.0
|
||||
|
||||
- Update misc/watcher.py to improve command line interface using Typer, and
|
||||
support `.env` specification of environment variables. Improved error
|
||||
messages. Thanks to @mflagg2814 for the PR that prompted this improvement.
|
||||
- Improved error message when a file cannot be read because we are running in
|
||||
a snap container.
|
||||
|
||||
## v15.2.0
|
||||
|
||||
- Added a Docker image based on Alpine Linux. This image is smaller than the
|
||||
Ubuntu-based image and may be useful in some situations. Currently hosted at
|
||||
jbarlow83/ocrmypdf-alpine. Currently not available in ARM flavor.
|
||||
- The Ubuntu Docker is now aliased to jbarlow83/ocrmypdf-ubuntu.
|
||||
- Updated Docker documentation.
|
||||
|
||||
## v15.1.0
|
||||
|
||||
- We now require Pillow 10.0.1, due a serious security vulnerability in all earlier
|
||||
versions of that dependency. The vulnerability concerns WebP images and could
|
||||
be triggered in OCRmyPDF when creating a PDF from a malicious WebP image.
|
||||
- Added some keyword arguments to `ocrmypdf.ocr` that were previously accepted
|
||||
but undocumented.
|
||||
- Documentation updates and typing improvements.
|
||||
|
||||
## v15.0.2
|
||||
|
||||
- Added Python 3.12 to test matrix.
|
||||
- Updated documentation for notes on Python 3.12, 32-bit support and some new
|
||||
features in v15.
|
||||
|
||||
## v15.0.1
|
||||
|
||||
- Wheels Python tag changed to py39.
|
||||
- Marked as a expected fail a test that fails on recent Ghostscript versions.
|
||||
- Clarified documentation and release notes around the extent of 32-bit support.
|
||||
- Updated installation documentation to changes in v15.
|
||||
|
||||
## v15.0.0
|
||||
|
||||
- Dropped support for Python 3.8.
|
||||
- Dropped support some older dependencies, specifically `coloredlogs` and
|
||||
`tqdm` in favor of rich - see `pyproject.toml` for details.
|
||||
Generally speaking, Ubuntu 22.04 is our new baseline system.
|
||||
- Tightened version requirements for some dependencies.
|
||||
- Dropped support for 32-bit Linux wheels. We strongly recommend a 64-bit operating
|
||||
system, and 64-bit versions of Python, Tesseract and Ghostscript to use OCRmyPDF.
|
||||
Many of our dependencies are dropping 32-bit builds (e.g. Pillow), and we are
|
||||
following suit. (Maintainers may still build 32-bit versions from source.)
|
||||
- Changed to trusted release for PyPI publishing.
|
||||
- pikepdf memory mapping is enabled again for improved performance, now that an
|
||||
issue with feature in pikepdf is fixed.
|
||||
- `ocrmypdf.helpers.calculate_downsample` previously had two variants, one
|
||||
that took a `PIL.Image` and one that took a `tuple[int, int]`. The latter
|
||||
was removed.
|
||||
- The snap version of ocrmypdf is now based on Ubuntu core22.
|
||||
- We now account for situations where a small portion of an image on a page is drawn
|
||||
at high DPI (resolution). Previously, the entire page would be rasterized at the
|
||||
highest resolution of any feature, which caused performance problems. Now,
|
||||
the page is rasterized
|
||||
at a resolution based on the average DPI of the page, weighted by the area that
|
||||
each feature occupies. Typically, small areas of high resolution in PDFs are
|
||||
errors or quirks from the repeated use of assets and high resolution is not
|
||||
beneficial. {issue}`1010,1104,1004,1079,1010`
|
||||
- Ghostscript color conversion strategy is now configurable using
|
||||
`--color-conversion-strategy`. {issue}`1143`
|
||||
- JBIG2 threshold for optimization is now configurable using
|
||||
`--jbig2-threshold`. {issue}`1133`
|
||||
|
||||
@@ -0,0 +1,291 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v16
|
||||
|
||||
## v16.13.0
|
||||
|
||||
- Added detection and repair for Ghostscript 10.6 JPEG corruption. When GS 10.6
|
||||
truncates JPEG data by 1-15 bytes, OCRmyPDF now restores the original image
|
||||
bytes from the input PDF. A warning is issued when GS 10.6+ is detected.
|
||||
{issue}`1603`
|
||||
- We continue to force re-optimization of JPEGs, since this catches some issues with corruption for situations where Ghostscript modifies an image. It is likely there are still cases where we cannot mitigate all corruption issues. {issue}`1585`
|
||||
- Fixed handling of PDF page boxes (ArtBox, BleedBox) which were not being
|
||||
processed correctly in some cases. {issue}`1181,1360`
|
||||
- Documentation: clarified podman usage instructions.
|
||||
|
||||
## v16.12.0
|
||||
|
||||
- Disable Ghostscript's subset fonts feature, which was found to corrupt text in certain
|
||||
PDFs. Thanks @mnaegler for identifying this issue. {issue}`1592`
|
||||
- Users of Ghostscript 10.6.0+ reported that Ghostscript seems to generate corrupted
|
||||
JPEGs. We force re-optimization of these JPEGs to mitigate the corruption until
|
||||
Ghostscript fixes the issue. {issue}`1585`
|
||||
- OCRmyPDF now avoids applying flate compression to large JPEG images, unless maximum
|
||||
optimization is requested, since flate+DCT compression reduces performances in PDF
|
||||
viewers with large images.
|
||||
- Updated Dockerfiles to use more recent base operating systems.
|
||||
- Updated build and test matrix to include Python 3.14.
|
||||
- Minor documentation improvements.
|
||||
- pikepdf >= 10.0.0 is now required.
|
||||
|
||||
## v16.11.1
|
||||
|
||||
- Fixed issue with Tesseract changing an error message related to skew. {issue}`1576`
|
||||
- Dropped macOS 13 from build-test matrix since it is no longer supported by Apple.
|
||||
|
||||
## v16.11.0
|
||||
|
||||
- Deprecated "semfree" plugin in favor of falling back to threads if the platform
|
||||
does not support semaphores. Fixes an issue with Python 3.14.
|
||||
- Fixed references to PDF/A compliances levels to be consistent with ISO nomenclature.
|
||||
Thanks @5HT2. {issue}`1557`
|
||||
- Fixed an issue around using plugin_manager as an argument. {issue}`1555`
|
||||
- Added OpenBSD install steps to README. {issue}`1554`
|
||||
- Removed PyPy from test matrix due to declining support in third party libraries.
|
||||
- Documentation improvements.
|
||||
|
||||
## v16.10.4
|
||||
|
||||
- Corrected build errors in Python 3.13.3 and 3.13.4.
|
||||
|
||||
## v16.10.3 (not released)
|
||||
|
||||
- Blocked optimization of images with pre-blended soft masks. {issue}`1536`
|
||||
- Fixed warning from hypothesis on running tests.
|
||||
- Release incomplete due to new test failures in Python 3.13.3 and 3.13.4.
|
||||
|
||||
## v16.10.2
|
||||
|
||||
- Blacklist pikepdf 9.8.0 due to an incompatible change.
|
||||
|
||||
## v16.10.1
|
||||
|
||||
- No changes affecting OCRmyPDF functionality for command line end users.
|
||||
- webservice: made page specification easier to find in UI.
|
||||
- webservice: fix download button downloads wrong file.
|
||||
- Converted project documentation from rST to Markdown.
|
||||
- Added README translation to Simplified Chinese. Thanks @HuaPai.
|
||||
- Modernized license specification in pyproject.toml.
|
||||
- Modernized SPDX license to REUSE.toml.
|
||||
|
||||
## v16.10.0
|
||||
|
||||
- Added hocr textangle processing, improving handling of text at angles.
|
||||
Thanks @0dinD {issue}`1467`
|
||||
- Docker documentation updates related to podman. Thanks @rugk. {issue}`1489,1488`
|
||||
- Dropped webservice.py's fragile use of ttyd. Instead, messages from ocrmypdf are
|
||||
printed to the console.
|
||||
- Fixed broken test test_hocrtransform_matches_sandwich, which had become
|
||||
an invalid test. Thanks @QuLogic for reporting.
|
||||
- Improved install instructions for Windows. Thanks @alex.
|
||||
|
||||
## v16.9.0
|
||||
|
||||
- Added hocr caption processing. Thanks @0dinD {issue}`1466`
|
||||
- ocrmypdf-alpine Docker image is now built with Alpine 3.21.
|
||||
- Fixed error handling of PDFs that contain invalid images with both ImageMask
|
||||
and ColorSpace defined. {issue}`1453`
|
||||
- Fixed test suite regression when only older Ghostscripts are installed.
|
||||
- Improved documetnation of \_progressbar.py. Thanks @QuentinFuxa. {issue}`1456`
|
||||
- Disabling building of documentation as PDF on ReadTheDocs, as this caused
|
||||
complex build issues deemed not worth solving.
|
||||
|
||||
## v16.8.0
|
||||
|
||||
- Upgraded webservice.py demonstration using streamlit. It's now possible to
|
||||
exercise most of OCRmyPDF's functionality in a simple web UI.
|
||||
- Added cache to Dockerfiles to improve build speed.
|
||||
- Fixed numerous formatting errors in the documentation that prevented some
|
||||
parts of documentation from generating correctly.
|
||||
- Improved OCR text rendering by suppressing negative-width spaces. Thanks
|
||||
@pajowu. {issue}`1446`
|
||||
- Improved detecting of invisible text when using `--redo-ocr`. Thanks
|
||||
@pajowu. {issue}`1448``
|
||||
|
||||
## v16.7.0
|
||||
|
||||
- Fixed further issues with Docker build and updated some versions.
|
||||
- Main Docker image returned to Ubuntu 24.04 since the fix in v16.6.2 resolved
|
||||
that concern.
|
||||
- Code that previously sent Ghostscript output to stdout has been changed to
|
||||
output to temporary files, since Ghostscript was doing that anyway internally.
|
||||
This is a modest efficiency improvement.
|
||||
- Fixed an issue with debug log output being parsed as rich markup. {issue}`1444`
|
||||
|
||||
## v16.6.2
|
||||
|
||||
- Remove invalid hyperlink annotations to satisfy Ghostscript 10.x during PDF/A
|
||||
conversion. {issue}`1425`
|
||||
|
||||
## v16.6.1
|
||||
|
||||
- Fixed some issues with Docker build, such as removing unnecessary content and using
|
||||
a stable Tesseract version.
|
||||
- Reverted Docker image to Ubuntu 22.04 to access older/more stable Ghostscript
|
||||
for now.
|
||||
- Clarified batch commands in documentation.
|
||||
- Fixed an issue with JSON serialization and pickling of HOCRResult. {issue}`1427`
|
||||
|
||||
## v16.6.0
|
||||
|
||||
- Fixed an issue where damaged PDFs would fail with `--redo-ocr`. {issue}`1403`
|
||||
- Fixed an error that prevented JBIG2 optimization on Windows if the image
|
||||
was optimized in an earlier step. {issue}`1396`
|
||||
- Fixed an error detecting the version of unpaper 7.0.0. {issue}`1409`
|
||||
- Fixed a performance regression when scanning pages. {issue}`1378`. Thanks @aliemjay.
|
||||
- Fixed Alpine Docker image by enforcing Alpine 3.19. Alpine 3.20 includes a
|
||||
defective version of Tesseract OCR and so is not usable.
|
||||
- Upgraded Ubuntu Docker image to use Ubuntu 24.04.
|
||||
- Build and test scripts/actions switched to uv.
|
||||
- When running in a container, we now remind the user that temporary folders
|
||||
are inside the container and may not be accessible.
|
||||
- Fixed Linux test coverage matrix, which was missing some key versions.
|
||||
|
||||
## v16.5.0
|
||||
|
||||
- Fixed issue with interpreting PDFs that have images with array masks.
|
||||
{issue}`1377`
|
||||
- Enabled testing on Python 3.13.
|
||||
- Fixed a test that did not work correctly but still passed. {issue}`1382`
|
||||
- Improved "PDF/A conversion failed" warning message to better describe implications.
|
||||
- Updated documentation to better explain OCR_JSON_SETTINGS in batch processing.
|
||||
- Build backend changed from setuptools to hatchling.
|
||||
|
||||
## v16.4.3
|
||||
|
||||
- Work around pdfminer.six issue where a token on the buffer boundary is incorrectly
|
||||
parsed as two tokens. {issue}`1361`
|
||||
- New rules are applied to stencil masks and explicit masks when calculating the
|
||||
optimal page DPI for rendering. {issue}`1362`
|
||||
- Fixed attempts to use an incompatible jbig2.EXE provided by TeX Live. {issue}`1363`
|
||||
|
||||
## v16.4.2
|
||||
|
||||
- Fixed order of filenames passed to Ghostscript for PDF/A generation. {issue}`1359`
|
||||
- Suppressed missing jbig2dec warning message. {issue}`1358`
|
||||
- Fixed calculation of image size when soft mask dimensions don't match image
|
||||
dimension. {issue}`1351`
|
||||
- Several fixes to documentation. Thanks to users Iris and JoKalliauer
|
||||
who contributed these changes.
|
||||
- Fixed error on processing PDFs that are missing certain image metadata. {issue}`1315`
|
||||
|
||||
## v16.4.1
|
||||
|
||||
- Fixed calculation of image printed area (used in finding weighted DPI for OCR).
|
||||
{issue}`1334`
|
||||
- Fixed "NotImplementedError: not sure how to get colorspace" error
|
||||
messages in logs which simply records a failure to optimize images with
|
||||
print production colorspaces. {issue}`1315`
|
||||
|
||||
## v16.4.0
|
||||
|
||||
- Selecting the `osd` and `equ` pseudo-languages with `-l/--language` now
|
||||
exits with an error when using Tesseract OCR, because these are not
|
||||
regular Tesseract languages but implementation details implemented.
|
||||
Using them can cause Tesseract to crash.
|
||||
- The hOCR renderer is more tolerant of extra whitespace in input files.
|
||||
- watcher.py now changes the output file extension to .pdf when the input is not
|
||||
.pdf.
|
||||
- Improved handling of PDFs that contain circularly referenced Form XObjects.
|
||||
{issue}`1321`
|
||||
- Fixed Alpine Docker image for ARM64, which was not building correctly.
|
||||
- Docker images now use pikepdf 9.0.0.
|
||||
- Prevent use of Tesseract OCR 5.4.0, a version with known regressions.
|
||||
- Disabled progressbar for "Linearizing" when `--no-progress-bar` set.
|
||||
- Fixed some tests that warn about missing JBIG2 decoding via pikepdf, by
|
||||
installing the necessary libraries during tests.
|
||||
|
||||
## v16.3.1
|
||||
|
||||
- Fixed a test suite failure with Ghostscript 10.03.0+. {issue}`1316`
|
||||
- Fixed an issue with the presentation of the "OCR" progress bar. {issue}`1313`
|
||||
|
||||
## v16.3.0
|
||||
|
||||
- Fixed progress bar not displaying for Ghostscript PDF/A conversion. {issue}`1313`
|
||||
- Added progress bar for linearization. {issue}`1313`
|
||||
- If `--rotate-pages-threshold` issued without `--rotate-pages` we now exit with
|
||||
an error since the user likely intended to use `--rotate-pages`. {issue}`1309`
|
||||
- If Tesseract hOCR gives an invalid line box, print an error message instead of
|
||||
exiting with an error. {issue}`1312`
|
||||
|
||||
## v16.2.0
|
||||
|
||||
- Fixed issue 'NoneType' object has no attribute 'get' when optimizing certain PDFs.
|
||||
{issue}`1293,1271`
|
||||
- Switched formatting from black to ruff.
|
||||
- Added support for sending sidecar output to io.BytesIO.
|
||||
- Added support for converting HEIF/HEIC images (the native image of iPhones and
|
||||
some other devices) to PDFs, when the appropriate pi-hief library is installed.
|
||||
This library is marked as a dependency, but maintainers may opt out if needed.
|
||||
- We now default to downsampling large images that would exceed Tesseract's internal
|
||||
limits, but only if it cause processing to fail. Previously, this behavior only
|
||||
occurred if specifically requested on command line. It can still be configured
|
||||
and disabled. See the --tesseract command line options.
|
||||
- Added Macports install instructions. Thanks @akierig.
|
||||
- Improved logging output when an unexpected error occurs while trying to obtain
|
||||
the version of a third party program.
|
||||
|
||||
## v16.1.2
|
||||
|
||||
- Fixed test suite failure when using Ghostscript 10.3.
|
||||
- Other minor corrections.
|
||||
|
||||
## v16.1.1
|
||||
|
||||
- Fixed PyPy 3.10 support.
|
||||
|
||||
## v16.1.0
|
||||
|
||||
- Improved hOCR renderer is now default for left to right languages.
|
||||
- Improved handling of rotated pages. Previously, OCR text might be missing for
|
||||
pages that were rotated with a /Rotate tag on the page entry.
|
||||
- Improved handling of cropped pages. Previously, in some cases a page with a
|
||||
crop box would not have its OCR applied correctly and misalignment between
|
||||
OCR text and visible text coudl occur.
|
||||
- Documentation improvements, especially installation instructions for less
|
||||
common platforms.
|
||||
|
||||
## v16.0.4
|
||||
|
||||
- Fixed some issues for left-to-right text with the new hOCR renderer. It is still
|
||||
not default yet but will be made so soon. Right-to-left text is still in progress.
|
||||
- Added an error to prevent use of several versions of Ghostscript that seem
|
||||
corrupt existing text in input PDFs. Newly generated OCR is not affected.
|
||||
For best results, use Ghostscript 10.02.1 or newer, which contains the fix
|
||||
for the issue.
|
||||
|
||||
## v16.0.3
|
||||
|
||||
- Changed minimum required Ghostscript to 9.54, to support users of RHEL 9 and its
|
||||
derivatives, since that is the latest version available there.
|
||||
- Removed warning message about CVE-2023-43115, on the assumption that most
|
||||
distributions have backported the patch by now.
|
||||
|
||||
## v16.0.2
|
||||
|
||||
- Temporarily changed PDF text renderer back to sandwich by default to address
|
||||
regressions in macOS Preview.
|
||||
|
||||
## v16.0.1
|
||||
|
||||
- Fixed text rendering issue with new hOCR text renderer - extraneous byte order
|
||||
marks.
|
||||
- Tightened dependencies.
|
||||
|
||||
## v16.0.0
|
||||
|
||||
- Added OCR text renderer, combined the best ideas of Tesseract's PDF
|
||||
generator and the older hOCR transformer renderer. The result is a hopefully
|
||||
permanent fix for wordssmushedtogetherwithoutspaces issues in extracted text,
|
||||
better registration/position of text on skewed baselines {issue}`1009`,
|
||||
fixes to character output when the German Fraktur script is used {issue}`1191`,
|
||||
proper rendering of right to left languages (Arabic, Hebrew, Persian) {issue}`1157`.
|
||||
Asian languages may still have excessive word breaks compared to expectations.
|
||||
The new renderer is the default; the old sandwich renderer is still available
|
||||
using `--pdf-renderer sandwich`; the old hOCR renderer is no more.
|
||||
- The `ocrmypdf.hocrtransform` API has changed substantially.
|
||||
- Support for Python 3.9 has been dropped. Python 3.10+ is now required.
|
||||
- pikepdf >= 8.8.0 is now required.
|
||||
|
||||
@@ -0,0 +1,433 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# v17
|
||||
|
||||
## v17.9.0
|
||||
|
||||
- OCRmyPDF now uses any Noto font installed on the system, not just the two
|
||||
dozen script families it knows by name ({issue}`1722`). Previously a document
|
||||
in, say, Cherokee or Vai was rendered with the glyphless fallback font even
|
||||
though the matching font was installed — a common situation on macOS, which
|
||||
ships around a hundred script-specific Noto faces. When the named fonts
|
||||
cannot cover a word, OCRmyPDF now searches the installed fonts for one that
|
||||
can.
|
||||
- The "no installed font has glyphs" warning now names the characters it could
|
||||
not render, with their codepoints and Unicode names, so it is clear which
|
||||
font to install. Text that mixes scripts no single font covers is now
|
||||
reported as such, instead of advising the user to install fonts they may
|
||||
already have.
|
||||
- Fixed the macOS font installation instructions, which recommended a Homebrew
|
||||
package (`font-noto`) that does not exist ({issue}`1722`). Homebrew has no
|
||||
single Noto package; each family is a separate cask. The Fedora package name
|
||||
was also corrected to `google-noto-fonts-all`.
|
||||
- Font providers may now implement the optional `GlyphSearchingFontProvider`
|
||||
protocol to participate in coverage-based font search.
|
||||
- Fixed `--jpeg-quality`/`--jpg-quality` having no effect on the CLI: the
|
||||
value was silently dropped before reaching the optimizer, which then
|
||||
always used its own built-in default JPEG quality regardless of what was
|
||||
requested ({issue}`1723`). The same bug affected the Python API's
|
||||
`jpg_quality` parameter. `ocrmypdf.ocr()` now accepts `jpeg_quality`
|
||||
(matching the CLI flag name) as the canonical parameter; `jpg_quality`
|
||||
still works but is deprecated.
|
||||
- Hardened PDF parsing against malformed (non-dictionary) `/Resources`,
|
||||
`/XObject`, and `/FontDescriptor` entries, which previously crashed
|
||||
`ocrmypdf.ocr()` with `AttributeError`/`TypeError`/`ValueError` on
|
||||
otherwise-processable files, both during PDF/A font scanning and general
|
||||
image scanning ({issue}`1713`). Thanks @mvanhorn for the initial fix.
|
||||
- Release process improvements: fixed a CI bug where every push to main
|
||||
after a release was tagged would incorrectly revert the just-published
|
||||
GitHub release back to draft status.
|
||||
|
||||
## v17.8.1
|
||||
|
||||
- Improved the `--tesseract-pagesegmode` help text to point to
|
||||
`tesseract --help-extra`, since Tesseract 5.5.2 moved the page segmentation
|
||||
mode documentation there from `tesseract --help`. Thanks @sokai.
|
||||
- Internal refactoring: completed a project-wide mypy type-checking pass
|
||||
(`--check-untyped-defs` is now enabled, and the mypy pre-commit hook is now
|
||||
blocking rather than advisory), fixing several latent edge-case bugs
|
||||
surfaced along the way.
|
||||
- Release process improvements: migrated from pre-commit to prek for local
|
||||
git hooks, and added a dedicated lint job to CI.
|
||||
- Improved typing strictness for `Path`.
|
||||
|
||||
## v17.8.0
|
||||
|
||||
- `--output-type auto` (the default) again produces PDF/A whenever it can,
|
||||
matching OCRmyPDF 16's "PDF/A by default" behavior. It first tries the fast
|
||||
Ghostscript-free conversion (validated by veraPDF when available) and now
|
||||
falls back to Ghostscript when that cannot produce PDF/A, only emitting a
|
||||
regular PDF when even Ghostscript cannot safely convert (for example, an
|
||||
input with non-embedded CID/CJK fonts, per {issue}`1561`). A consequence is
|
||||
that the default path may once again invoke Ghostscript, which is slower and
|
||||
may transcode images; use `--output-type pdf` to skip PDF/A conversion
|
||||
entirely.
|
||||
- Fixed detection of veraPDF 1.30.0 and newer: recent builds print JVM
|
||||
warnings before their version string, which caused OCRmyPDF to report
|
||||
veraPDF as unavailable and skip the fast PDF/A path.
|
||||
- OCRmyPDF no longer silently corrupts a non-embedded CID (CJK) text layer when
|
||||
producing PDF/A ({issue}`1561`). PDF/A requires all fonts to be embedded, so
|
||||
Ghostscript substitutes and re-embeds non-embedded CID fonts — such as the OCR
|
||||
text layer Adobe Acrobat adds to scanned CJK documents — which mangles the
|
||||
text and destroys searchability. OCRmyPDF now detects non-embedded CID fonts
|
||||
before conversion: with `--output-type auto` (the default) it produces a
|
||||
regular PDF and preserves the existing text layer, and with an explicit
|
||||
`--output-type pdfa*` it stops with an error rather than emit corrupted
|
||||
output. Use `--output-type pdf` to keep the text layer, or `--force-ocr` to
|
||||
rebuild it with embedded fonts.
|
||||
- Writing the output PDF to standard output (`ocrmypdf input.pdf -`) is now
|
||||
protected against corruption at the operating system level. Previously
|
||||
OCRmyPDF relied on no in-process code — third-party libraries, plugins, or
|
||||
stray `print()` calls — ever writing to stdout; a single accidental write
|
||||
would silently corrupt the PDF. The command line program now saves the real
|
||||
stdout at startup, before plugins are loaded or any worker process/thread is
|
||||
started, and redirects file descriptor 1 to stderr, so that only OCRmyPDF's
|
||||
final PDF output can reach stdout. A consequence is that a plugin which
|
||||
intentionally prints to stdout will have that output redirected to stderr.
|
||||
- Added the public API function {func}`ocrmypdf.configure_stdout_protection`,
|
||||
which installs this same protection. Like {func}`ocrmypdf.configure_logging`,
|
||||
it is optional and intended for callers that want command-line-like behavior;
|
||||
applications that manage their own standard output should not call it.
|
||||
- Fixed an uncaught `UnicodeDecodeError` when processing a PDF whose
|
||||
`/DocumentInfo` dictionary contains a `/Name` key encoded in Latin-1 (or
|
||||
another non-UTF-8 encoding), such as `/Saks#e5r`. `repair_docinfo_nuls` now
|
||||
treats such a block as malformed, logs a message, and continues instead of
|
||||
crashing the pipeline ({issue}`1540`). Current pikepdf releases tolerate these
|
||||
keys by surrogate-escaping them, but older versions raised while iterating the
|
||||
dictionary.
|
||||
|
||||
## v17.7.1
|
||||
|
||||
- Fixed a severe, Windows-specific performance regression in the "Scanning
|
||||
contents" phase, most visible with `--redo-ocr` ({issue}`1662`). Since
|
||||
v16.4.3, OCRmyPDF forced pdfminer's read buffer to 256 MiB to work around a
|
||||
pdfminer bug that mishandled tokens split across the buffer boundary
|
||||
({issue}`1361`). On Windows, CPython's `BufferedReader.read()` eagerly
|
||||
allocates a buffer of the requested size on every read, so the oversized
|
||||
buffer made each of pdfminer's thousands of reads cost tens of milliseconds
|
||||
(this allocation is lazy, and effectively free, on Linux). The underlying
|
||||
pdfminer bug was fixed upstream in pdfminer.six 20250327
|
||||
([#1030](https://github.com/pdfminer/pdfminer.six/pull/1030)), with a
|
||||
follow-up for tokens split across streams in 20260107
|
||||
([#1158](https://github.com/pdfminer/pdfminer.six/pull/1158)), so the
|
||||
workaround has been removed and the minimum pdfminer.six version raised to
|
||||
20260107.
|
||||
- The font discovery used to build the OCR text layer now finds variable fonts
|
||||
such as `NotoSansArabic[wdth,wght].ttf`, the form shipped by Homebrew casks
|
||||
and current Google Fonts releases. Previously only static `-Regular.ttf`/`.otf`
|
||||
files were matched, so users who had installed the correct Noto font still got
|
||||
the glyphless fallback and a "No font found" warning ({issue}`1652`).
|
||||
- Font discovery is now language-aware for CJK: each Chinese, Japanese, and
|
||||
Korean language maps to its own per-language Noto family (NotoSansSC, TC, HK,
|
||||
JP, KR), with the pan-CJK super font kept as a shared fallback, since the
|
||||
per-language fonts are region subsets that may lack glyphs from other scripts.
|
||||
- The warning shown when no installed font has glyphs for some text was reworded
|
||||
to explain the consequence — the text is still added as a searchable, copyable
|
||||
layer but appears blank when highlighted in a viewer — and to name the specific
|
||||
font family to install.
|
||||
|
||||
## v17.7.0
|
||||
|
||||
- The Docker images now run as a non-root user (`app`, uid/gid 1000) by default
|
||||
rather than as root, as a defense-in-depth measure. If you bind-mount a
|
||||
directory for input and output, you may now need to add a `--user` argument so
|
||||
the container can write to it; the correct value differs for rootless Docker,
|
||||
Podman, and rootful Docker, and is described in the Docker documentation.
|
||||
Piping the input and output through stdin/stdout still works with no
|
||||
permission setup.
|
||||
- The Docker images now default their working directory to `/data`, so files in
|
||||
a directory mounted there can be given as relative paths without an explicit
|
||||
`--workdir`.
|
||||
- The Ubuntu Docker image now installs Tesseract 5 from the Ubuntu archive
|
||||
instead of the third-party `alex-p/tesseract-ocr5` PPA, and the base images
|
||||
were updated to Ubuntu 26.04 and Alpine 3.24.
|
||||
- Fixed a missing space in the error message shown when OCRmyPDF cannot access
|
||||
its working directory inside a Docker container.
|
||||
- Updated packaged dependencies, including the optional web service stack
|
||||
(starlette, tornado, python-multipart) and cryptography.
|
||||
|
||||
## v17.6.0
|
||||
|
||||
- When the optimizer encounters an image it cannot process (for example, an
|
||||
exotic colorspace that cannot be transcoded), it now logs a concise warning
|
||||
that the image was left unchanged rather than printing an alarming
|
||||
traceback. The output file was already valid in these cases; only the
|
||||
reporting was misleading. The full traceback is still available at debug
|
||||
verbosity (`-v 1`) ({issue}`846`).
|
||||
- `--pdfa-image-compression=auto` (the default) now selects lossless image
|
||||
compression at `-O0` so Ghostscript no longer transcodes lossless images to
|
||||
JPEG during PDF/A generation. At `-O1` and above, `auto` continues to defer
|
||||
to Ghostscript's heuristic, which may recompress images lossily. `-O1` (the
|
||||
default level) is kept as a historical exception because coercing it to
|
||||
lossless can substantially bloat output; users who want guaranteed lossless
|
||||
image handling should pass `--pdfa-image-compression=lossless` or use `-O0`
|
||||
({issue}`1124`).
|
||||
- `--pdfa-image-compression=lossless` now passes existing JPEG images through
|
||||
unchanged rather than re-encoding them with a lossless codec. Re-encoding an
|
||||
already-lossy JPEG losslessly cannot recover quality and only inflates the
|
||||
file, so JPEGs are preserved while non-JPEG images are encoded losslessly.
|
||||
- OCRmyPDF now validates and repairs malformed page-boundary boxes
|
||||
(``/MediaBox``, ``/CropBox``, ``/TrimBox``, ``/ArtBox``, ``/BleedBox``) in its
|
||||
input, following the PDF 2.0 specification. Coordinates written in invalid
|
||||
exponential notation are reinterpreted ({issue}`1398`); rectangles whose
|
||||
corners are given in reversed order are normalized, which previously crashed
|
||||
with ``NegativeDimensionError`` ({issue}`1526`); and a crop/trim/art/bleed box
|
||||
that falls outside the MediaBox is clamped to their intersection, or discarded
|
||||
when that intersection is empty, which previously produced an output with a
|
||||
zero-height effective page that some viewers refused to open ({issue}`1400`).
|
||||
When a box is discarded, clamped, or reinterpreted, OCRmyPDF logs a warning
|
||||
recommending visual inspection of the output. Thanks @ajdlinux for the initial
|
||||
fix in PR #1691.
|
||||
- OCRmyPDF now discards an embedded Adobe full-text search index
|
||||
(``/Root/PieceInfo/SearchIndex``) from its output. This proprietary index,
|
||||
produced by Acrobat's "Embed Index" feature, is read only by Adobe Acrobat;
|
||||
other viewers ignore it and search the text on the fly. Because any change to
|
||||
a PDF invalidates the index, retaining it after OCRmyPDF rewrites the document
|
||||
would leave a stale index that returns incorrect search results in Acrobat.
|
||||
Modern viewers rebuild a search index on demand, so there is no loss of
|
||||
search capability.
|
||||
- OCRmyPDF now discards embedded per-page thumbnail images (the optional
|
||||
``/Thumb`` image XObject on a page) from its output. OCRmyPDF alters page
|
||||
appearance (deskew, clean, rasterize, re-render) and plugins may edit pages
|
||||
arbitrarily, so a retained thumbnail would be stale and no longer match its
|
||||
page. Embedded thumbnails are a navigation aid that modern viewers generate
|
||||
on demand, so there is no loss of functionality.
|
||||
- Fixed a regression in OCR quality for PDFs that paint a 1-bit image mask
|
||||
(stencil) with a gray or colored fill color. Previously such pages were
|
||||
rasterized as 1-bit black-and-white before OCR, so Ghostscript dithered
|
||||
mid-tone text into an unreadable stipple and Tesseract failed to recognize
|
||||
it. The rasterizer now inspects the fill color used to paint a mask and
|
||||
promotes the page to grayscale or full color as needed, so the distinction
|
||||
is preserved for the OCR engine. This applies to both the Ghostscript and
|
||||
pypdfium rasterizers. {issue}`1688`
|
||||
- The default 1-bit raster device for Ghostscript is now ``pngmonod``
|
||||
(error-diffusion) instead of ``pngmono`` (ordered dithering). It produces
|
||||
better input for OCR on faint or anti-aliased scans at negligible cost and
|
||||
no change to output file size, since the rasterized image is an
|
||||
intermediate that is discarded after OCR.
|
||||
- When rasterizing pages with Ghostscript, OCRmyPDF now enables text and
|
||||
graphics anti-aliasing (``-dTextAlphaBits=4 -dGraphicsAlphaBits=4``) for the
|
||||
grayscale and color raster devices. Ghostscript 10.x renders aliased glyphs
|
||||
that OCR frequently misreads as extra word breaks or substituted characters;
|
||||
anti-aliasing materially improves OCR accuracy on the Ghostscript
|
||||
rasterization path, especially for small fonts at moderate resolution. The
|
||||
1-bit monochrome devices are unaffected, since they perform their own
|
||||
anti-aliased downscaling and older Ghostscript versions reject alpha-bit
|
||||
options on them. Note that the default rasterizer (``--rasterizer auto``)
|
||||
prefers pypdfium2, which already anti-aliases; this change benefits users who
|
||||
select ``--rasterizer ghostscript`` or do not have pypdfium2 installed.
|
||||
OCRmyPDF now also logs which rasterizer rendered each page at debug verbosity
|
||||
(``-v 1``), and the ``--rasterizer`` help text explains the OCR-quality
|
||||
trade-off, to make such reports easier to diagnose. {issue}`1439`
|
||||
- When Tesseract reports a page with many diacritics, OCRmyPDF still logs its
|
||||
interpreted "lots of diacritics - possibly poor OCR" hint, but now also emits
|
||||
Tesseract's raw message at debug verbosity (``-v 1``) so the original wording
|
||||
is available for diagnosis. {issue}`1566`
|
||||
- Added ``--mode strip``, which removes the invisible OCR text layer from a PDF
|
||||
in place. Unlike ``--ocr-engine none --force-ocr``, it does not rasterize the
|
||||
page, so images and visible content are preserved unchanged and the output is
|
||||
smaller rather than larger. Only text drawn as invisible (PDF text render mode
|
||||
3) is removed; some OCR engines -- and OCRmyPDF v2.2 and earlier -- express
|
||||
text as visible glyphs covered by an opaque image, and that text cannot be
|
||||
removed this way. {issue}`1435`
|
||||
|
||||
## v17.5.0
|
||||
|
||||
- Added support for the ``end`` alias in ``--pages``, denoting the last page
|
||||
of the document. For example, ``--pages 3-end`` OCRs from page 3 through
|
||||
the final page. {issue}`1615`
|
||||
- Added ``--ghostscript-jpeg-quality`` and ``--ghostscript-jpeg-maxdpi``
|
||||
advanced options for tuning Ghostscript's PDF/A output. The optimizer's
|
||||
``--jpeg-quality`` remains the recommended file-size control.
|
||||
- Fixed pypdfium2 rasterizer clipping content when the CropBox was smaller
|
||||
than the MediaBox (e.g. JSTOR or cropped PDFs). {issue}`1685`
|
||||
- Fixed Form XObject cycle detection in the optimizer's image xref scan.
|
||||
Self-referential or DAG-shaped Form graphs (notably from PowerPoint
|
||||
exports) previously produced floods of recursion warnings and could hang
|
||||
for minutes. {issue}`1321`
|
||||
- Tesseract config errors are now surfaced as ``TesseractConfigError`` with
|
||||
actionable guidance, instead of crashing later with a confusing
|
||||
``FileNotFoundError`` on the missing hOCR output. {issue}`1687`
|
||||
- Refreshed the Chinese README translation. Thanks @cislunarspace.
|
||||
- Internal refactoring of the ``_exec`` and ``subprocess`` modules to
|
||||
separate probing from execution.
|
||||
- CI dependency updates.
|
||||
|
||||
## v17.4.2
|
||||
|
||||
- Fixed Python API unconditionally overriding ``PIL.Image.MAX_IMAGE_PIXELS``
|
||||
when the caller did not explicitly set ``max_image_mpixels``. Host
|
||||
applications (e.g. Paperless-NGX) that configure the PIL limit before
|
||||
invoking ``ocrmypdf.ocr()`` now have their setting respected. The CLI
|
||||
default of 250 megapixels is unchanged. {issue}`1665`
|
||||
- Updated uv.lock to avoid pinning a vulnerable version of Pillow. {issue}`1666`
|
||||
|
||||
## v17.4.1
|
||||
|
||||
- Fixed RTL text extraction order in the fpdf2 renderer. Arabic lam-alef
|
||||
ligatures and other multi-character CMap entries were garbled by the bidi
|
||||
algorithm during text extraction. {issue}`1655`
|
||||
- Fixed ``work_folder`` not being set in ``PdfContext`` options when using
|
||||
the Python API. Thanks @bluebox-steven. {issue}`1613`
|
||||
- Updated Ghostscript JPEG corruption warning to include the detected version
|
||||
number, confirming the bug persists in Ghostscript 10.7.0.
|
||||
- Internal refactoring.
|
||||
- CI dependency updates.
|
||||
|
||||
## v17.4.0
|
||||
|
||||
- Added ``--no-overwrite`` / ``-n`` option to prevent overwriting output files.
|
||||
If the destination file already exists, OCRmyPDF exits with code 5
|
||||
(``OutputFileAccessError``). {issue}`1642`
|
||||
- Fixed text layer stretching in the fpdf2 renderer for widely-spaced words.
|
||||
The horizontal scaling (Tz) was incorrectly stretched to fill inter-word gaps
|
||||
instead of relying on Td positioning, causing text selection to highlight far
|
||||
beyond the actual word boundaries. {issue}`1635`
|
||||
- Fixed ``optimize=2`` or ``optimize=3`` crash when using the Python API without
|
||||
explicitly setting ``jpg_quality`` or ``png_quality``. {issue}`1641`
|
||||
- Fixed ``verapdf`` availability check crashing with ``NotADirectoryError`` on
|
||||
some platforms. {issue}`1638`
|
||||
|
||||
## v17.3.0
|
||||
|
||||
- Fixed Python API ignoring the ``language`` parameter, always defaulting to
|
||||
``eng``. The API now correctly maps ``language`` to OcrOptions ``languages``
|
||||
and splits ``+``-separated codes (e.g. ``eng+deu``) to match CLI behavior.
|
||||
{issue}`1640`
|
||||
- Fixed Python API producing empty OCR output because ``tesseract_timeout``
|
||||
defaulted to 0, causing Tesseract to time out immediately. The default is
|
||||
now ``None``, falling back to the plugin's 180-second timeout. {issue}`1636`
|
||||
- Fixed OCR text layer displacement on PDFs with non-zero MediaBox origins
|
||||
(e.g. JSTOR or cropped PDFs). The coordinate transformation matrix is now
|
||||
always computed, not skipped when rotation is zero. {issue}`1630`
|
||||
- Restored image overlay support (``--image``) for the hocrtransform tool,
|
||||
enabling sandwich PDF output with the fpdf2 renderer. {issue}`1634`
|
||||
- Docker: updated Alpine base image to 3.23.
|
||||
- Documentation restructured into per-major-version release notes files.
|
||||
- Release process improvements.
|
||||
|
||||
## v17.2.0
|
||||
|
||||
- Fixed incorrect word spacing in poppler-based PDF viewers and tools (Evince,
|
||||
pdftotext, and others) where words on the same line appeared separated by
|
||||
double newlines. This works around a poppler bug where Tz (horizontal scaling)
|
||||
is not carried across BT/ET boundaries. {issue}`1632`
|
||||
- Fixed OCR text layer being visible instead of invisible due to incorrect fpdf2
|
||||
text rendering mode attribute. This caused OCR text to appear when images were
|
||||
removed from the PDF. {issue}`1631`
|
||||
- Fixed OCR text layer misalignment with non-zero mediabox origins, which
|
||||
affected cropped PDFs and JSTOR PDFs generated by iText. The ``--redo-ocr``
|
||||
mode would shift text vertically on these files. {issue}`1630`
|
||||
- Fixed Ghostscript rasterization failure with very low DPI values (below 10).
|
||||
OCRmyPDF now renders at a minimum of 10 DPI and resizes the output to match
|
||||
the originally requested dimensions. {issue}`1612`
|
||||
|
||||
## v17.1.0
|
||||
|
||||
- Added `--tagged-pdf-mode` to allow skipping the TaggedPDF error message, if desired.
|
||||
- Fixed an issue where deflated JPEGs (FlateDecode + DCTDecode) were counted as
|
||||
lossless images for the purpose of determining whether to compress to JPEG,
|
||||
causing file size inflation with some workflows (`--mode force` in particular).
|
||||
|
||||
## v17.0.1
|
||||
|
||||
- Fixed output file size inflation when using pypdfium as rasterizer and force-ocr
|
||||
mode.
|
||||
|
||||
## v17.0.0
|
||||
|
||||
**Breaking changes**
|
||||
|
||||
- **Plugin interface migration**: Plugin hooks now receive `OcrOptions` objects instead of
|
||||
`argparse.Namespace` objects. Most plugins will continue working due to duck-typing
|
||||
compatibility, but plugin developers should update their type hints from `Namespace`
|
||||
to `OcrOptions`.
|
||||
- Built-in plugins no longer modify options in-place, improving immutability and
|
||||
code clarity.
|
||||
- **Lossy JBIG2 removed**: The `--jbig2-lossy` and `--jbig2-page-group-size` options have been
|
||||
removed due to well-documented risks of character substitution errors. These options are now
|
||||
deprecated and will emit warnings if used. Only lossless JBIG2 compression is supported.
|
||||
- **PDF/A output behavior change**: If neither Ghostscript nor verapdf is installed,
|
||||
`--output-type auto` (the new default) will produce a standard PDF instead of PDF/A. This is
|
||||
a change from previous versions where Ghostscript was required and PDF/A was always produced.
|
||||
This configuration is rare but users should be aware of the change.
|
||||
|
||||
**New features**
|
||||
|
||||
- **pypdfium2 rasterizer**: Added optional pypdfium2-based PDF rasterization plugin as an
|
||||
alternative to Ghostscript for page rendering. Use `--rasterizer pypdfium` to enable
|
||||
(requires `pip install pypdfium2`). The default `--rasterizer auto` prefers pypdfium when
|
||||
available and falls back to Ghostscript.
|
||||
- **Pluggable OCR engines**: New `--ocr-engine` option allows selecting OCR engines:
|
||||
- `auto` (default): Uses Tesseract
|
||||
- `tesseract`: Explicit Tesseract selection
|
||||
- `none`: Skip OCR entirely for PDF processing-only workflows
|
||||
|
||||
This prepares the foundation for future third-party OCR engine plugins.
|
||||
- **Smart PDF/A conversion**: New `--output-type auto` (now the default) produces best-effort
|
||||
PDF/A output without requiring Ghostscript when the verapdf validator is available. Falls back
|
||||
to traditional Ghostscript conversion when needed.
|
||||
- **verapdf integration**: Added optional verapdf validation for fast PDF/A conversion. When
|
||||
available, OCRmyPDF attempts speculative PDF/A conversion using pikepdf, validates with verapdf,
|
||||
and skips Ghostscript if validation passes.
|
||||
- **Optional Ghostscript**: As a consequence of the changes above, Ghostscript is no longer a required dependency. It is optional.
|
||||
- **fpdf2 text renderer**: Replaced legacy hOCR text renderer with new fpdf2-based implementation,
|
||||
providing better multilingual support and more accurate text positioning.
|
||||
- **Improved Occulta glyphless font**: The new Occulta font provides better handling of
|
||||
zero-width markers and double-width CJK characters for accurate text layer positioning.
|
||||
- **Expanded multilingual font support**: Added FontProvider infrastructure with language-aware
|
||||
font selection for Devanagari (Hindi, Sanskrit, Marathi, Nepali), CJK (Chinese, Japanese,
|
||||
Korean), Arabic script, and many other scripts. System font discovery reduces package size.
|
||||
- **Simplified mode selection**: New `--mode` (`-m`) argument consolidates processing options:
|
||||
- `default`: Error if text is found (standard behavior)
|
||||
- `force`: Rasterize all content and run OCR (replaces `--force-ocr`)
|
||||
- `skip`: Skip pages with existing text (replaces `--skip-text`)
|
||||
- `redo`: Re-OCR pages, stripping old text layer (replaces `--redo-ocr`)
|
||||
|
||||
Legacy flags remain as silent aliases for backward compatibility.
|
||||
|
||||
**API improvements**
|
||||
|
||||
- Centralized validation logic in the `OcrOptions` Pydantic model
|
||||
- Removed scattered option mutation throughout the codebase
|
||||
- Better type safety for plugin development
|
||||
- Simplified plugin option handling
|
||||
- New `OcrElement`, `OcrClass`, and `BoundingBox` exports for OCR engine plugin developers
|
||||
- Extended `OcrEngine` ABC with `generate_ocr()` method for direct OCR tree output, eliding the need to translate a modern engine's output to hOCR or directly write to PDF.
|
||||
|
||||
**Bug fixes**
|
||||
|
||||
- Fixed double-compression of already-deflated JPEGs.
|
||||
- Fixed tesseract_cache plugin to properly handle cache misses.
|
||||
- Fixed handling of PDF page boxes (ArtBox, BleedBox) which were not being processed correctly.
|
||||
- Added thread safety lock to pypdfium plugin for concurrent operations.
|
||||
- Improved pdfminer.six compatibility with explicit word spacing.
|
||||
|
||||
**Documentation**
|
||||
|
||||
- Updated cookbook to replace deprecated `--tesseract-timeout 0` with `--ocr-engine none`.
|
||||
- Added comprehensive plugin documentation for new OCR engine framework.
|
||||
|
||||
**Dependency changes**
|
||||
|
||||
- Requires: one of `pypdfium2` or `ghostscript` for PDF rasterization (PDF to image)
|
||||
- Preferred: both
|
||||
- Requires: one of `verapdf` or `ghostscript` for PDF/A generation
|
||||
- Preferred: both
|
||||
- Recommended: `pypdfium2` for PDF rasterization (new dependency)
|
||||
- Recommended: `ghostscript` (used to be Required)
|
||||
- Recommended: Noto fonts for improved OCR text positioning
|
||||
- Optional: `verapdf` for fast PDF/A validation (new dependency)
|
||||
- Requires: `fpdf2` for text layer rendering (new dependency)
|
||||
- Recommended: replace `typer` with `cyclopts` in misc scripts (new dependency)
|
||||
- See docs/maintainers.md for details.
|
||||
|
||||
**Migration guide for plugin developers**
|
||||
|
||||
- Update imports: `from ocrmypdf._options import OcrOptions`
|
||||
- Update type hints: `def check_options(options: OcrOptions)` instead of `options: Namespace`
|
||||
- Attribute access remains unchanged: `options.languages`, `options.output_type`, etc.
|
||||
- Remove any in-place option modifications - compute values at point of use instead
|
||||
- Most existing plugins will continue working without changes due to duck-typing
|
||||
|
||||
+2
-3
@@ -16,7 +16,6 @@ from __future__ import annotations
|
||||
|
||||
import filecmp
|
||||
import logging
|
||||
import os
|
||||
import posixpath
|
||||
import shutil
|
||||
import sys
|
||||
@@ -39,7 +38,7 @@ script_dir = Path(__file__).parent
|
||||
# set archive_dir to a path for backup original documents. Leave empty if not required.
|
||||
archive_dir = "/pdfbak"
|
||||
|
||||
start_dir = Path(sys.argv[1]) if len(sys.argv) > 1 else Path(".")
|
||||
start_dir = Path(sys.argv[1]) if len(sys.argv) > 1 else Path()
|
||||
|
||||
if len(sys.argv) > 2:
|
||||
log_file = Path(sys.argv[2])
|
||||
@@ -68,7 +67,7 @@ for filename in start_dir.glob("**/*.pdf"):
|
||||
try:
|
||||
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||
except OSError:
|
||||
os.makedirs(posixpath.dirname(archive_filename))
|
||||
Path(posixpath.dirname(archive_filename)).mkdir(parents=True)
|
||||
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||
try:
|
||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Helper script for bisecting PDFs to find a page with an issue."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
|
||||
@@ -21,18 +21,18 @@ __ocrmypdf_arguments()
|
||||
--subject (set metadata)
|
||||
--keywords (set metadata)
|
||||
--rotate-pages (rotate pages to correct orientation)
|
||||
--remove-background (attempt to remove background from pages)
|
||||
--deskew (fix small horizontal alignment skew)
|
||||
--clean (clean document images before OCR)
|
||||
--clean-final (clean document images and keep result)
|
||||
--unpaper-args (a quoted string of arguments to pass to unpaper)
|
||||
--oversample (oversample images to this DPI)
|
||||
--remove-vectors (don\'t send vector objects to OCR)
|
||||
--threshold (threshold images before OCR)
|
||||
--mode (processing mode for pages with existing text)
|
||||
--force-ocr (OCR documents that already have printable text)
|
||||
--skip-text (skip OCR on any pages that already contain text)
|
||||
--redo-ocr (redo OCR on any pages that seem to have OCR already)
|
||||
--invalidate-digital-signatures (remove digital signatures from PDF)
|
||||
--tagged-pdf-mode (control behavior for Tagged PDFs)
|
||||
--skip-big (skip OCR on pages larger than this many MPixels)
|
||||
--optimize (select optimization level)
|
||||
--jpeg-quality (JPEG quality [0..100])
|
||||
@@ -42,9 +42,14 @@ __ocrmypdf_arguments()
|
||||
--pages (apply OCR to only the specified pages)
|
||||
--max-image-mpixels (image decompression bomb threshold)
|
||||
--pdf-renderer (select PDF renderer options)
|
||||
--ocr-engine (OCR engine to use)
|
||||
--rasterizer (PDF page rasterizer)
|
||||
--rotate-pages-threshold (page rotation confidence)
|
||||
--pdfa-image-compression (set PDF/A image compression options)
|
||||
--ghostscript-jpeg-quality (Ghostscript JPEG quality during PDF/A [0..100])
|
||||
--ghostscript-jpeg-maxdpi (cap Ghostscript image DPI during PDF/A)
|
||||
--fast-web-view (if file size if above this amount in MB linearize PDF)
|
||||
--continue-on-soft-render-error (continue after recoverable render errors)
|
||||
--plugin (name of plugin to import)
|
||||
--keep-temporary-files (keep temporary files (debug)
|
||||
--tesseract-config (set custom tesseract config file)
|
||||
@@ -52,6 +57,10 @@ __ocrmypdf_arguments()
|
||||
--tesseract-oem (set tesseract --oem)
|
||||
--tesseract-thresholding (set tesseract image thresholding)
|
||||
--tesseract-timeout (maximum number of seconds to wait for OCR)
|
||||
--tesseract-non-ocr-timeout (maximum seconds for non-OCR operations)
|
||||
--tesseract-downsample-large-images (downsample large images before OCR)
|
||||
--no-tesseract-downsample-large-images (do not downsample large images)
|
||||
--tesseract-downsample-above (downsample images larger than this pixel size)
|
||||
--user-words (specify location of user words file)
|
||||
--user-patterns (specify location of user patterns file)
|
||||
--no-progress-bar (disable the progress bar)
|
||||
@@ -68,7 +77,8 @@ __ocrmypdf_arguments()
|
||||
|
||||
__ocrmypdf_output-type()
|
||||
{
|
||||
local choices="pdfa (output a PDF/A (default))
|
||||
local choices="auto (best-effort PDF/A without Ghostscript (default))
|
||||
pdfa (output a PDF/A-2b)
|
||||
pdf (output a standard PDF)
|
||||
pdfa-1 (output a PDF/A-1b)
|
||||
pdfa-2 (output a PDF/A-2b)
|
||||
@@ -114,10 +124,11 @@ __ocrmypdf_optimize()
|
||||
|
||||
__ocrmypdf_pdf-renderer()
|
||||
{
|
||||
local choices="auto (auto select PDF renderer)
|
||||
hocr (use hOCR renderer)
|
||||
hocrdebug (uses hOCR renderer in debug mode, showing recognized text)
|
||||
sandwich (use sandwich renderer)"
|
||||
local choices="auto (auto select PDF renderer, uses fpdf2)
|
||||
fpdf2 (use fpdf2 renderer with full language support)
|
||||
sandwich (use sandwich renderer)
|
||||
hocr (use hOCR renderer - deprecated)
|
||||
hocrdebug (uses hOCR renderer in debug mode - deprecated)"
|
||||
|
||||
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||
|
||||
@@ -210,6 +221,58 @@ UseDeviceIndependentColor (convert with device independent color)"
|
||||
fi
|
||||
}
|
||||
|
||||
__ocrmypdf_mode()
|
||||
{
|
||||
local choices="default (error if text is found)
|
||||
force (rasterize all content and run OCR)
|
||||
skip (skip pages with existing text)
|
||||
redo (re-OCR pages, replacing old invisible text)"
|
||||
|
||||
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||
# Remove description if only one completion exists
|
||||
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||
fi
|
||||
}
|
||||
|
||||
__ocrmypdf_tagged-pdf-mode()
|
||||
{
|
||||
local choices="default (error if --mode is default, otherwise warn)
|
||||
ignore (always warn but continue processing)"
|
||||
|
||||
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||
# Remove description if only one completion exists
|
||||
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||
fi
|
||||
}
|
||||
|
||||
__ocrmypdf_ocr-engine()
|
||||
{
|
||||
local choices="auto (select best available engine)
|
||||
tesseract (use Tesseract OCR)
|
||||
none (skip OCR entirely)"
|
||||
|
||||
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||
# Remove description if only one completion exists
|
||||
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||
fi
|
||||
}
|
||||
|
||||
__ocrmypdf_rasterizer()
|
||||
{
|
||||
local choices="auto (prefer pypdfium, fall back to Ghostscript)
|
||||
ghostscript (use Ghostscript rasterizer)
|
||||
pypdfium (use pypdfium rasterizer - faster)"
|
||||
|
||||
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||
# Remove description if only one completion exists
|
||||
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||
fi
|
||||
}
|
||||
|
||||
__ocrmypdf_check_previous()
|
||||
{
|
||||
case $prev in
|
||||
@@ -241,6 +304,22 @@ __ocrmypdf_check_previous()
|
||||
__ocrmypdf_pdf-renderer
|
||||
return 0
|
||||
;;
|
||||
-m|--mode)
|
||||
__ocrmypdf_mode
|
||||
return 0
|
||||
;;
|
||||
--tagged-pdf-mode)
|
||||
__ocrmypdf_tagged-pdf-mode
|
||||
return 0
|
||||
;;
|
||||
--ocr-engine)
|
||||
__ocrmypdf_ocr-engine
|
||||
return 0
|
||||
;;
|
||||
--rasterizer)
|
||||
__ocrmypdf_rasterizer
|
||||
return 0
|
||||
;;
|
||||
--pdfa-image-compression)
|
||||
__ocrmypdf_pdfa-image-compression
|
||||
return 0
|
||||
@@ -260,7 +339,9 @@ __ocrmypdf_check_previous()
|
||||
|
||||
--title|--author|--subject|--keywords|--unpaper-args|--pages|--plugin|\
|
||||
--jpeg-quality|--png-quality|--image-dpi|--oversample|--skip-big|--max-image-mpixels|\
|
||||
--tesseract-timeout|--rotate-pages-threshold|--fast-web-view)
|
||||
--ghostscript-jpeg-quality|--ghostscript-jpeg-maxdpi|\
|
||||
--tesseract-timeout|--tesseract-non-ocr-timeout|--tesseract-downsample-above|\
|
||||
--rotate-pages-threshold|--fast-web-view)
|
||||
# argument required but no completions available
|
||||
return 0
|
||||
;;
|
||||
|
||||
@@ -11,13 +11,27 @@ complete -c ocrmypdf -s r -l rotate-pages -d "rotate pages to correct orientatio
|
||||
complete -c ocrmypdf -s d -l deskew -d "fix small horizontal alignment skew"
|
||||
complete -c ocrmypdf -s c -l clean -d "clean document images before OCR"
|
||||
complete -c ocrmypdf -s i -l clean-final -d "clean document images and keep result"
|
||||
complete -c ocrmypdf -x -l unpaper-args -d "quoted string of arguments to pass to unpaper"
|
||||
complete -c ocrmypdf -l remove-vectors -d "don't send vector objects to OCR"
|
||||
|
||||
function __fish_ocrmypdf_mode
|
||||
echo -e "default\t"(_ "error if text is found")
|
||||
echo -e "force\t"(_ "rasterize all content and run OCR")
|
||||
echo -e "skip\t"(_ "skip pages with existing text")
|
||||
echo -e "redo\t"(_ "re-OCR pages, replacing old invisible text")
|
||||
end
|
||||
complete -c ocrmypdf -x -s m -l mode -a '(__fish_ocrmypdf_mode)' -d "processing mode for pages with existing text"
|
||||
complete -c ocrmypdf -s f -l force-ocr -d "OCR documents that already have printable text"
|
||||
complete -c ocrmypdf -s s -l skip-text -d "skip OCR on any pages that already contain text"
|
||||
complete -c ocrmypdf -l redo-ocr -d "redo OCR on any pages that seem to have OCR already"
|
||||
complete -c ocrmypdf -l invalidate-digital-signatures -d "invalidate digital signatures and allow OCR to proceed"
|
||||
|
||||
function __fish_ocrmypdf_tagged_pdf_mode
|
||||
echo -e "default\t"(_ "error if --mode is default, otherwise warn")
|
||||
echo -e "ignore\t"(_ "always warn but continue processing")
|
||||
end
|
||||
complete -c ocrmypdf -x -l tagged-pdf-mode -a '(__fish_ocrmypdf_tagged_pdf_mode)' -d "control behavior for Tagged PDFs"
|
||||
|
||||
complete -c ocrmypdf -s k -l keep-temporary-files -d "keep temporary files (debug)"
|
||||
|
||||
function __fish_ocrmypdf_languages
|
||||
@@ -32,7 +46,8 @@ complete -c ocrmypdf -x -s l -l language -a '(__fish_ocrmypdf_languages)' -d lan
|
||||
complete -c ocrmypdf -x -l image-dpi -d "assume this DPI if input image DPI is unknown"
|
||||
|
||||
function __fish_ocrmypdf_output_type
|
||||
echo -e "pdfa\t"(_ "output a PDF/A (default)")
|
||||
echo -e "auto\t"(_ "best-effort PDF/A without requiring Ghostscript (default)")
|
||||
echo -e "pdfa\t"(_ "output a PDF/A-2b")
|
||||
echo -e "pdf\t"(_ "output a standard PDF")
|
||||
echo -e "pdfa-1\t"(_ "output a PDF/A-1b")
|
||||
echo -e "pdfa-2\t"(_ "output a PDF/A-2b")
|
||||
@@ -42,13 +57,28 @@ end
|
||||
complete -c ocrmypdf -x -l output-type -a '(__fish_ocrmypdf_output_type)' -d "select PDF output options"
|
||||
|
||||
function __fish_ocrmypdf_pdf_renderer
|
||||
echo -e "auto\t"(_ "auto select PDF renderer")
|
||||
echo -e "hocr\t"(_ "use hOCR renderer")
|
||||
echo -e "hocrdebug\t"(_ "uses hOCR renderer in debug mode, showing recognized text")
|
||||
echo -e "auto\t"(_ "auto select PDF renderer (default, uses fpdf2)")
|
||||
echo -e "fpdf2\t"(_ "use fpdf2 renderer with full language support")
|
||||
echo -e "sandwich\t"(_ "use sandwich renderer")
|
||||
echo -e "hocr\t"(_ "use hOCR renderer (deprecated)")
|
||||
echo -e "hocrdebug\t"(_ "uses hOCR renderer in debug mode (deprecated)")
|
||||
end
|
||||
complete -c ocrmypdf -x -l pdf-renderer -a '(__fish_ocrmypdf_pdf_renderer)' -d "select PDF renderer options"
|
||||
|
||||
function __fish_ocrmypdf_ocr_engine
|
||||
echo -e "auto\t"(_ "select best available engine (default)")
|
||||
echo -e "tesseract\t"(_ "use Tesseract OCR")
|
||||
echo -e "none\t"(_ "skip OCR entirely")
|
||||
end
|
||||
complete -c ocrmypdf -x -l ocr-engine -a '(__fish_ocrmypdf_ocr_engine)' -d "OCR engine to use"
|
||||
|
||||
function __fish_ocrmypdf_rasterizer
|
||||
echo -e "auto\t"(_ "prefer pypdfium, fall back to Ghostscript (default)")
|
||||
echo -e "ghostscript\t"(_ "use Ghostscript rasterizer")
|
||||
echo -e "pypdfium\t"(_ "use pypdfium rasterizer (faster)")
|
||||
end
|
||||
complete -c ocrmypdf -x -l rasterizer -a '(__fish_ocrmypdf_rasterizer)' -d "PDF page rasterizer"
|
||||
|
||||
function __fish_ocrmypdf_optimize
|
||||
echo -e "0\t"(_ "do not optimize")
|
||||
echo -e "1\t"(_ "do safe, lossless optimizations (default)")
|
||||
@@ -72,6 +102,8 @@ function __fish_ocrmypdf_pdfa_compression
|
||||
echo -e "lossless\t"(_ "convert color and grayscale images to lossless (PNG)")
|
||||
end
|
||||
complete -c ocrmypdf -x -l pdfa-image-compression -a '(__fish_ocrmypdf_pdfa_compression)' -d "set PDF/A image compression options"
|
||||
complete -c ocrmypdf -x -l ghostscript-jpeg-quality -d "Ghostscript JPEG quality during PDF/A [0..100]"
|
||||
complete -c ocrmypdf -x -l ghostscript-jpeg-maxdpi -d "cap Ghostscript image DPI during PDF/A"
|
||||
|
||||
complete -c ocrmypdf -x -s j -l jobs -d "how many worker processes to use"
|
||||
complete -c ocrmypdf -x -l title -d "set metadata"
|
||||
@@ -124,11 +156,17 @@ end
|
||||
complete -c ocrmypdf -x -l tesseract-thresholding -a '(__fish_ocrmypdf_tesseract_thresholding)' -d "set tesseract thresholding method (needs Tesseract 5.x)"
|
||||
|
||||
complete -c ocrmypdf -x -l tesseract-timeout -d "maximum number of seconds to wait for OCR"
|
||||
complete -c ocrmypdf -x -l tesseract-non-ocr-timeout -d "maximum seconds to wait for non-OCR operations"
|
||||
complete -c ocrmypdf -l tesseract-downsample-large-images -d "downsample large images before OCR"
|
||||
complete -c ocrmypdf -l no-tesseract-downsample-large-images -d "do not downsample large images"
|
||||
complete -c ocrmypdf -x -l tesseract-downsample-above -d "downsample images larger than this pixel size"
|
||||
complete -c ocrmypdf -x -l rotate-pages-threshold -d "page rotation confidence"
|
||||
|
||||
complete -c ocrmypdf -r -l user-words -d "specify location of user words file"
|
||||
complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file"
|
||||
complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF"
|
||||
complete -c ocrmypdf -l continue-on-soft-render-error -d "continue processing after recoverable render errors"
|
||||
complete -c ocrmypdf -r -l plugin -d "name of plugin to import"
|
||||
|
||||
function __fish_ocrmypdf_color_conversion_strategy
|
||||
echo -e "LeaveColorUnchanged\t"(_ "do not convert color spaces (default)")
|
||||
|
||||
@@ -6,12 +6,19 @@ services:
|
||||
ocrmypdf:
|
||||
restart: always
|
||||
container_name: ocrmypdf
|
||||
image: jbarlow83/ocrmypdf
|
||||
image: jbarlow83/ocrmypdf-alpine
|
||||
volumes:
|
||||
- "/media/scan:/input"
|
||||
- "/mnt/scan:/output"
|
||||
environment:
|
||||
- OCR_OUTPUT_DIRECTORY_YEAR_MONTH=0
|
||||
# The image runs as the non-root "app" user (uid 1000) by default. The
|
||||
# correct value here depends on your runtime, so that the watcher can write
|
||||
# to the /output bind mount and the files end up owned by you:
|
||||
# rootful Docker -> your host uid:gid
|
||||
# rootless Docker -> "0:0" (container root maps to your host user)
|
||||
# Podman -> your host uid:gid, plus `userns_mode: "keep-id"`
|
||||
# See docs/docker.md ("Bind-mounted volumes") for the reasoning.
|
||||
user: "<SET TO YOUR USER ID>:<SET TO YOUR GROUP ID>"
|
||||
entrypoint: python3
|
||||
command: watcher.py
|
||||
command: /app/watcher.py
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
||||
# SPDX-License-Identifier: MIT
|
||||
#
|
||||
# Batch web interface for OCRmyPDF.
|
||||
#
|
||||
# docker compose -f misc/docker-compose.webui.yml up --build
|
||||
#
|
||||
# Then open http://localhost:8000/
|
||||
#
|
||||
# There is no authentication. Run this on a trusted network, or put it behind
|
||||
# a reverse proxy that handles TLS and access control.
|
||||
---
|
||||
services:
|
||||
ocrmypdf-webui:
|
||||
build:
|
||||
context: ..
|
||||
dockerfile: .docker/Dockerfile
|
||||
image: ocrmypdf-webui
|
||||
container_name: ocrmypdf-webui
|
||||
restart: unless-stopped
|
||||
|
||||
# The image's default entrypoint is the ocrmypdf CLI; override it to start
|
||||
# the web server instead.
|
||||
entrypoint: ["/app/.venv/bin/python3", "-m", "webui"]
|
||||
|
||||
ports:
|
||||
- "8772:8000"
|
||||
|
||||
environment:
|
||||
# Files OCR'd concurrently. Each one also uses OCR_JOBS threads
|
||||
# internally, so WORKERS x OCR_JOBS should be roughly your core count.
|
||||
OCRMYPDF_WEBUI_WORKERS: "2"
|
||||
OCRMYPDF_WEBUI_OCR_JOBS: "2"
|
||||
# Limits on what a single submission may contain.
|
||||
OCRMYPDF_WEBUI_MAX_FILES: "50"
|
||||
OCRMYPDF_WEBUI_MAX_UPLOAD_MB: "500"
|
||||
# Uploads and results are deleted this many seconds after the batch was
|
||||
# submitted, whether or not they were downloaded.
|
||||
OCRMYPDF_WEBUI_BATCH_TTL_SECONDS: "3600"
|
||||
# Give up on any single file that takes longer than this.
|
||||
OCRMYPDF_WEBUI_JOB_TIMEOUT_SECONDS: "1800"
|
||||
|
||||
# Uploads and results are transient, so keep them in RAM and out of the
|
||||
# container's writable layer. Size this above the largest batch you expect;
|
||||
# drop this block to use ordinary container storage instead.
|
||||
tmpfs:
|
||||
- /var/tmp/ocrmypdf-webui:size=4g,mode=1777
|
||||
|
||||
healthcheck:
|
||||
test:
|
||||
- CMD
|
||||
- /app/.venv/bin/python3
|
||||
- "-c"
|
||||
- "import urllib.request;urllib.request.urlopen('http://127.0.0.1:8000/healthz').read()"
|
||||
interval: 30s
|
||||
timeout: 5s
|
||||
start_period: 15s
|
||||
retries: 3
|
||||
|
||||
security_opt:
|
||||
- no-new-privileges:true
|
||||
@@ -37,8 +37,8 @@ def do_column(label, suffix, d):
|
||||
env[k] = v
|
||||
args = shlex.split(
|
||||
cli.format(
|
||||
in_=os.path.join(d, "input.pdf"),
|
||||
out=os.path.join(d, f"output{suffix}.pdf"),
|
||||
in_=Path(d) / "input.pdf",
|
||||
out=Path(d) / f"output{suffix}.pdf",
|
||||
)
|
||||
)
|
||||
with st.expander("Environment variables", expanded=bool(env_text.strip())):
|
||||
@@ -106,10 +106,10 @@ def main():
|
||||
)
|
||||
)
|
||||
|
||||
doc1 = pymupdf.open(os.path.join(d, "output1.pdf"))
|
||||
doc2 = pymupdf.open(os.path.join(d, "output2.pdf"))
|
||||
doc1 = pymupdf.open(Path(d, "output1.pdf"))
|
||||
doc2 = pymupdf.open(Path(d, "output2.pdf"))
|
||||
for i, page1_2 in enumerate(zip(doc1, doc2, strict=False)):
|
||||
st.write(f"Page {i+1}")
|
||||
st.write(f"Page {i + 1}")
|
||||
page1, page2 = page1_2
|
||||
col1, col2 = st.columns(2)
|
||||
with col1, st.container(border=True):
|
||||
|
||||
+3
-4
@@ -5,7 +5,6 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from tempfile import TemporaryDirectory
|
||||
@@ -60,10 +59,10 @@ def main():
|
||||
Path(d, "2.pdf").write_bytes(pdf_bytes2)
|
||||
|
||||
with st.expander("Text"):
|
||||
doc1 = pymupdf.open(os.path.join(d, "1.pdf"))
|
||||
doc2 = pymupdf.open(os.path.join(d, "2.pdf"))
|
||||
doc1 = pymupdf.open(Path(d, "1.pdf"))
|
||||
doc2 = pymupdf.open(Path(d, "2.pdf"))
|
||||
for i, page1_2 in enumerate(zip(doc1, doc2, strict=False)):
|
||||
st.write(f"Page {i+1}")
|
||||
st.write(f"Page {i + 1}")
|
||||
page1, page2 = page1_2
|
||||
col1, col2 = st.columns(2)
|
||||
with col1, st.container(border=True):
|
||||
|
||||
@@ -23,7 +23,7 @@ def main(
|
||||
engine: Annotated[str, cyclopts.Parameter()] = 'pdftotext',
|
||||
):
|
||||
"""Compare text in PDFs."""
|
||||
with open(pdf1, 'rb') as f1, open(pdf2, 'rb') as f2:
|
||||
with pdf1.open('rb') as f1, pdf2.open('rb') as f2:
|
||||
text1 = run(
|
||||
['pdftotext', '-layout', '-', '-'],
|
||||
stdin=f1,
|
||||
|
||||
+10
-9
@@ -13,13 +13,14 @@ import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
# pylint: disable=logging-not-lazy
|
||||
|
||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
||||
script_dir = Path(os.path.realpath(__file__)).parent
|
||||
timestamp = time.strftime("%Y-%m-%d-%H%M_")
|
||||
log_file = script_dir + '/' + timestamp + 'ocrmypdf.log'
|
||||
log_file = script_dir / (timestamp + 'ocrmypdf.log')
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s %(message)s',
|
||||
@@ -33,10 +34,10 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||
logging.info(dir_name)
|
||||
os.chdir(dir_name)
|
||||
for filename in file_list:
|
||||
file_stem, file_ext = os.path.splitext(filename)
|
||||
file_stem, file_ext = Path(filename).stem, Path(filename).suffix
|
||||
if file_ext != '.pdf':
|
||||
continue
|
||||
full_path = os.path.join(dir_name, filename)
|
||||
full_path = Path(dir_name, filename)
|
||||
timestamp_ocr = time.strftime("%Y-%m-%d-%H%M_OCR_")
|
||||
filename_ocr = timestamp_ocr + file_stem + '.pdf'
|
||||
# create string for pdf processing
|
||||
@@ -52,10 +53,10 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||
'-',
|
||||
]
|
||||
logging.info(cmd)
|
||||
full_path_ocr = os.path.join(dir_name, filename_ocr)
|
||||
full_path_ocr = Path(dir_name, filename_ocr)
|
||||
with (
|
||||
open(filename, 'rb') as input_file,
|
||||
open(full_path_ocr, 'wb') as output_file,
|
||||
Path(filename).open('rb') as input_file,
|
||||
full_path_ocr.open('wb') as output_file,
|
||||
):
|
||||
proc = subprocess.run(
|
||||
cmd,
|
||||
@@ -67,8 +68,8 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||
errors='ignore',
|
||||
)
|
||||
logging.info(proc.stderr)
|
||||
os.chmod(full_path_ocr, 0o664)
|
||||
os.chmod(full_path, 0o664)
|
||||
full_path_ocr.chmod(0o664)
|
||||
full_path.chmod(0o664)
|
||||
full_path_ocr_archive = sys.argv[2]
|
||||
full_path_archive = sys.argv[2] + '/no_ocr'
|
||||
shutil.move(full_path_ocr, full_path_ocr_archive)
|
||||
|
||||
+2
-2
@@ -13,7 +13,7 @@ import logging
|
||||
import shutil
|
||||
import sys
|
||||
import time
|
||||
from enum import Enum
|
||||
from enum import StrEnum
|
||||
from pathlib import Path
|
||||
from typing import Annotated, Any
|
||||
|
||||
@@ -35,7 +35,7 @@ app = cyclopts.App(name="ocrmypdf-watcher")
|
||||
log = logging.getLogger('ocrmypdf-watcher')
|
||||
|
||||
|
||||
class LoggingLevelEnum(str, Enum):
|
||||
class LoggingLevelEnum(StrEnum):
|
||||
"""Enum for logging levels."""
|
||||
|
||||
DEBUG = "DEBUG"
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# prek pre-commit configuration — https://prek.j178.dev
|
||||
#
|
||||
# The local/system hooks below invoke the project's OWN pinned tools (ruff/mypy
|
||||
# from uv.lock) and mirror .github/workflows/build.yml's lint job exactly, so
|
||||
# they can never drift from CI's versions or rules. prek installs nothing of
|
||||
# its own for them — "system" language just execs whatever `uv run` resolves.
|
||||
#
|
||||
# The pre-commit/pre-commit-hooks repo hooks below are generic file checks with
|
||||
# no project-local tool equivalent, so they're kept as a normal (non-local) repo.
|
||||
#
|
||||
# Run all checks manually: `uv run prek run --all-files`
|
||||
# Install the git hooks: `uv run prek install`
|
||||
|
||||
default_install_hook_types = ["pre-commit", "pre-push"]
|
||||
default_stages = ["pre-commit"]
|
||||
|
||||
[[repos]]
|
||||
repo = "https://github.com/pre-commit/pre-commit-hooks"
|
||||
rev = "v4.4.0"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "check-case-conflict"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "check-merge-conflict"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "check-toml"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "check-yaml"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "debug-statements"
|
||||
|
||||
[[repos]]
|
||||
repo = "local"
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "ruff-format"
|
||||
name = "ruff format (check)"
|
||||
language = "system"
|
||||
entry = "uv run ruff format --check ."
|
||||
types = ["python"]
|
||||
pass_filenames = false
|
||||
require_serial = true
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "ruff-check"
|
||||
name = "ruff check"
|
||||
language = "system"
|
||||
entry = "uv run ruff check ."
|
||||
types = ["python"]
|
||||
pass_filenames = false
|
||||
require_serial = true
|
||||
|
||||
[[repos.hooks]]
|
||||
id = "mypy"
|
||||
name = "mypy"
|
||||
language = "system"
|
||||
entry = "uv run mypy src/ocrmypdf"
|
||||
types = ["python"]
|
||||
pass_filenames = false
|
||||
require_serial = true
|
||||
+41
-14
@@ -1,22 +1,21 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
[build-system]
|
||||
requires = ["hatchling", "hatch-vcs"]
|
||||
requires = ["hatchling"]
|
||||
build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "ocrmypdf"
|
||||
dynamic = ["version"]
|
||||
version = "17.9.0"
|
||||
description = "OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched"
|
||||
readme = "README.md"
|
||||
license = "MPL-2.0"
|
||||
requires-python = ">=3.11"
|
||||
dependencies = [
|
||||
"deprecation>=2.1.0",
|
||||
"fpdf2>=2.8.0",
|
||||
"img2pdf>=0.5",
|
||||
"packaging>=20",
|
||||
"pdfminer.six>=20220319",
|
||||
"pdfminer.six>=20260107", # fixes parsing of tokens split across the read buffer/streams (gh #1361)
|
||||
"pi-heif", # Heif image format - maintainers: if this is removed, it will NOT break
|
||||
"pikepdf>=10",
|
||||
"Pillow>=10.0.1",
|
||||
@@ -24,6 +23,7 @@ dependencies = [
|
||||
"pydantic>=2.12.5",
|
||||
"pypdfium2>=5.0.0",
|
||||
"rich>=13",
|
||||
"typing-extensions>=4.12; python_version < '3.13'",
|
||||
"uharfbuzz>=0.53.2",
|
||||
]
|
||||
authors = [{ name = "James R. Barlow", email = "james@purplerock.ca" }]
|
||||
@@ -49,21 +49,29 @@ keywords = ["PDF", "OCR", "optical character recognition", "PDF/A", "scanning"]
|
||||
Documentation = "https://ocrmypdf.readthedocs.io/"
|
||||
Source = "https://github.com/ocrmypdf/OCRmyPDF"
|
||||
Tracker = "https://github.com/ocrmypdf/OCRmyPDF/issues"
|
||||
Changelog = "https://github.com/ocrmypdf/OCRmyPDF/docs/release_notes.md"
|
||||
Changelog = "https://github.com/ocrmypdf/OCRmyPDF/tree/main/docs/releasenotes"
|
||||
|
||||
[project.optional-dependencies]
|
||||
# User-installable features - use `uv sync --extra <name>` or `pip install ocrmypdf[name]`
|
||||
watcher = ["watchdog>=1.0.2", "cyclopts>=3", "python-dotenv"]
|
||||
webservice = ["streamlit>=1.41.0"]
|
||||
# Batch web interface (webui/): multi-file upload, background OCR, zip download
|
||||
# Plain uvicorn rather than uvicorn[standard]: uvloop/httptools have no musl
|
||||
# wheels and would have to be compiled for the Alpine image, and the event
|
||||
# loop is never the bottleneck here — Tesseract is.
|
||||
webui = [
|
||||
"fastapi>=0.115",
|
||||
"uvicorn>=0.30",
|
||||
"python-multipart>=0.0.18",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
ocrmypdf = "ocrmypdf.__main__:run"
|
||||
|
||||
[tool.hatch.version]
|
||||
source = "vcs"
|
||||
|
||||
[tool.hatch.build.hooks.vcs]
|
||||
version-file = "src/ocrmypdf/_version.py"
|
||||
[tool.hatch.build.targets.wheel]
|
||||
# Stated explicitly so top-level directories that are not part of the
|
||||
# distribution (webui/, misc/, tests/) never get picked up by autodetection.
|
||||
packages = ["src/ocrmypdf"]
|
||||
|
||||
[tool.distutils.bdist_wheel]
|
||||
python-tag = "py311"
|
||||
@@ -104,18 +112,28 @@ filterwarnings = [
|
||||
]
|
||||
|
||||
[tool.mypy]
|
||||
check_untyped_defs = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = [
|
||||
'pluggy',
|
||||
'img2pdf',
|
||||
'pdfminer.*',
|
||||
'reportlab.*',
|
||||
'fitz',
|
||||
'libxmp.utils',
|
||||
'pypdfium2',
|
||||
'uharfbuzz',
|
||||
'pi_heif',
|
||||
]
|
||||
ignore_missing_imports = true
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
# Test functions are not required to annotate their return type (almost
|
||||
# always None); it's a low-value hint that would just be noise here.
|
||||
module = 'tests.*'
|
||||
disallow_untyped_defs = false
|
||||
disallow_incomplete_defs = false
|
||||
|
||||
[tool.ruff]
|
||||
target-version = "py311"
|
||||
exclude = ["src/ocrmypdf/_version.py"] # Autogenerated
|
||||
@@ -131,6 +149,7 @@ exclude = ["src/ocrmypdf/_version.py"] # Autogenerated
|
||||
"SIM", # simplify
|
||||
"B", # flake8-bugbear
|
||||
"ICN", # flake8-import-conventions
|
||||
"PTH", # flake8-use-pathlib
|
||||
]
|
||||
ignore = [
|
||||
"B028", # warning with no explicit stacklevel
|
||||
@@ -139,7 +158,7 @@ ignore = [
|
||||
]
|
||||
|
||||
[tool.ruff.lint.isort]
|
||||
known-first-party = ["ocrmypdf"]
|
||||
known-first-party = ["ocrmypdf", "webui"]
|
||||
required-imports = ["from __future__ import annotations"]
|
||||
|
||||
[tool.ruff.lint.flake8-import-conventions]
|
||||
@@ -162,7 +181,15 @@ quote-style = "preserve"
|
||||
|
||||
[dependency-groups]
|
||||
# Developer-only tools - use `uv sync --group <name>`
|
||||
dev = ["mypy>=1.13.0", "ipykernel>=6.29.5", "reportlab>=4.4.4"]
|
||||
dev = [
|
||||
"mypy>=1.13.0",
|
||||
"ruff>=0.14.11",
|
||||
"prek>=0.4.8",
|
||||
"ipykernel>=6.29.5",
|
||||
"reportlab>=4.4.4",
|
||||
"cyclopts>=4.5.1",
|
||||
"pygithub>=2.9.1",
|
||||
]
|
||||
test = [
|
||||
# Core testing framework
|
||||
"coverage[toml]>=6.2",
|
||||
@@ -174,7 +201,6 @@ test = [
|
||||
"python-xmp-toolkit==2.0.1", # also requires apt-get install libexempi3
|
||||
"reportlab>=3.6.8",
|
||||
# Type stubs for testing
|
||||
"types-Pillow",
|
||||
"types-humanfriendly",
|
||||
# Extended test capabilities (merged from extended_test)
|
||||
"pymupdf>=1.24.14",
|
||||
@@ -183,6 +209,7 @@ docs = [
|
||||
"myst-parser>=4.0.1",
|
||||
"sphinx",
|
||||
"sphinx-issues",
|
||||
"sphinx-reredirects",
|
||||
"sphinx-rtd-theme",
|
||||
"sphinxcontrib-mermaid",
|
||||
]
|
||||
|
||||
@@ -11,7 +11,7 @@ from ocrmypdf import helpers, hocrtransform, pdfa, pdfinfo
|
||||
from ocrmypdf._concurrent import Executor
|
||||
from ocrmypdf._defaults import PROGRAM_NAME
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._options import OcrOptions, TaggedPdfMode
|
||||
from ocrmypdf._pipelines._common import (
|
||||
configure_debug_logging,
|
||||
)
|
||||
@@ -19,6 +19,7 @@ from ocrmypdf._version import __version__
|
||||
from ocrmypdf.api import (
|
||||
Verbosity,
|
||||
configure_logging,
|
||||
configure_stdout_protection,
|
||||
ocr,
|
||||
)
|
||||
from ocrmypdf.exceptions import (
|
||||
@@ -53,6 +54,7 @@ __all__ = [
|
||||
'BoundingBox',
|
||||
'configure_debug_logging',
|
||||
'configure_logging',
|
||||
'configure_stdout_protection',
|
||||
'DpiError',
|
||||
'EncryptedPdfError',
|
||||
'Executor',
|
||||
@@ -78,6 +80,7 @@ __all__ = [
|
||||
'PriorOcrFoundError',
|
||||
'PROGRAM_NAME',
|
||||
'SubprocessOutputError',
|
||||
'TaggedPdfMode',
|
||||
'TesseractConfigError',
|
||||
'UnsupportedImageFormatError',
|
||||
'Verbosity',
|
||||
|
||||
@@ -16,7 +16,7 @@ from contextlib import suppress
|
||||
from ocrmypdf import __version__
|
||||
from ocrmypdf._pipelines.ocr import run_pipeline_cli
|
||||
from ocrmypdf._validation import check_options
|
||||
from ocrmypdf.api import Verbosity, configure_logging
|
||||
from ocrmypdf.api import Verbosity, configure_logging, configure_stdout_protection
|
||||
from ocrmypdf.cli import get_options_and_plugins
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
@@ -39,12 +39,17 @@ def sigbus(*args):
|
||||
|
||||
def run(args=None):
|
||||
"""Run the ocrmypdf command line interface."""
|
||||
# Protect the real stdout before loading plugins or starting any worker
|
||||
# processes/threads, so that only our final PDF output can reach it and
|
||||
# stray writes from plugins or libraries are diverted to stderr.
|
||||
configure_stdout_protection()
|
||||
|
||||
options, plugin_manager = get_options_and_plugins(args=args)
|
||||
|
||||
with suppress(AttributeError, PermissionError):
|
||||
os.nice(5)
|
||||
|
||||
verbosity = options.verbose
|
||||
verbosity = Verbosity(options.verbose)
|
||||
if not os.isatty(sys.stderr.fileno()):
|
||||
options.progress_bar = False
|
||||
if options.quiet:
|
||||
|
||||
@@ -8,7 +8,7 @@ from __future__ import annotations
|
||||
import threading
|
||||
from abc import ABC, abstractmethod
|
||||
from collections.abc import Callable, Iterable
|
||||
from typing import Any, TypeVar
|
||||
from typing import Any, TypeVar, cast
|
||||
|
||||
from ocrmypdf._progressbar import NullProgressBar, ProgressBar
|
||||
|
||||
@@ -72,7 +72,10 @@ class Executor(ABC):
|
||||
if not task_finished:
|
||||
task_finished = _task_finished_noop
|
||||
if not task:
|
||||
task = _task_noop
|
||||
# _task_noop always returns None, but T is unbound here (it's
|
||||
# only meaningful when a real task is supplied); task_finished's
|
||||
# own no-op default accepts Any, so this is safe.
|
||||
task = cast('Callable[..., T]', _task_noop)
|
||||
|
||||
with self.pool_lock:
|
||||
self._execute(
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
"""Probe helper for external executables.
|
||||
|
||||
Each ``ocrmypdf._exec.<tool>`` module describes its external program with a
|
||||
module-level :class:`ToolProbe` and delegates ``version()`` / ``available()``
|
||||
to it. This separates the "is the tool installed and suitable?" question
|
||||
(probing) from the "run the tool" question (execution). Work functions stay
|
||||
as pure module-level functions so they are trivially picklable for use in
|
||||
subprocess workers.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ToolProbe:
|
||||
"""Describes how to detect an external executable and its version.
|
||||
|
||||
Attributes:
|
||||
program: The program name as it appears on PATH (or a full path).
|
||||
version_arg: The argument that elicits a version string.
|
||||
version_regex: A regex with a capturing group that extracts the
|
||||
version from the program's output.
|
||||
version_cls: A :class:`packaging.version.Version` subclass, used for
|
||||
tools with non-standard version strings (e.g. Tesseract).
|
||||
env: Optional environment overrides applied when probing the version.
|
||||
also_catch: Additional exception types that should be treated as
|
||||
"not available" by :meth:`available`. :class:`OSError` is useful
|
||||
for tools like verapdf whose launcher may fail with non-standard
|
||||
errors when the JVM is missing.
|
||||
"""
|
||||
|
||||
program: str
|
||||
version_arg: str = '--version'
|
||||
version_regex: str = r'(\d+(\.\d+)*)'
|
||||
version_cls: type[Version] = Version
|
||||
env: Mapping[str, str] | None = None
|
||||
also_catch: tuple[type[BaseException], ...] = ()
|
||||
|
||||
def version(self) -> Version:
|
||||
"""Return the installed version of the program.
|
||||
|
||||
Raises:
|
||||
MissingDependencyError: if the program cannot be found or its
|
||||
version string cannot be parsed.
|
||||
"""
|
||||
raw = get_version(
|
||||
self.program,
|
||||
version_arg=self.version_arg,
|
||||
regex=self.version_regex,
|
||||
env=self.env,
|
||||
)
|
||||
return self.version_cls(raw)
|
||||
|
||||
def available(self) -> bool:
|
||||
"""Return whether a usable version of the program is installed."""
|
||||
try:
|
||||
self.version()
|
||||
except MissingDependencyError:
|
||||
return False
|
||||
except self.also_catch:
|
||||
return False
|
||||
return True
|
||||
@@ -16,6 +16,7 @@ from subprocess import PIPE, CalledProcessError
|
||||
from packaging.version import Version
|
||||
from PIL import Image, UnidentifiedImageError
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import (
|
||||
ColorConversionNeededError,
|
||||
InputFileError,
|
||||
@@ -23,7 +24,7 @@ from ocrmypdf.exceptions import (
|
||||
)
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
||||
from ocrmypdf.subprocess import run, run_polling_stderr
|
||||
|
||||
COLOR_CONVERSION_STRATEGIES = frozenset(
|
||||
[
|
||||
@@ -69,11 +70,19 @@ class DuplicateFilter(logging.Filter):
|
||||
return True
|
||||
|
||||
|
||||
log.addFilter(DuplicateFilter(log))
|
||||
PROBE = ToolProbe(program=GS)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version(GS))
|
||||
def _ensure_log_filter_installed() -> None:
|
||||
"""Idempotently attach the duplicate-suppressing filter to the GS logger.
|
||||
|
||||
Called at the top of each work function so the filter is present in the
|
||||
main process *and* in any subprocess worker that calls Ghostscript.
|
||||
"""
|
||||
if not any(isinstance(f, DuplicateFilter) for f in log.filters):
|
||||
log.addFilter(DuplicateFilter(log))
|
||||
|
||||
|
||||
def _gs_error_reported(stream) -> bool:
|
||||
@@ -96,8 +105,8 @@ def _gs_devicen_reported(stream) -> bool:
|
||||
|
||||
|
||||
def rasterize_pdf(
|
||||
input_file: os.PathLike,
|
||||
output_file: os.PathLike,
|
||||
input_file: Path,
|
||||
output_file: Path,
|
||||
*,
|
||||
raster_device: GhostscriptRasterDevice,
|
||||
raster_dpi: Resolution,
|
||||
@@ -123,10 +132,37 @@ def rasterize_pdf(
|
||||
use_cropbox: If True, rasterize the CropBox instead of MediaBox.
|
||||
Default is False (use MediaBox).
|
||||
"""
|
||||
_ensure_log_filter_installed()
|
||||
raster_dpi = raster_dpi.round(6)
|
||||
if not page_dpi:
|
||||
page_dpi = raster_dpi
|
||||
|
||||
# Ghostscript may fail with very low DPI values (below 10). If the requested
|
||||
# DPI is too low, use a minimum of 10 DPI and resize the output afterward.
|
||||
MIN_RASTER_DPI = 10
|
||||
needs_low_dpi_resize = (
|
||||
raster_dpi.x < MIN_RASTER_DPI or raster_dpi.y < MIN_RASTER_DPI
|
||||
)
|
||||
if needs_low_dpi_resize:
|
||||
effective_dpi = Resolution(
|
||||
max(raster_dpi.x, MIN_RASTER_DPI), max(raster_dpi.y, MIN_RASTER_DPI)
|
||||
)
|
||||
else:
|
||||
effective_dpi = raster_dpi
|
||||
|
||||
# Anti-alias text and vector graphics when rendering to a contone device.
|
||||
# Ghostscript 10.x renders aliased glyphs that OCR frequently misreads as
|
||||
# extra word breaks; anti-aliasing empirically improves OCR accuracy on the
|
||||
# Ghostscript path, especially for small fonts at moderate DPI (#1439).
|
||||
# The 1-bit mono devices do not accept alpha bits (older Ghostscript
|
||||
# rejects them) and pngmonod performs its own anti-aliased downscaling.
|
||||
mono_devices = (GhostscriptRasterDevice.PNGMONO, GhostscriptRasterDevice.PNGMONOD)
|
||||
antialias_args = (
|
||||
[]
|
||||
if raster_device in mono_devices
|
||||
else ['-dTextAlphaBits=4', '-dGraphicsAlphaBits=4']
|
||||
)
|
||||
|
||||
args_gs = (
|
||||
[
|
||||
GS,
|
||||
@@ -137,8 +173,9 @@ def rasterize_pdf(
|
||||
f'-sDEVICE={raster_device}',
|
||||
f'-dFirstPage={pageno}',
|
||||
f'-dLastPage={pageno}',
|
||||
f'-r{raster_dpi.x:f}x{raster_dpi.y:f}',
|
||||
f'-r{effective_dpi.x:f}x{effective_dpi.y:f}',
|
||||
]
|
||||
+ antialias_args
|
||||
+ (['-dUseCropBox'] if use_cropbox else [])
|
||||
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
||||
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||
@@ -172,7 +209,18 @@ def rasterize_pdf(
|
||||
)
|
||||
|
||||
try:
|
||||
im: Image.Image
|
||||
with Image.open(output_file) as im:
|
||||
if needs_low_dpi_resize:
|
||||
# Resize to the dimensions that would have resulted from the
|
||||
# original low DPI request
|
||||
scale_x = raster_dpi.x / effective_dpi.x
|
||||
scale_y = raster_dpi.y / effective_dpi.y
|
||||
new_size = (
|
||||
max(1, int(round(im.width * scale_x))),
|
||||
max(1, int(round(im.height * scale_y))),
|
||||
)
|
||||
im = im.resize(new_size, Image.Resampling.LANCZOS)
|
||||
if rotation is not None:
|
||||
log.debug("Rotating output by %i", rotation)
|
||||
# rotation is a clockwise angle and Image.ROTATE_* is
|
||||
@@ -241,15 +289,18 @@ class GhostscriptFollower:
|
||||
|
||||
def generate_pdfa(
|
||||
pdf_pages,
|
||||
output_file: os.PathLike,
|
||||
output_file: Path,
|
||||
*,
|
||||
compression: str,
|
||||
color_conversion_strategy: str,
|
||||
jpeg_quality: int | None = None,
|
||||
jpeg_maxdpi: int | None = None,
|
||||
pdf_version: str = '1.5',
|
||||
pdfa_part: str = '2',
|
||||
progressbar_class=None,
|
||||
stop_on_error: bool = False,
|
||||
):
|
||||
_ensure_log_filter_installed()
|
||||
# Ghostscript's compression is all or nothing. We can either force all images
|
||||
# to JPEG, force all to Flate/PNG, or let it decide how to encode the images.
|
||||
# In most case it's best to let it decide.
|
||||
@@ -263,6 +314,11 @@ def generate_pdfa(
|
||||
]
|
||||
elif compression == 'lossless':
|
||||
compression_args = [
|
||||
# Re-encoding an existing JPEG with a lossless codec only inflates
|
||||
# its size: the lossy data is already baked in, so there is nothing
|
||||
# to gain. Pass JPEGs through untouched and apply lossless (Flate)
|
||||
# encoding only to images that are not already JPEG.
|
||||
"-dPassThroughJPEGImages=true",
|
||||
"-dAutoFilterColorImages=false",
|
||||
"-dColorImageFilter=/FlateEncode",
|
||||
"-dAutoFilterGrayImages=false",
|
||||
@@ -284,6 +340,35 @@ def generate_pdfa(
|
||||
# Windows has lots of fatal "permission denied" errors
|
||||
stop_on_error = False
|
||||
|
||||
# `-dJPEGQ=N` tells Ghostscript to use a JPEG quality of N, IF it decides
|
||||
# to transcode an image to JPEG. When there are existing JPEG images,
|
||||
# Ghostscript uses passthrough mode, so the quality level is not changed.
|
||||
# OCRmyPDF's optimizer separately uses the `--jpeg-quality` command line
|
||||
# option to potentially re-encode JPEG images, regardless of whether
|
||||
# Ghostscript decided to transcode them to JPEG or not.
|
||||
# `jpeg_quality=0` is meaningful to Ghostscript (maximum compression), so
|
||||
# only fall back to the default when the value is None.
|
||||
effective_jpeg_quality = jpeg_quality if jpeg_quality is not None else 95
|
||||
|
||||
# Downsampling images is a blunt-force way to reduce file size and almost
|
||||
# always degrades quality more than lowering JPEG quality at the original
|
||||
# resolution. We expose this for users with very specific needs (e.g.
|
||||
# producing very small files for screen-only viewing); the optimizer is
|
||||
# usually a better choice.
|
||||
downsample_args: list[str] = []
|
||||
if jpeg_maxdpi is not None:
|
||||
downsample_args = [
|
||||
"-dDownsampleColorImages=true",
|
||||
"-dColorImageDownsampleThreshold=1.0",
|
||||
"-dDownsampleGrayImages=true",
|
||||
"-dGrayImageDownsampleThreshold=1.0",
|
||||
"-dDownsampleMonoImages=true",
|
||||
"-dMonoImageDownsampleThreshold=1.0",
|
||||
f"-dColorImageResolution={jpeg_maxdpi}",
|
||||
f"-dGrayImageResolution={jpeg_maxdpi}",
|
||||
f"-dMonoImageResolution={jpeg_maxdpi}",
|
||||
]
|
||||
|
||||
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
||||
# is set; see:
|
||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699392
|
||||
@@ -300,8 +385,9 @@ def generate_pdfa(
|
||||
]
|
||||
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||
+ compression_args
|
||||
+ downsample_args
|
||||
+ [
|
||||
"-dJPEGQ=95",
|
||||
f"-dJPEGQ={effective_jpeg_quality}", # See note above on JPEG quality
|
||||
"-dSubsetFonts=false", # Prevents GS from messing up some encodings
|
||||
f"-dPDFA={pdfa_part}",
|
||||
"-dPDFACompatibilityPolicy=1",
|
||||
@@ -339,4 +425,8 @@ def generate_pdfa(
|
||||
for part in stderr.split('****'):
|
||||
log.error(part)
|
||||
if _gs_devicen_reported(stderr):
|
||||
raise ColorConversionNeededError()
|
||||
# Ghostscript could not normalize the DeviceN colorspace for PDF/A,
|
||||
# even if the user requested a conversion strategy. The output is
|
||||
# liable to render blank in some viewers, so raise regardless of the
|
||||
# strategy and tailor the guidance to what was attempted.
|
||||
raise ColorConversionNeededError(color_conversion_strategy)
|
||||
|
||||
@@ -9,21 +9,23 @@ from subprocess import PIPE, CalledProcessError
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
_PROBE = ToolProbe(program='jbig2', version_regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
try:
|
||||
version = get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
return _PROBE.version()
|
||||
except CalledProcessError as e:
|
||||
# TeX Live for Windows provides an incompatible jbig2.EXE which may
|
||||
# be on the PATH.
|
||||
raise MissingDependencyError('jbig2enc') from e
|
||||
return Version(version)
|
||||
|
||||
|
||||
def available():
|
||||
def available() -> bool:
|
||||
try:
|
||||
version()
|
||||
except MissingDependencyError:
|
||||
@@ -33,7 +35,7 @@ def available():
|
||||
|
||||
def convert_single(cwd, infile, outfile, threshold):
|
||||
args = ['jbig2', '--pdf', '-t', str(threshold), infile]
|
||||
with open(outfile, 'wb') as fstdout:
|
||||
with outfile.open('wb') as fstdout:
|
||||
proc = run(args, cwd=cwd, stdout=fstdout, stderr=PIPE)
|
||||
proc.check_returncode()
|
||||
return proc
|
||||
|
||||
@@ -8,22 +8,12 @@ from __future__ import annotations
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE
|
||||
|
||||
from packaging.version import Version
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('pngquant', regex=r'(\d+(\.\d+)*).*'))
|
||||
|
||||
|
||||
def available():
|
||||
try:
|
||||
version()
|
||||
except MissingDependencyError:
|
||||
return False
|
||||
return True
|
||||
PROBE = ToolProbe(program='pngquant', version_regex=r'(\d+(\.\d+)*).*')
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max: int):
|
||||
@@ -35,7 +25,7 @@ def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max:
|
||||
quality_min: Minimum quality to use
|
||||
quality_max: Maximum quality to use
|
||||
"""
|
||||
with open(input_file, 'rb') as input_stream:
|
||||
with input_file.open('rb') as input_stream:
|
||||
args = [
|
||||
'pngquant',
|
||||
'--force',
|
||||
|
||||
@@ -17,13 +17,14 @@ from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import (
|
||||
MissingDependencyError,
|
||||
SubprocessOutputError,
|
||||
TesseractConfigError,
|
||||
)
|
||||
from ocrmypdf.pluginspec import OrientationConfidence
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -115,8 +116,13 @@ class TesseractVersion(Version):
|
||||
)
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return TesseractVersion(get_version('tesseract', regex=r'tesseract\s(.+)'))
|
||||
PROBE = ToolProbe(
|
||||
program='tesseract',
|
||||
version_regex=r'tesseract\s(.+)',
|
||||
version_cls=TesseractVersion,
|
||||
)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def has_thresholding() -> bool:
|
||||
@@ -287,12 +293,14 @@ def tesseract_log_output(stream: bytes) -> None:
|
||||
|
||||
lines = text.splitlines()
|
||||
for line in lines:
|
||||
if line.startswith(
|
||||
("Tesseract Open Source", "Warning in pixReadMem")
|
||||
):
|
||||
if line.startswith(("Tesseract Open Source", "Warning in pixReadMem")):
|
||||
continue
|
||||
elif 'diacritics' in line:
|
||||
tlog.warning("lots of diacritics - possibly poor OCR")
|
||||
# Surface the raw Tesseract message at debug level so users can see
|
||||
# exactly what Tesseract reported (e.g. the affected count) without
|
||||
# losing the interpreted hint above (#1566).
|
||||
tlog.debug(line.strip())
|
||||
elif line.startswith('OSD: Weak margin'):
|
||||
tlog.warning("unsure about page orientation")
|
||||
elif 'Error in pixScanForForeground' in line:
|
||||
@@ -309,6 +317,23 @@ def tesseract_log_output(stream: bytes) -> None:
|
||||
tlog.warning(line.strip())
|
||||
elif 'read_params_file' in line.lower():
|
||||
tlog.error(line.strip())
|
||||
# Tesseract emits "read_params_file: Can't open <name>" when it
|
||||
# cannot locate a config file (e.g. 'hocr', 'txt') in its
|
||||
# tessdata configs/ directory, then exits 0 without producing
|
||||
# the requested output. Promote to a hard error so the user
|
||||
# sees the root cause instead of a downstream FileNotFoundError.
|
||||
if "Can't open" in line:
|
||||
missing = line.split("Can't open", 1)[1].strip()
|
||||
else:
|
||||
missing = line.strip()
|
||||
raise TesseractConfigError(
|
||||
f"Tesseract cannot open its config file '{missing}'. "
|
||||
"This usually means Tesseract is installed but its config "
|
||||
"files are missing from the tessdata configs/ directory. "
|
||||
"On Debian/Ubuntu, ensure the 'tesseract-ocr' package is "
|
||||
"fully installed. If you set TESSDATA_PREFIX, verify its "
|
||||
"configs/ subdirectory contains the required files."
|
||||
)
|
||||
else:
|
||||
tlog.info(line.strip())
|
||||
|
||||
@@ -389,6 +414,12 @@ def generate_hocr(
|
||||
raise SubprocessOutputError() from e
|
||||
else:
|
||||
tesseract_log_output(stdout)
|
||||
if not output_hocr.exists():
|
||||
raise SubprocessOutputError(
|
||||
"Tesseract exited successfully but did not produce the "
|
||||
f"expected hOCR output at {output_hocr}. Tesseract output:\n"
|
||||
+ (stdout.decode(errors='replace') if stdout else '(empty)')
|
||||
)
|
||||
# The sidecar text file will get the suffix .txt; rename it to
|
||||
# whatever caller wants it named
|
||||
with suppress(FileNotFoundError):
|
||||
@@ -457,6 +488,12 @@ def generate_pdf(
|
||||
stdout = p.stdout
|
||||
with suppress(FileNotFoundError):
|
||||
prefix.with_suffix('.txt').replace(output_text)
|
||||
if not output_pdf.exists():
|
||||
raise SubprocessOutputError(
|
||||
"Tesseract exited successfully but did not produce the "
|
||||
f"expected PDF output at {output_pdf}. Tesseract output:\n"
|
||||
+ (stdout.decode(errors='replace') if stdout else '(empty)')
|
||||
)
|
||||
except TimeoutExpired:
|
||||
page_timedout(timeout)
|
||||
use_skip_page(output_pdf, output_text)
|
||||
|
||||
@@ -7,7 +7,6 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
from collections.abc import Iterator
|
||||
from contextlib import contextmanager
|
||||
from decimal import Decimal
|
||||
@@ -15,11 +14,11 @@ from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
from packaging.version import Version
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import SubprocessOutputError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
# unpaper documentation:
|
||||
# https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md
|
||||
@@ -47,8 +46,9 @@ class UnpaperImageTooLargeError(Exception):
|
||||
super().__init__(self.message)
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('unpaper', regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)'))
|
||||
PROBE = ToolProbe(program='unpaper', version_regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)')
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
@contextmanager
|
||||
@@ -101,13 +101,6 @@ def run_unpaper(
|
||||
) from e
|
||||
|
||||
|
||||
def validate_custom_args(args: str) -> list[str]:
|
||||
unpaper_args = shlex.split(args)
|
||||
if any(('/' in arg or arg == '.' or arg == '..') for arg in unpaper_args):
|
||||
raise ValueError('No filenames allowed in --unpaper-args')
|
||||
return unpaper_args
|
||||
|
||||
|
||||
def clean(
|
||||
input_file: Path,
|
||||
output_file: Path,
|
||||
|
||||
@@ -11,10 +11,9 @@ from pathlib import Path
|
||||
from subprocess import PIPE
|
||||
from typing import NamedTuple
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf._exec._probe import ToolProbe
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import run
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -27,18 +26,13 @@ class ValidationResult(NamedTuple):
|
||||
message: str
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
"""Get verapdf version."""
|
||||
return Version(get_version('verapdf', regex=r'veraPDF (\d+(\.\d+)*)'))
|
||||
|
||||
|
||||
def available() -> bool:
|
||||
"""Check if verapdf is available."""
|
||||
try:
|
||||
version()
|
||||
except MissingDependencyError:
|
||||
return False
|
||||
return True
|
||||
PROBE = ToolProbe(
|
||||
program='verapdf',
|
||||
version_regex=r'veraPDF (\d+(\.\d+)*)',
|
||||
also_catch=(OSError,),
|
||||
)
|
||||
version = PROBE.version
|
||||
available = PROBE.available
|
||||
|
||||
|
||||
def output_type_to_flavour(output_type: str) -> str:
|
||||
|
||||
+133
-14
@@ -6,11 +6,12 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from collections.abc import Collection
|
||||
from contextlib import suppress
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
from typing import TYPE_CHECKING, cast
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from ocrmypdf.hocrtransform import OcrElement
|
||||
@@ -18,6 +19,7 @@ if TYPE_CHECKING:
|
||||
from pikepdf import (
|
||||
Dictionary,
|
||||
Name,
|
||||
Object,
|
||||
Operator,
|
||||
Page,
|
||||
Pdf,
|
||||
@@ -117,6 +119,10 @@ def _build_text_layer_ctm(
|
||||
):
|
||||
"""Build transformation matrix to align text layer with page content.
|
||||
|
||||
Always computes the full CTM to handle non-zero page origins (e.g.,
|
||||
JSTOR PDFs with MediaBox like [0, 100, 595, 982]) and minor scale
|
||||
differences due to DPI rounding.
|
||||
|
||||
Args:
|
||||
text_width: Width of text layer mediabox.
|
||||
text_height: Height of text layer mediabox.
|
||||
@@ -127,11 +133,8 @@ def _build_text_layer_ctm(
|
||||
text_rotation: Rotation in degrees (clockwise) to apply to text layer.
|
||||
|
||||
Returns:
|
||||
pikepdf.Matrix transformation matrix, or None if no rotation needed.
|
||||
pikepdf.Matrix transformation matrix, or None if identity.
|
||||
"""
|
||||
if text_rotation == 0:
|
||||
return None
|
||||
|
||||
from pikepdf import Matrix
|
||||
|
||||
wt, ht = text_width, text_height
|
||||
@@ -153,7 +156,14 @@ def _build_text_layer_ctm(
|
||||
scale_y = page_height / ht if ht else 1.0
|
||||
scale = Matrix().scaled(scale_x, scale_y)
|
||||
|
||||
return translate @ rotate @ scale @ untranslate @ corner
|
||||
ctm = translate @ rotate @ scale @ untranslate @ corner
|
||||
|
||||
# Return None if the result is effectively identity
|
||||
identity = Matrix()
|
||||
if ctm == identity:
|
||||
return None
|
||||
|
||||
return ctm
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -173,9 +183,13 @@ def strip_invisible_text(pdf: Pdf, page: Page):
|
||||
render_mode_stack = []
|
||||
text_objects = []
|
||||
|
||||
for operands, operator in parse_content_stream(page, ''):
|
||||
for instruction in parse_content_stream(page, ''):
|
||||
operands, operator = instruction.operands, instruction.operator
|
||||
if operator == Operator('Tr'):
|
||||
render_mode = operands[0]
|
||||
# operands[0] is already a plain int under pikepdf's default
|
||||
# (implicit) conversion mode, or a pikepdf.Object under explicit
|
||||
# conversion mode; int() handles both.
|
||||
render_mode = int(operands[0])
|
||||
|
||||
if operator == Operator('q'):
|
||||
render_mode_stack.append(render_mode)
|
||||
@@ -199,10 +213,103 @@ def strip_invisible_text(pdf: Pdf, page: Page):
|
||||
stream.extend(text_objects)
|
||||
text_objects.clear()
|
||||
|
||||
content_stream = unparse_content_stream(stream)
|
||||
# pikepdf's Collection[...] parameter doesn't structurally match our
|
||||
# _ObjectList-based tuples even though it works fine at runtime.
|
||||
content_stream = unparse_content_stream(
|
||||
cast('list[tuple[Collection[Object], Operator]]', stream)
|
||||
)
|
||||
page.Contents = Stream(pdf, content_stream)
|
||||
|
||||
|
||||
def discard_text_search_index(pdf: Pdf) -> bool:
|
||||
"""Discard an embedded Adobe full-text search index from the catalog.
|
||||
|
||||
Adobe Acrobat can embed a full-text search index in the document catalog at
|
||||
``/Root/PieceInfo/SearchIndex``. It is built from the page text, and only
|
||||
Acrobat reads it; other viewers ignore it and search the text on the fly.
|
||||
Any change to the PDF invalidates the index, so once OCRmyPDF rewrites the
|
||||
document (editing the text layer, rasterizing, optimizing) a retained index
|
||||
would be stale and return incorrect search results in Acrobat. We cannot
|
||||
update this vendor-private data, so we discard it; modern viewers rebuild a
|
||||
search index on demand. Returns True if the catalog was modified.
|
||||
"""
|
||||
try:
|
||||
pieceinfo = pdf.Root.get(Name.PieceInfo)
|
||||
if not isinstance(pieceinfo, Dictionary) or Name.SearchIndex not in pieceinfo:
|
||||
return False
|
||||
del pieceinfo[Name.SearchIndex]
|
||||
log.debug(
|
||||
"Discarded embedded text search index "
|
||||
"(/Root/PieceInfo/SearchIndex) because the PDF was rewritten; "
|
||||
"it would otherwise be stale."
|
||||
)
|
||||
# Drop an empty PieceInfo rather than leave a husk behind.
|
||||
if len(pieceinfo) == 0:
|
||||
del pdf.Root.PieceInfo
|
||||
return True
|
||||
except (KeyError, TypeError, AttributeError):
|
||||
return False
|
||||
|
||||
|
||||
def discard_page_thumbnails(pdf: Pdf) -> int:
|
||||
"""Discard embedded per-page thumbnail images.
|
||||
|
||||
A page object may carry an optional ``/Thumb`` image XObject — a miniature
|
||||
rendering of the page (ISO 32000-2, 12.3.4). It is only a navigation aid and
|
||||
modern viewers generate page thumbnails on demand. OCRmyPDF alters page
|
||||
appearance (deskew, clean, rasterize, re-render) and plugins may edit pages
|
||||
arbitrarily, so any retained thumbnail would be stale and misrepresent its
|
||||
page. We discard them; viewers rebuild thumbnails as needed. Returns the
|
||||
number of thumbnails removed.
|
||||
"""
|
||||
removed = 0
|
||||
for page in pdf.pages:
|
||||
pageobj = page.obj
|
||||
if Name.Thumb in pageobj:
|
||||
del pageobj[Name.Thumb]
|
||||
removed += 1
|
||||
if removed:
|
||||
log.debug(
|
||||
"Discarded %d embedded page thumbnail(s) (/Thumb) because the PDF "
|
||||
"was rewritten; they would otherwise be stale.",
|
||||
removed,
|
||||
)
|
||||
return removed
|
||||
|
||||
|
||||
def discard_structure_tree(pdf: Pdf) -> bool:
|
||||
"""Discard the logical structure (tagged-PDF) tree from the document.
|
||||
|
||||
The structure tree (``/Root/StructTreeRoot``, ``/Root/MarkInfo``) maps
|
||||
marked content in the page content streams to semantic elements via MCIDs.
|
||||
When OCRmyPDF rasterizes pages (force) or strips and rewrites the text layer
|
||||
(redo), those MCIDs are destroyed or renumbered, leaving the tree dangling
|
||||
and inconsistent with the new content. We cannot rebuild it to match, so we
|
||||
discard it; the page-level ``/StructParents`` keys go too. Returns True if
|
||||
the catalog was modified.
|
||||
"""
|
||||
modified = False
|
||||
try:
|
||||
if Name.StructTreeRoot in pdf.Root:
|
||||
del pdf.Root.StructTreeRoot
|
||||
modified = True
|
||||
if Name.MarkInfo in pdf.Root:
|
||||
del pdf.Root.MarkInfo
|
||||
modified = True
|
||||
for page in pdf.pages:
|
||||
if Name.StructParents in page.obj:
|
||||
del page.obj[Name.StructParents]
|
||||
modified = True
|
||||
except (KeyError, TypeError, AttributeError):
|
||||
return modified
|
||||
if modified:
|
||||
log.debug(
|
||||
"Discarded the logical structure tree (/Root/StructTreeRoot) "
|
||||
"because the PDF was re-OCR'd; it would otherwise be stale."
|
||||
)
|
||||
return modified
|
||||
|
||||
|
||||
class OcrGrafter:
|
||||
"""Manages grafting text-only PDFs onto regular PDFs."""
|
||||
|
||||
@@ -245,6 +352,14 @@ class OcrGrafter:
|
||||
ocr_tree: OCR tree for fpdf2 renderer.
|
||||
autorotate_correction: Orientation correction in degrees (0, 90, 180, 270).
|
||||
"""
|
||||
if self.context.options.mode == ProcessingMode.strip_text:
|
||||
# Strip mode: remove the invisible OCR text layer in place without
|
||||
# rasterizing or grafting anything. Honor --pages if specified.
|
||||
options = self.context.options
|
||||
if not options.pages or pageno in options.pages:
|
||||
strip_invisible_text(self.pdf_base, self.pdf_base.pages[pageno])
|
||||
return
|
||||
|
||||
if ocr_output and ocr_tree:
|
||||
raise ValueError(
|
||||
'Cannot specify both ocr_output and ocr_tree for fpdf2 renderer'
|
||||
@@ -311,9 +426,9 @@ class OcrGrafter:
|
||||
|
||||
def finalize(self):
|
||||
# Can have hocr OR parsed pages OR neither (no OCR), but not both
|
||||
assert not (
|
||||
self.fpdf2_hocr_pages and self.fpdf2_parsed_pages
|
||||
), "Can't have both hocr and ocrtree pages"
|
||||
assert not (self.fpdf2_hocr_pages and self.fpdf2_parsed_pages), (
|
||||
"Can't have both hocr and ocrtree pages"
|
||||
)
|
||||
|
||||
if self.fpdf2_hocr_pages:
|
||||
# Render all pages with fpdf2, then graft
|
||||
@@ -323,11 +438,15 @@ class OcrGrafter:
|
||||
if self.fpdf2_parsed_pages:
|
||||
self._render_and_graft_fpdf2_pages()
|
||||
|
||||
discard_text_search_index(self.pdf_base)
|
||||
discard_page_thumbnails(self.pdf_base)
|
||||
if self.context.options.mode in (ProcessingMode.force, ProcessingMode.redo):
|
||||
discard_structure_tree(self.pdf_base)
|
||||
self.pdf_base.save(self.output_file)
|
||||
self.pdf_base.close()
|
||||
return self.output_file
|
||||
|
||||
def _parse_hocr_pages(self):
|
||||
def _parse_hocr_pages(self) -> list[Fpdf2ParsedPage]:
|
||||
"""Render all pages to multi-page PDF with shared fonts, then graft."""
|
||||
from ocrmypdf.hocrtransform.hocr_parser import HocrParser
|
||||
|
||||
@@ -447,7 +566,7 @@ class OcrGrafter:
|
||||
xobj.Type = Name.XObject
|
||||
xobj.Subtype = Name.Form
|
||||
xobj.FormType = 1
|
||||
xobj.BBox = mediabox
|
||||
xobj.BBox = base_mediabox
|
||||
|
||||
# Copy resources from text page's Resources to xobj
|
||||
# We need to handle this carefully since text_page is from a foreign PDF
|
||||
|
||||
@@ -36,6 +36,7 @@ class PdfContext:
|
||||
plugin_manager,
|
||||
):
|
||||
self.options = options
|
||||
self.options.work_folder = work_folder
|
||||
self.work_folder = work_folder
|
||||
self.origin = origin
|
||||
self.pdfinfo = pdfinfo
|
||||
|
||||
@@ -7,7 +7,6 @@ from __future__ import annotations
|
||||
|
||||
import datetime as dt
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
@@ -88,8 +87,12 @@ def repair_docinfo_nuls(pdf):
|
||||
if isinstance(v, str) and b'\x00' in bytes(v):
|
||||
pdf.docinfo[k] = bytes(v).replace(b'\x00', b'')
|
||||
modified = True
|
||||
except TypeError:
|
||||
# TypeError can also be raised if dictionary items are unexpected types
|
||||
except (TypeError, UnicodeDecodeError):
|
||||
# TypeError: DocumentInfo is not a dictionary, or its items are
|
||||
# unexpected types.
|
||||
# UnicodeDecodeError: a DocumentInfo key or value contains bytes that
|
||||
# are not valid PDFDocEncoding/UTF-16, e.g. a Latin-1 /Name key such as
|
||||
# /Saks#e5r. Older pikepdf raised while iterating such a block (#1540).
|
||||
log.error("File contains a malformed DocumentInfo block - continuing anyway.")
|
||||
return modified
|
||||
|
||||
@@ -99,7 +102,7 @@ def should_linearize(working_file: Path, context: PdfContext) -> bool:
|
||||
|
||||
For smaller files, linearization is not worth the effort.
|
||||
"""
|
||||
filesize = os.stat(working_file).st_size
|
||||
filesize = working_file.stat().st_size
|
||||
return filesize > (context.options.fast_web_view * 1_000_000)
|
||||
|
||||
|
||||
|
||||
+103
-30
@@ -8,6 +8,7 @@ from __future__ import annotations
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import unicodedata
|
||||
from collections.abc import Sequence
|
||||
from enum import StrEnum
|
||||
@@ -28,7 +29,7 @@ log = logging.getLogger(__name__)
|
||||
|
||||
# Module-level registry for plugin option models
|
||||
# This is populated by setup_plugin_infrastructure() after plugins are loaded
|
||||
_plugin_option_models: dict[str, type] = {}
|
||||
_plugin_option_models: dict[str, type[BaseModel]] = {}
|
||||
|
||||
PathOrIO = BinaryIO | IOBase | Path | str | bytes
|
||||
|
||||
@@ -42,16 +43,59 @@ class ProcessingMode(StrEnum):
|
||||
- ``force``: Rasterize all content and run OCR regardless of existing text
|
||||
- ``skip``: Skip OCR on pages that already have text
|
||||
- ``redo``: Re-OCR pages, stripping old invisible text layer
|
||||
- ``strip``: Remove the invisible OCR text layer in place; do not OCR
|
||||
"""
|
||||
|
||||
default = 'default'
|
||||
force = 'force'
|
||||
skip = 'skip'
|
||||
redo = 'redo'
|
||||
# User-facing value is '--mode strip'; the member is named strip_text to
|
||||
# avoid shadowing str.strip on this str-based enum.
|
||||
strip_text = 'strip'
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges: str) -> set[int]:
|
||||
"""Convert page range string to set of page numbers."""
|
||||
class TaggedPdfMode(StrEnum):
|
||||
"""Control behavior when encountering a Tagged PDF.
|
||||
|
||||
Tagged PDFs often indicate documents generated from office applications
|
||||
that may not need OCR. This enum controls how OCRmyPDF handles them:
|
||||
|
||||
- ``default``: Error if ProcessingMode is default, otherwise warn
|
||||
- ``ignore``: Always warn but continue processing (never error)
|
||||
"""
|
||||
|
||||
default = 'default'
|
||||
ignore = 'ignore'
|
||||
|
||||
|
||||
def _has_end_alias(ranges: str) -> bool:
|
||||
"""Return True if the page range string uses the ``end`` alias."""
|
||||
return 'end' in ranges.lower()
|
||||
|
||||
|
||||
def _resolve_page_token(token: str, total_pages: int | None) -> int:
|
||||
"""Convert a single page-number token to a 1-based integer.
|
||||
|
||||
The literal ``end`` (case-insensitive) is resolved to ``total_pages``. If
|
||||
``total_pages`` is None, an error is raised.
|
||||
"""
|
||||
if token.lower() == 'end':
|
||||
if total_pages is None:
|
||||
raise BadArgsError(
|
||||
"'end' was used in --pages but the total page count is not yet known"
|
||||
)
|
||||
return total_pages
|
||||
return int(token)
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges: str, total_pages: int | None = None) -> set[int]:
|
||||
"""Convert page range string to set of 0-based page numbers.
|
||||
|
||||
The token ``end`` (case-insensitive) is an alias for the last page of the
|
||||
document. It is resolved using ``total_pages``; if ``end`` appears in the
|
||||
string and ``total_pages`` is None, a :class:`BadArgsError` is raised.
|
||||
"""
|
||||
pages: list[int] = []
|
||||
page_groups = ranges.replace(' ', '').split(',')
|
||||
for group in page_groups:
|
||||
@@ -60,10 +104,15 @@ def _pages_from_ranges(ranges: str) -> set[int]:
|
||||
try:
|
||||
start, end = group.split('-')
|
||||
except ValueError:
|
||||
pages.append(int(group) - 1)
|
||||
try:
|
||||
pages.append(_resolve_page_token(group, total_pages) - 1)
|
||||
except ValueError:
|
||||
raise BadArgsError(f"invalid page number '{group}'") from None
|
||||
else:
|
||||
try:
|
||||
new_pages = list(range(int(start) - 1, int(end)))
|
||||
start_n = _resolve_page_token(start, total_pages)
|
||||
end_n = _resolve_page_token(end, total_pages)
|
||||
new_pages = list(range(start_n - 1, end_n))
|
||||
if not new_pages:
|
||||
raise BadArgsError(
|
||||
f"invalid page subrange '{start}-{end}'"
|
||||
@@ -142,14 +191,13 @@ class OcrOptions(BaseModel):
|
||||
remove_background: bool = False
|
||||
remove_vectors: bool = False
|
||||
oversample: int = 0
|
||||
unpaper_args: str | list[str] | None = (
|
||||
None # Can be string or list after validation
|
||||
)
|
||||
unpaper_args: list[str] | None = None
|
||||
|
||||
# OCR behavior
|
||||
skip_big: float | None = None
|
||||
pages: str | set[int] | None = None # Can be string or set after validation
|
||||
invalidate_digital_signatures: bool = False
|
||||
tagged_pdf_mode: TaggedPdfMode = TaggedPdfMode.default
|
||||
|
||||
# Metadata
|
||||
title: str | None = None
|
||||
@@ -159,23 +207,25 @@ class OcrOptions(BaseModel):
|
||||
|
||||
# Optimization
|
||||
optimize: int = 1
|
||||
jpg_quality: int | None = None
|
||||
jpeg_quality: int | None = None
|
||||
png_quality: int | None = None
|
||||
jbig2_threshold: float = 0.85
|
||||
|
||||
# Compatibility alias for plugins that expect jpeg_quality
|
||||
# Deprecated compatibility alias for code that still uses the old field name
|
||||
@property
|
||||
def jpeg_quality(self):
|
||||
"""Compatibility alias for jpg_quality."""
|
||||
return self.jpg_quality
|
||||
def jpg_quality(self):
|
||||
"""Deprecated compatibility alias for jpeg_quality."""
|
||||
return self.jpeg_quality
|
||||
|
||||
@jpeg_quality.setter
|
||||
def jpeg_quality(self, value):
|
||||
"""Compatibility alias for jpg_quality."""
|
||||
self.jpg_quality = value
|
||||
@jpg_quality.setter
|
||||
def jpg_quality(self, value):
|
||||
"""Deprecated compatibility alias for jpeg_quality."""
|
||||
self.jpeg_quality = value
|
||||
|
||||
# Output behavior
|
||||
no_overwrite: bool = False
|
||||
|
||||
# Advanced options
|
||||
max_image_mpixels: float = 250.0
|
||||
max_image_mpixels: float | None = None
|
||||
pdf_renderer: str = 'auto'
|
||||
ocr_engine: str = 'auto'
|
||||
rasterizer: str = 'auto'
|
||||
@@ -190,7 +240,7 @@ class OcrOptions(BaseModel):
|
||||
tesseract_pagesegmode: int | None = None
|
||||
tesseract_oem: int | None = None
|
||||
tesseract_thresholding: int | None = None
|
||||
tesseract_timeout: float = 0.0
|
||||
tesseract_timeout: float | None = None
|
||||
tesseract_non_ocr_timeout: float | None = None
|
||||
tesseract_downsample_above: int = 32767
|
||||
tesseract_downsample_large_images: bool | None = None
|
||||
@@ -284,7 +334,7 @@ class OcrOptions(BaseModel):
|
||||
@classmethod
|
||||
def validate_max_image_mpixels(cls, v):
|
||||
"""Validate max image megapixels."""
|
||||
if v < 0:
|
||||
if v is not None and v < 0:
|
||||
raise ValueError("max_image_mpixels must be non-negative")
|
||||
return v
|
||||
|
||||
@@ -315,15 +365,37 @@ class OcrOptions(BaseModel):
|
||||
@field_validator('pages')
|
||||
@classmethod
|
||||
def validate_pages_format(cls, v):
|
||||
"""Convert page ranges string to set of page numbers."""
|
||||
"""Convert page ranges string to set of page numbers.
|
||||
|
||||
If the string uses the ``end`` alias, the original string is preserved
|
||||
so that resolution can happen later, once the document's page count is
|
||||
known.
|
||||
"""
|
||||
if v is None:
|
||||
return v
|
||||
if isinstance(v, set):
|
||||
return v # Already processed
|
||||
if _has_end_alias(v):
|
||||
# Defer resolution until total page count is known
|
||||
return v
|
||||
|
||||
# Convert string ranges to set of page numbers
|
||||
return _pages_from_ranges(v)
|
||||
|
||||
@field_validator('unpaper_args', mode='before')
|
||||
@classmethod
|
||||
def validate_unpaper_args(cls, v):
|
||||
"""Normalize unpaper_args from string to list and validate security."""
|
||||
if v is None:
|
||||
return v
|
||||
if isinstance(v, str):
|
||||
v = shlex.split(v)
|
||||
if isinstance(v, list):
|
||||
if any(('/' in arg or arg == '.' or arg == '..') for arg in v):
|
||||
raise ValueError('No filenames allowed in --unpaper-args')
|
||||
return v
|
||||
raise ValueError(f'unpaper_args must be a string or list, got {type(v)}')
|
||||
|
||||
@model_validator(mode='before')
|
||||
@classmethod
|
||||
def handle_special_cases(cls, data):
|
||||
@@ -392,7 +464,7 @@ class OcrOptions(BaseModel):
|
||||
):
|
||||
raise ValueError(
|
||||
"Since you specified `--output-type none`, the output file "
|
||||
f"{self.output_file} cannot be produced. Set the output file to "
|
||||
f"{str(self.output_file)} cannot be produced. Set the output file to "
|
||||
f"`-` to suppress this message."
|
||||
)
|
||||
return self
|
||||
@@ -503,7 +575,7 @@ class OcrOptions(BaseModel):
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def register_plugin_models(cls, models: dict[str, type]) -> None:
|
||||
def register_plugin_models(cls, models: dict[str, type[BaseModel]]) -> None:
|
||||
"""Register plugin option model classes for nested access.
|
||||
|
||||
Args:
|
||||
@@ -551,6 +623,13 @@ class OcrOptions(BaseModel):
|
||||
value = getattr(self, flat_name)
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
# Plugin-scoped fields that aren't in the central OcrOptions
|
||||
# registry: argparse stores them in extra_attrs under the
|
||||
# namespace_field name.
|
||||
elif flat_name in self.extra_attrs:
|
||||
value = self.extra_attrs[flat_name]
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
# Also check direct field name (for fields like jbig2_lossy)
|
||||
elif field_name in OcrOptions.model_fields:
|
||||
value = getattr(self, field_name)
|
||||
@@ -563,12 +642,6 @@ class OcrOptions(BaseModel):
|
||||
value = self.optimize
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
elif namespace == 'optimize' and field_name == 'jpeg_quality':
|
||||
# jpg_quality maps to jpeg_quality
|
||||
if 'jpg_quality' in OcrOptions.model_fields:
|
||||
value = self.jpg_quality
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
|
||||
# Create and cache the plugin options instance
|
||||
instance = model_class(**kwargs)
|
||||
|
||||
@@ -0,0 +1,253 @@
|
||||
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
||||
# SPDX-FileCopyrightText: 2025 ajdlinux
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Validate and repair malformed page-boundary boxes.
|
||||
|
||||
A page's boundary boxes (``/MediaBox``, ``/CropBox``, ``/TrimBox``, ``/ArtBox``,
|
||||
``/BleedBox``) are sometimes malformed in ways that PDF readers tolerate but
|
||||
that crash or corrupt downstream processing. This module normalizes them in
|
||||
place following the PDF 2.0 specification (ISO 32000-2:2020):
|
||||
|
||||
- **Non-decimal coordinates** (§7.3.3): a coordinate written in exponential
|
||||
notation is invalid PDF number syntax and is stored by qpdf/pikepdf as a
|
||||
string. We coerce it back to a number (issue #1398).
|
||||
- **Reversed corners** (§7.9.5): a rectangle is "a pair of diagonally opposite
|
||||
corners"; ``[llx lly urx ury]`` is only the typical order. We normalize to
|
||||
``[min_x, min_y, max_x, max_y]`` (issue #1526).
|
||||
- **Sub-box outside the MediaBox** (§14.11.2): "If the bounds of the crop,
|
||||
trim, bleed or art box extends outside of the bounds of the media box, a
|
||||
processor shall treat the box as its intersection with the media box." We
|
||||
clamp to that intersection, or discard the sub-box (so it inherits the
|
||||
MediaBox) when the intersection is empty (issue #1400).
|
||||
|
||||
A rectangle is treated as empty when its width or height is ``<= 0``; PDF 2.0
|
||||
permits zero-dimension rectangles and defines no minimum page size, so no other
|
||||
size floor is imposed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
from collections.abc import Iterable, Mapping, Sequence
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pikepdf
|
||||
from pikepdf import Name
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
_SUBBOXES = ('CropBox', 'TrimBox', 'ArtBox', 'BleedBox')
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BoxRepair:
|
||||
"""A single change made to a page box.
|
||||
|
||||
Attributes:
|
||||
box: The box name, e.g. ``"CropBox"``.
|
||||
kind: One of ``"reordered"`` (reversed corners normalized; lossless),
|
||||
``"recoded"`` (non-numeric/exponential coordinate coerced),
|
||||
``"clamped"`` (sub-box clamped to the MediaBox), ``"discarded"``
|
||||
(sub-box removed because its MediaBox intersection was empty), or
|
||||
``"degenerate_mediabox"`` (MediaBox has zero width or height).
|
||||
"""
|
||||
|
||||
box: str
|
||||
kind: str
|
||||
|
||||
|
||||
def _read_box(values: Sequence) -> tuple[list[float], bool, bool] | None:
|
||||
"""Coerce a box array to floats and normalize corner order.
|
||||
|
||||
Returns ``(normalized_values, recoded, reordered)`` where ``recoded`` is
|
||||
True if any element needed string/exponential coercion and ``reordered`` is
|
||||
True if the corners were given in non-standard order. Returns None if the
|
||||
array is not four finite numbers.
|
||||
"""
|
||||
if len(values) != 4:
|
||||
return None
|
||||
nums: list[float] = []
|
||||
recoded = False
|
||||
for v in values:
|
||||
try:
|
||||
n = float(v)
|
||||
except (TypeError, ValueError):
|
||||
try:
|
||||
n = float(str(v))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
recoded = True
|
||||
if not math.isfinite(n):
|
||||
return None
|
||||
nums.append(n)
|
||||
x0, y0, x1, y1 = nums
|
||||
normalized = [min(x0, x1), min(y0, y1), max(x0, x1), max(y0, y1)]
|
||||
reordered = normalized != nums
|
||||
return normalized, recoded, reordered
|
||||
|
||||
|
||||
def coerce_box(values: Iterable) -> list[float]:
|
||||
"""Return box values coerced to floats with corner order normalized.
|
||||
|
||||
Robust against exponential/string coordinates and reversed corners, so
|
||||
callers that only need to read a box (e.g. dimension calculations) do not
|
||||
crash on malformed input. Falls back to best-effort per-element coercion if
|
||||
the array is not four numbers.
|
||||
"""
|
||||
values = list(values)
|
||||
result = _read_box(values)
|
||||
if result is not None:
|
||||
return result[0]
|
||||
coerced = []
|
||||
for v in values:
|
||||
try:
|
||||
coerced.append(float(v))
|
||||
except (TypeError, ValueError):
|
||||
coerced.append(float(str(v)))
|
||||
return coerced
|
||||
|
||||
|
||||
def _is_empty(box: Sequence[float]) -> bool:
|
||||
"""A rectangle is empty when its width or height is non-positive."""
|
||||
return (box[2] - box[0]) <= 0 or (box[3] - box[1]) <= 0
|
||||
|
||||
|
||||
def repair_page_boxes(page: pikepdf.Page) -> list[BoxRepair]:
|
||||
"""Validate and repair the boundary boxes of a single page, in place.
|
||||
|
||||
Returns the list of changes made (empty if the page was already valid).
|
||||
Only boxes that actually change are written back, so valid pages are left
|
||||
untouched. Performs no logging or I/O.
|
||||
"""
|
||||
repairs: list[BoxRepair] = []
|
||||
|
||||
# MediaBox is the reference rectangle; read it inheritance-aware.
|
||||
mediabox: list[float] | None = None
|
||||
try:
|
||||
mb_result = _read_box(list(page.mediabox.as_list()))
|
||||
except (AttributeError, KeyError, RuntimeError):
|
||||
mb_result = None
|
||||
if mb_result is not None:
|
||||
mediabox, recoded, reordered = mb_result
|
||||
if reordered:
|
||||
repairs.append(BoxRepair('MediaBox', 'reordered'))
|
||||
if recoded:
|
||||
repairs.append(BoxRepair('MediaBox', 'recoded'))
|
||||
if recoded or reordered:
|
||||
page.obj.MediaBox = pikepdf.Array(mediabox)
|
||||
if _is_empty(mediabox):
|
||||
repairs.append(BoxRepair('MediaBox', 'degenerate_mediabox'))
|
||||
mediabox = None # don't clamp against a degenerate reference
|
||||
|
||||
for box in _SUBBOXES:
|
||||
name = Name('/' + box)
|
||||
if name not in page.obj:
|
||||
continue
|
||||
try:
|
||||
sub_result = _read_box(list(page.obj[name]))
|
||||
except (TypeError, RuntimeError):
|
||||
continue
|
||||
if sub_result is None:
|
||||
continue
|
||||
values, recoded, reordered = sub_result
|
||||
if reordered:
|
||||
repairs.append(BoxRepair(box, 'reordered'))
|
||||
if recoded:
|
||||
repairs.append(BoxRepair(box, 'recoded'))
|
||||
if recoded or reordered:
|
||||
page.obj[name] = pikepdf.Array(values)
|
||||
|
||||
if mediabox is None:
|
||||
continue
|
||||
intersection = [
|
||||
max(values[0], mediabox[0]),
|
||||
max(values[1], mediabox[1]),
|
||||
min(values[2], mediabox[2]),
|
||||
min(values[3], mediabox[3]),
|
||||
]
|
||||
if _is_empty(intersection):
|
||||
del page.obj[name]
|
||||
repairs.append(BoxRepair(box, 'discarded'))
|
||||
elif intersection != values:
|
||||
page.obj[name] = pikepdf.Array(intersection)
|
||||
repairs.append(BoxRepair(box, 'clamped'))
|
||||
|
||||
return repairs
|
||||
|
||||
|
||||
# Per-kind log severity and message template ({box} is substituted).
|
||||
_KIND_MESSAGES: dict[str, tuple[int, str]] = {
|
||||
'discarded': (
|
||||
logging.WARNING,
|
||||
'{box} lies outside the MediaBox and was discarded; '
|
||||
'the full page will be shown',
|
||||
),
|
||||
'clamped': (
|
||||
logging.WARNING,
|
||||
'{box} extended beyond the MediaBox and was clamped to it',
|
||||
),
|
||||
'recoded': (
|
||||
logging.WARNING,
|
||||
'{box} used invalid (e.g. exponential) coordinates, which were reinterpreted',
|
||||
),
|
||||
'degenerate_mediabox': (
|
||||
logging.WARNING,
|
||||
'MediaBox has zero width or height and could not be repaired; '
|
||||
'output may be invalid',
|
||||
),
|
||||
'reordered': (
|
||||
logging.DEBUG,
|
||||
'{box} corners were reversed and have been normalized',
|
||||
),
|
||||
}
|
||||
|
||||
# Kinds that change page appearance and warrant manual review of the output.
|
||||
_INSPECT_KINDS = frozenset({'discarded', 'clamped', 'recoded'})
|
||||
_INSPECT = ' Please visually inspect the output PDF.'
|
||||
|
||||
|
||||
def _format_pages(pagenos: Iterable[int]) -> str:
|
||||
"""Format 0-based page numbers as a compact 1-based range string."""
|
||||
nums = sorted(p + 1 for p in pagenos)
|
||||
ranges: list[tuple[int, int]] = []
|
||||
start = prev = nums[0]
|
||||
for n in nums[1:]:
|
||||
if n == prev + 1:
|
||||
prev = n
|
||||
continue
|
||||
ranges.append((start, prev))
|
||||
start = prev = n
|
||||
ranges.append((start, prev))
|
||||
return ', '.join(f'{a}' if a == b else f'{a}-{b}' for a, b in ranges)
|
||||
|
||||
|
||||
def summarize_box_repairs(
|
||||
repairs_by_page: Mapping[int, Sequence[BoxRepair]],
|
||||
) -> list[tuple[int, str]]:
|
||||
"""Aggregate per-page repairs into ``(log_level, message)`` pairs.
|
||||
|
||||
Repairs are grouped by ``(kind, box)`` so a defect shared across many pages
|
||||
yields a single message listing the affected pages, rather than one message
|
||||
per page.
|
||||
"""
|
||||
groups: dict[tuple[str, str], set[int]] = {}
|
||||
for pageno, repairs in repairs_by_page.items():
|
||||
for repair in repairs:
|
||||
groups.setdefault((repair.kind, repair.box), set()).add(pageno)
|
||||
|
||||
messages: list[tuple[int, str]] = []
|
||||
for (kind, box), pages in sorted(groups.items()):
|
||||
level, template = _KIND_MESSAGES[kind]
|
||||
text = f'Page(s) {_format_pages(pages)}: {template.format(box=box)}.'
|
||||
if kind in _INSPECT_KINDS:
|
||||
text += _INSPECT
|
||||
messages.append((level, text))
|
||||
return messages
|
||||
|
||||
|
||||
def log_box_repairs(repairs_by_page: Mapping[int, Sequence[BoxRepair]]) -> None:
|
||||
"""Emit aggregated log messages for the repairs made across all pages."""
|
||||
for level, message in summarize_box_repairs(repairs_by_page):
|
||||
log.log(level, message)
|
||||
+183
-77
@@ -28,23 +28,29 @@ from ocrmypdf._concurrent import Executor
|
||||
from ocrmypdf._exec import unpaper
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._metadata import repair_docinfo_nuls
|
||||
from ocrmypdf._options import OcrOptions, ProcessingMode
|
||||
from ocrmypdf._options import OcrOptions, PathOrIO, ProcessingMode, TaggedPdfMode
|
||||
from ocrmypdf._pageboxes import log_box_repairs, repair_page_boxes
|
||||
from ocrmypdf._stdoutprotect import get_protected_stdout_fd
|
||||
from ocrmypdf.exceptions import (
|
||||
ColorConversionNeededError,
|
||||
DigitalSignatureError,
|
||||
DpiError,
|
||||
EncryptedPdfError,
|
||||
InputFileError,
|
||||
NonEmbeddedFontsError,
|
||||
PriorOcrFoundError,
|
||||
SubprocessOutputError,
|
||||
TaggedPDFError,
|
||||
UnsupportedImageFormatError,
|
||||
)
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
|
||||
from ocrmypdf.pdfa import (
|
||||
file_claims_pdfa,
|
||||
find_nonembedded_cid_fonts,
|
||||
generate_pdfa_ps,
|
||||
speculative_pdfa_conversion,
|
||||
)
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, FloatRect, PageInfo, PdfInfo
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, FloatRect, Ink, PageInfo, PdfInfo
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice, OrientationConfidence
|
||||
|
||||
try:
|
||||
@@ -116,8 +122,7 @@ def triage_image_file(input_file: Path, output_file: Path, options: OcrOptions)
|
||||
|
||||
if im.mode in ('RGBA', 'LA'):
|
||||
raise UnsupportedImageFormatError(
|
||||
"The input image has an alpha channel. Remove the alpha "
|
||||
"channel first."
|
||||
"The input image has an alpha channel. Remove the alpha channel first."
|
||||
)
|
||||
|
||||
if 'iccprofile' not in im.info:
|
||||
@@ -135,7 +140,7 @@ def triage_image_file(input_file: Path, output_file: Path, options: OcrOptions)
|
||||
layout_fun = img2pdf.get_fixed_dpi_layout_fun(
|
||||
Resolution(options.image_dpi, options.image_dpi)
|
||||
)
|
||||
with open(output_file, 'wb') as outf:
|
||||
with output_file.open('wb') as outf:
|
||||
img2pdf.convert(
|
||||
os.fspath(input_file),
|
||||
layout_fun=layout_fun,
|
||||
@@ -154,7 +159,7 @@ def _pdf_guess_version(input_file: Path, search_window=1024) -> str:
|
||||
|
||||
Returns empty string if not found, indicating file is probably not PDF.
|
||||
"""
|
||||
with open(input_file, 'rb') as f:
|
||||
with input_file.open('rb') as f:
|
||||
signature = f.read(search_window)
|
||||
m = re.search(rb'%PDF-(\d\.\d)', signature)
|
||||
if m:
|
||||
@@ -175,6 +180,12 @@ def triage(
|
||||
)
|
||||
try:
|
||||
with pikepdf.open(input_file) as pdf:
|
||||
repairs_by_page = {
|
||||
n: repairs
|
||||
for n, page in enumerate(pdf.pages)
|
||||
if (repairs := repair_page_boxes(page))
|
||||
}
|
||||
log_box_repairs(repairs_by_page)
|
||||
pdf.save(output_file)
|
||||
except pikepdf.PdfError as e:
|
||||
raise InputFileError() from e
|
||||
@@ -250,15 +261,21 @@ def validate_pdfinfo_options(context: PdfContext) -> None:
|
||||
"image of the form and all filled form fields. The output PDF "
|
||||
"will be 'flattened' and will no longer be fillable."
|
||||
)
|
||||
if pdfinfo.is_tagged:
|
||||
if options.mode != ProcessingMode.default:
|
||||
log.warning(
|
||||
"This PDF is marked as a Tagged PDF. This often indicates "
|
||||
"that the PDF was generated from an office document and does "
|
||||
"not need OCR. PDF pages processed by OCRmyPDF may not be "
|
||||
"tagged correctly."
|
||||
)
|
||||
else:
|
||||
if pdfinfo.is_tagged or pdfinfo.has_structure_tree:
|
||||
log.warning(
|
||||
"This PDF contains structural markup (it is a Tagged PDF or "
|
||||
"carries a logical structure tree). This often indicates that the "
|
||||
"PDF was generated from an office document or is otherwise born "
|
||||
"digital, and does not need OCR. OCRmyPDF cannot rebuild this "
|
||||
"structure to match new text, so any page it re-OCRs with "
|
||||
"--force-ocr or --redo-ocr will have its structural markup "
|
||||
"discarded."
|
||||
)
|
||||
if (
|
||||
options.tagged_pdf_mode == TaggedPdfMode.default
|
||||
and options.mode == ProcessingMode.default
|
||||
):
|
||||
log.info("Use --tagged-pdf-mode ignore to ignore Tagged PDFs.")
|
||||
raise TaggedPDFError()
|
||||
context.plugin_manager.validate(pdfinfo=pdfinfo, options=options)
|
||||
|
||||
@@ -322,6 +339,11 @@ def is_ocr_required(page_context: PageContext) -> bool:
|
||||
pageinfo = page_context.pageinfo
|
||||
options = page_context.options
|
||||
|
||||
if options.mode == ProcessingMode.strip_text:
|
||||
# Strip mode removes the OCR text layer in place; it never rasterizes
|
||||
# or runs OCR. The stripping happens in OcrGrafter.graft_page.
|
||||
return False
|
||||
|
||||
ocr_required = True
|
||||
|
||||
if options.pages and pageinfo.pageno not in options.pages:
|
||||
@@ -505,6 +527,49 @@ def calculate_raster_dpi(page_context: PageContext):
|
||||
return canvas_dpi, page_dpi
|
||||
|
||||
|
||||
def _select_raster_device(pageinfo: PageInfo) -> GhostscriptRasterDevice:
|
||||
"""Choose the minimum raster device that preserves the page's color depth.
|
||||
|
||||
The device escalates from 1-bit mono through grayscale, indexed, and full
|
||||
color as required by the page's images, image masks, and vector content.
|
||||
Image masks are painted with the current fill color, so a mask painted in
|
||||
gray or color escalates the device even though the mask itself is 1-bit.
|
||||
"""
|
||||
colorspaces = [
|
||||
GhostscriptRasterDevice.PNGMONOD,
|
||||
GhostscriptRasterDevice.PNGGRAY,
|
||||
GhostscriptRasterDevice.PNG256,
|
||||
GhostscriptRasterDevice.PNG16M,
|
||||
]
|
||||
device_idx = 0
|
||||
|
||||
def at_least(colorspace):
|
||||
return max(device_idx, colorspaces.index(colorspace))
|
||||
|
||||
for image in pageinfo.images:
|
||||
if image.type_ == 'stencil':
|
||||
# The fill color used to paint the mask, not the 1-bit mask data,
|
||||
# determines the color depth OCR needs.
|
||||
if image.ink == Ink.color:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
elif image.ink == Ink.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
continue
|
||||
if image.bpc > 1:
|
||||
if image.color == Colorspace.index:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG256)
|
||||
elif image.color == Colorspace.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
else:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
if pageinfo.has_vector:
|
||||
log.debug(f"Page has vector content, using {GhostscriptRasterDevice.PNG16M}")
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
return colorspaces[device_idx]
|
||||
|
||||
|
||||
def rasterize(
|
||||
input_file: Path,
|
||||
page_context: PageContext,
|
||||
@@ -526,39 +591,13 @@ def rasterize(
|
||||
Returns:
|
||||
Path: The output PNG file path.
|
||||
"""
|
||||
colorspaces = [
|
||||
GhostscriptRasterDevice.PNGMONO,
|
||||
GhostscriptRasterDevice.PNGGRAY,
|
||||
GhostscriptRasterDevice.PNG256,
|
||||
GhostscriptRasterDevice.PNG16M,
|
||||
]
|
||||
device_idx = 0
|
||||
|
||||
if remove_vectors is None:
|
||||
remove_vectors = page_context.options.remove_vectors
|
||||
|
||||
output_file = page_context.get_path(f'rasterize{output_tag}.png')
|
||||
pageinfo = page_context.pageinfo
|
||||
|
||||
def at_least(colorspace):
|
||||
return max(device_idx, colorspaces.index(colorspace))
|
||||
|
||||
for image in pageinfo.images:
|
||||
if image.type_ != 'image':
|
||||
continue # ignore masks
|
||||
if image.bpc > 1:
|
||||
if image.color == Colorspace.index:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG256)
|
||||
elif image.color == Colorspace.gray:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNGGRAY)
|
||||
else:
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
if pageinfo.has_vector:
|
||||
log.debug(f"Page has vector content, using {GhostscriptRasterDevice.PNG16M}")
|
||||
device_idx = at_least(GhostscriptRasterDevice.PNG16M)
|
||||
|
||||
device = colorspaces[device_idx]
|
||||
device = _select_raster_device(pageinfo)
|
||||
|
||||
log.debug(
|
||||
f"Rasterize with {device}, rotation {correction}, mediabox {pageinfo.mediabox}"
|
||||
@@ -644,6 +683,7 @@ def create_ocr_image(image: Path, page_context: PageContext) -> Path:
|
||||
"""
|
||||
output_file = page_context.get_path('ocr.png')
|
||||
options = page_context.options
|
||||
im: Image.Image
|
||||
with Image.open(image) as im:
|
||||
log.debug('resolution %r', im.info['dpi'])
|
||||
|
||||
@@ -735,7 +775,8 @@ def ocr_engine_direct(
|
||||
def should_visible_page_image_use_jpg(pageinfo: PageInfo) -> bool:
|
||||
"""Determines whether the visible page image should be saved as a JPEG.
|
||||
|
||||
If all images were JPEGs originally, permit a JPEG as output.
|
||||
If all images were JPEGs originally (including FlateDecode+DCTDecode),
|
||||
permit a JPEG as output.
|
||||
|
||||
Args:
|
||||
pageinfo: The PageInfo object containing information about the page.
|
||||
@@ -744,7 +785,7 @@ def should_visible_page_image_use_jpg(pageinfo: PageInfo) -> bool:
|
||||
A boolean indicating whether the visible page image should be saved as a JPEG.
|
||||
"""
|
||||
return bool(pageinfo.images) and all(
|
||||
im.enc == Encoding.jpeg for im in pageinfo.images
|
||||
im.enc in (Encoding.jpeg, Encoding.flate_jpeg) for im in pageinfo.images
|
||||
)
|
||||
|
||||
|
||||
@@ -792,7 +833,7 @@ def create_pdf_page_from_image(
|
||||
|
||||
# Create a new single page PDF to hold
|
||||
bio = BytesIO()
|
||||
with open(image, 'rb') as imfile:
|
||||
with image.open('rb') as imfile:
|
||||
log.debug('convert')
|
||||
|
||||
layout_fun = img2pdf.get_layout_fun(pagesize)
|
||||
@@ -941,6 +982,12 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext) -
|
||||
# pikepdf can deal with this, but we make the world a better place by
|
||||
# stamping them out as soon as possible.
|
||||
with pikepdf.open(input_pdf) as pdf_file:
|
||||
# Ghostscript would substitute and re-embed any non-embedded CID font to
|
||||
# satisfy PDF/A, corrupting CJK text (e.g. an Acrobat OCR layer) in the
|
||||
# process. Refuse rather than silently damage the user's text layer.
|
||||
nonembedded = find_nonembedded_cid_fonts(pdf_file)
|
||||
if nonembedded:
|
||||
raise NonEmbeddedFontsError(nonembedded)
|
||||
if repair_docinfo_nuls(pdf_file):
|
||||
pdf_file.save(fix_docinfo_file)
|
||||
else:
|
||||
@@ -1034,14 +1081,46 @@ def try_speculative_pdfa(input_pdf: Path, context: PdfContext) -> Path | None:
|
||||
return None
|
||||
|
||||
|
||||
def _ghostscript_pdfa_fallback(input_pdf: Path, context: PdfContext) -> Path | None:
|
||||
"""Best-effort PDF/A conversion via Ghostscript for 'auto' output type.
|
||||
|
||||
Returns the converted PDF/A path, or None if Ghostscript is unavailable,
|
||||
fails, or cannot produce valid PDF/A. Never raises: 'auto' mode degrades to
|
||||
a regular PDF instead of erroring or emitting corrupted output.
|
||||
|
||||
Args:
|
||||
input_pdf: Path to the PDF to convert.
|
||||
context: The PDF context.
|
||||
"""
|
||||
from ocrmypdf._exec import ghostscript
|
||||
|
||||
if not ghostscript.available():
|
||||
return None
|
||||
try:
|
||||
ps_stub = generate_postscript_stub(context)
|
||||
gs_out = convert_to_pdfa(input_pdf, ps_stub, context)
|
||||
except (
|
||||
SubprocessOutputError,
|
||||
ColorConversionNeededError,
|
||||
NonEmbeddedFontsError,
|
||||
) as e:
|
||||
log.info('Auto mode: Ghostscript could not produce PDF/A (%s)', e)
|
||||
return None
|
||||
if not file_claims_pdfa(gs_out)['pass']:
|
||||
log.info('Auto mode: Ghostscript output is not valid PDF/A')
|
||||
return None
|
||||
return gs_out
|
||||
|
||||
|
||||
def try_auto_pdfa(input_pdf: Path, context: PdfContext) -> tuple[Path, str]:
|
||||
"""Best-effort PDF/A for 'auto' output type.
|
||||
|
||||
This function attempts to produce PDF/A without requiring Ghostscript:
|
||||
1. If verapdf is available, tries speculative conversion with validation
|
||||
2. Without verapdf, passes through as PDF/A if safe (input already PDF/A
|
||||
or force-ocr was used)
|
||||
3. Falls back to regular PDF if neither condition is met
|
||||
Order of attempts, first success wins:
|
||||
1. Non-embedded CID fonts -> regular PDF (Ghostscript would corrupt them).
|
||||
2. Speculative conversion validated by verapdf (no Ghostscript).
|
||||
3. Without verapdf, pass through if already PDF/A or rebuilt with force-ocr.
|
||||
4. Ghostscript conversion (best-effort; failures fall through).
|
||||
5. Regular PDF if none of the above produced PDF/A.
|
||||
|
||||
Args:
|
||||
input_pdf: Path to the PDF to convert
|
||||
@@ -1053,25 +1132,42 @@ def try_auto_pdfa(input_pdf: Path, context: PdfContext) -> tuple[Path, str]:
|
||||
"""
|
||||
from ocrmypdf._exec import verapdf
|
||||
|
||||
# If verapdf available, try speculative conversion with validation
|
||||
# Non-embedded CID fonts cannot be made PDF/A without Ghostscript font
|
||||
# substitution that corrupts CID/CJK text. Rather than risk an existing
|
||||
# text layer, downgrade to a regular PDF (the same outcome as any other
|
||||
# case where best-effort PDF/A is not achievable).
|
||||
with pikepdf.open(input_pdf) as pdf_file:
|
||||
nonembedded = find_nonembedded_cid_fonts(pdf_file)
|
||||
if nonembedded:
|
||||
log.info(
|
||||
"Auto mode: input has non-embedded CID fonts (%s) that cannot be "
|
||||
"converted to PDF/A without corrupting the text; outputting a "
|
||||
"regular PDF. Use --output-type pdf to select this explicitly.",
|
||||
', '.join(sorted(nonembedded)),
|
||||
)
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
# Cheap path: speculative conversion validated by verapdf (no Ghostscript).
|
||||
if verapdf.available():
|
||||
result = try_speculative_pdfa(input_pdf, context)
|
||||
if result is not None:
|
||||
return (result, 'pdfa')
|
||||
# verapdf validation failed - fall through to regular PDF
|
||||
log.info(
|
||||
'Auto mode: speculative PDF/A validation failed, outputting regular PDF'
|
||||
)
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
# Without verapdf, check if we can pass through as PDF/A
|
||||
if _is_safe_pdfa(input_pdf, context.options):
|
||||
# Pass through as-is (no modifications needed)
|
||||
log.info('Auto mode: speculative PDF/A validation failed')
|
||||
elif _is_safe_pdfa(input_pdf, context.options):
|
||||
# No verapdf, but the input is already PDF/A or was rebuilt with
|
||||
# --force-ocr, so we can pass it through without Ghostscript.
|
||||
log.info('Auto mode: passing through as PDF/A (input already compliant)')
|
||||
return (input_pdf, 'pdfa')
|
||||
|
||||
# Fall through to regular PDF
|
||||
log.info('Auto mode: no verapdf available and input is not PDF/A, outputting PDF')
|
||||
# Fall back to Ghostscript to produce real PDF/A (v16 behavior). Best-effort:
|
||||
# if Ghostscript is unavailable or cannot safely produce PDF/A, keep a
|
||||
# regular PDF rather than error.
|
||||
gs_out = _ghostscript_pdfa_fallback(input_pdf, context)
|
||||
if gs_out is not None:
|
||||
log.info('Auto mode: produced PDF/A via Ghostscript')
|
||||
return (gs_out, 'pdfa')
|
||||
|
||||
log.info('Auto mode: could not produce PDF/A, outputting regular PDF')
|
||||
return (input_pdf, 'pdf')
|
||||
|
||||
|
||||
@@ -1103,7 +1199,7 @@ def should_linearize(working_file: Path, context: PdfContext) -> bool:
|
||||
|
||||
For smaller files, linearization is not worth the effort.
|
||||
"""
|
||||
filesize = os.stat(working_file).st_size
|
||||
filesize = working_file.stat().st_size
|
||||
return filesize > (context.options.fast_web_view * 1_000_000)
|
||||
|
||||
|
||||
@@ -1189,7 +1285,8 @@ def enumerate_compress_ranges(
|
||||
A tuple containing a range of indices and the corresponding element.
|
||||
If the element is None, the range represents a skipped range of indices.
|
||||
"""
|
||||
skipped_from, index = None, None
|
||||
skipped_from: int | None = None
|
||||
index: int | None = None
|
||||
for index, txt_file in enumerate(iterable):
|
||||
index += 1
|
||||
if txt_file:
|
||||
@@ -1201,6 +1298,9 @@ def enumerate_compress_ranges(
|
||||
if skipped_from is None:
|
||||
skipped_from = index
|
||||
if skipped_from is not None:
|
||||
# skipped_from can only be set inside the loop above, so the loop
|
||||
# must have run at least once and index is guaranteed to be an int.
|
||||
assert index is not None
|
||||
yield (skipped_from, index), None
|
||||
|
||||
|
||||
@@ -1212,7 +1312,7 @@ def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext) -> Pat
|
||||
and returns the path to the merged file.
|
||||
"""
|
||||
output_file = context.get_path('sidecar.txt')
|
||||
with open(output_file, 'w', encoding="utf-8") as stream:
|
||||
with output_file.open('w', encoding="utf-8") as stream:
|
||||
for (from_, to_), txt_file in enumerate_compress_ranges(txt_files):
|
||||
if from_ != 1:
|
||||
stream.write('\f') # Form feed between pages for all pages after first
|
||||
@@ -1227,24 +1327,28 @@ def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext) -> Pat
|
||||
return output_file
|
||||
|
||||
|
||||
def copy_final(
|
||||
input_file: Path, output_file: str | Path | BinaryIO, original_file: Path | None
|
||||
) -> None:
|
||||
def copy_final(input_file: Path, output_file: PathOrIO) -> None:
|
||||
"""Copy the final temporary file to the output destination.
|
||||
|
||||
Args:
|
||||
input_file (Path): The intermediate input file to copy.
|
||||
output_file (str | Path | BinaryIO): The output file to copy to.
|
||||
original_file: The original file to copy attributes from.
|
||||
|
||||
Returns:
|
||||
None
|
||||
input_file: The intermediate input file to copy.
|
||||
output_file: The output file to copy to.
|
||||
"""
|
||||
log.debug('%s -> %s', input_file, output_file)
|
||||
with input_file.open('rb') as input_stream:
|
||||
if output_file == '-':
|
||||
copyfileobj(input_stream, sys.stdout.buffer) # type: ignore[misc]
|
||||
sys.stdout.flush()
|
||||
fd = get_protected_stdout_fd()
|
||||
if fd is not None:
|
||||
# Stdout protection is active: write to the preserved real
|
||||
# stdout. dup the saved fd so the with-block's close() does not
|
||||
# close our long-lived descriptor.
|
||||
with os.fdopen(os.dup(fd), 'wb') as stdout_stream:
|
||||
copyfileobj(input_stream, stdout_stream)
|
||||
stdout_stream.flush()
|
||||
else:
|
||||
# No protection installed (e.g. plain API use): legacy behavior.
|
||||
copyfileobj(input_stream, sys.stdout.buffer) # type: ignore[misc]
|
||||
sys.stdout.flush()
|
||||
elif hasattr(output_file, 'writable'):
|
||||
output_stream = cast(BinaryIO, output_file)
|
||||
copyfileobj(input_stream, output_stream) # type: ignore[misc]
|
||||
@@ -1254,5 +1358,7 @@ def copy_final(
|
||||
# At this point we overwrite the output_file specified by the user
|
||||
# use copyfileobj because then we use open() to create the file and
|
||||
# get the appropriate umask, ownership, etc.
|
||||
with open(output_file, 'w+b') as output_stream:
|
||||
# The `hasattr` check above already ruled out stream-like objects.
|
||||
assert isinstance(output_file, str | bytes | os.PathLike)
|
||||
with Path(os.fsdecode(output_file)).open('w+b') as output_stream:
|
||||
copyfileobj(input_stream, output_stream)
|
||||
|
||||
@@ -329,10 +329,13 @@ def setup_pipeline(
|
||||
# Note: OcrOptions is immutable, so we can't modify options.jobs directly
|
||||
# The jobs field should already be set correctly during OcrOptions creation
|
||||
|
||||
# Apply PIL max image pixels side effect
|
||||
PIL.Image.MAX_IMAGE_PIXELS = int(options.max_image_mpixels * 1_000_000)
|
||||
if PIL.Image.MAX_IMAGE_PIXELS == 0:
|
||||
PIL.Image.MAX_IMAGE_PIXELS = None # type: ignore
|
||||
# Apply PIL max image pixels side effect only when explicitly requested.
|
||||
# When None, leave PIL.Image.MAX_IMAGE_PIXELS as the host application
|
||||
# configured it. The CLI passes its own default (250.0) via argparse.
|
||||
if options.max_image_mpixels is not None:
|
||||
PIL.Image.MAX_IMAGE_PIXELS = int(options.max_image_mpixels * 1_000_000)
|
||||
if PIL.Image.MAX_IMAGE_PIXELS == 0:
|
||||
PIL.Image.MAX_IMAGE_PIXELS = None # type: ignore
|
||||
|
||||
pikepdf_enable_mmap()
|
||||
executor = setup_executor(plugin_manager)
|
||||
@@ -340,12 +343,17 @@ def setup_pipeline(
|
||||
|
||||
|
||||
def do_get_pdfinfo(pdf_path: Path, executor: Executor, options) -> PdfInfo:
|
||||
# Handle pages field - it might be a string that needs conversion
|
||||
# Handle pages field - it might be a string that needs conversion.
|
||||
# A string indicates the ``end`` alias was used and resolution was
|
||||
# deferred; we resolve it now using the document's actual page count.
|
||||
check_pages = options.pages
|
||||
if isinstance(check_pages, str):
|
||||
from ocrmypdf._options import _pages_from_ranges
|
||||
|
||||
check_pages = _pages_from_ranges(check_pages)
|
||||
with Pdf.open(pdf_path) as pdf:
|
||||
total_pages = len(pdf.pages)
|
||||
check_pages = _pages_from_ranges(check_pages, total_pages=total_pages)
|
||||
options.pages = check_pages
|
||||
|
||||
return get_pdfinfo(
|
||||
pdf_path,
|
||||
@@ -481,7 +489,7 @@ def postprocess(
|
||||
else:
|
||||
pdf_out = pdf_file
|
||||
if context.options.output_type == 'auto':
|
||||
# Best effort PDF/A - never uses Ghostscript
|
||||
# Best effort PDF/A - may use Ghostscript as a last resort
|
||||
pdf_out, actual_type = try_auto_pdfa(pdf_out, context)
|
||||
# Store actual output type for reporting
|
||||
context.options.extra_attrs['_actual_output_type'] = actual_type
|
||||
|
||||
@@ -98,8 +98,8 @@ def exec_hocr_to_ocr_pdf(context: PdfContext, executor: Executor) -> Sequence[st
|
||||
log.info("Postprocessing...")
|
||||
pdf, messages = postprocess(pdf, context, executor)
|
||||
|
||||
# Copy PDF file to destination (we don't know the input PDF file name)
|
||||
copy_final(pdf, options.output_file, None)
|
||||
# Copy PDF file to destination
|
||||
copy_final(pdf, options.output_file)
|
||||
return messages
|
||||
|
||||
|
||||
@@ -109,6 +109,9 @@ def run_hocr_to_ocr_pdf_pipeline(
|
||||
plugin_manager: OcrmypdfPluginManager,
|
||||
) -> ExitCode:
|
||||
"""Run pipeline to convert hOCR to final output PDF."""
|
||||
# The _hocr_to_ocr_pdf() API requires work_folder: Path and stores it on
|
||||
# options before this pipeline runs, so it is always set at this point.
|
||||
assert options.work_folder is not None
|
||||
with manage_work_folder(
|
||||
work_folder=options.work_folder, retain=True, print_location=False
|
||||
) as work_folder:
|
||||
|
||||
@@ -145,7 +145,7 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
||||
if options.sidecar:
|
||||
text = merge_sidecars(sidecars, context)
|
||||
# Copy text file to destination
|
||||
copy_final(text, options.sidecar, options.input_file)
|
||||
copy_final(text, options.sidecar)
|
||||
|
||||
# Merge layers to one single pdf
|
||||
pdf = ocrgraft.finalize()
|
||||
@@ -157,7 +157,7 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
||||
pdf, messages = postprocess(pdf, context, executor)
|
||||
|
||||
# Copy PDF file to destination
|
||||
copy_final(pdf, options.output_file, options.input_file)
|
||||
copy_final(pdf, options.output_file)
|
||||
return messages
|
||||
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import logging.handlers
|
||||
import os
|
||||
import shutil
|
||||
from functools import partial
|
||||
|
||||
@@ -91,6 +92,9 @@ def run_hocr_pipeline(
|
||||
"""Run pipeline to output hOCR."""
|
||||
if options.output_folder is None:
|
||||
raise ValueError("output_folder must be specified for hOCR pipeline")
|
||||
# This pipeline is only reachable via the _pdf_to_hocr() API, which
|
||||
# declares input_pdf: Path - streams and raw bytes paths are not supported.
|
||||
assert isinstance(options.input_file, str | os.PathLike)
|
||||
with manage_work_folder(
|
||||
work_folder=options.output_folder, retain=True, print_location=False
|
||||
) as work_folder:
|
||||
@@ -100,9 +104,7 @@ def run_hocr_pipeline(
|
||||
|
||||
# Gather pdfinfo and create context
|
||||
pdfinfo = do_get_pdfinfo(origin_pdf, executor, options)
|
||||
context = PdfContext(
|
||||
options, work_folder, options.input_file, pdfinfo, plugin_manager
|
||||
)
|
||||
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
||||
# Validate options are okay for this pdf
|
||||
validate_pdfinfo_options(context)
|
||||
exec_pdf_to_hocr(context, executor)
|
||||
|
||||
@@ -21,6 +21,7 @@ from pydantic import BaseModel
|
||||
import ocrmypdf.builtin_plugins
|
||||
from ocrmypdf import Executor, PdfContext, pluginspec
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._plugin_registry import PluginOptionRegistry
|
||||
from ocrmypdf._progressbar import ProgressBar
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pluginspec import OcrEngine
|
||||
@@ -53,10 +54,11 @@ class OcrmypdfPluginManager:
|
||||
self._plugins = plugins
|
||||
self._builtins = builtins
|
||||
self._pm = pluggy.PluginManager(*args, **kwargs)
|
||||
self._option_registry: PluginOptionRegistry | None = None
|
||||
self._setup_plugins()
|
||||
|
||||
@property
|
||||
def pluggy(self) -> pluggy.PluginManager:
|
||||
def pluggy_manager(self) -> pluggy.PluginManager:
|
||||
"""Access the underlying pluggy.PluginManager for advanced use cases.
|
||||
|
||||
This is useful for plugins that need to call methods like set_blocked()
|
||||
@@ -74,7 +76,8 @@ class OcrmypdfPluginManager:
|
||||
return state
|
||||
|
||||
def __setstate__(self, state):
|
||||
self.__init__(
|
||||
OcrmypdfPluginManager.__init__(
|
||||
self,
|
||||
*state['init_args'],
|
||||
plugins=state['plugins'],
|
||||
builtins=state['builtins'],
|
||||
@@ -86,10 +89,10 @@ class OcrmypdfPluginManager:
|
||||
|
||||
# 1. Register builtins
|
||||
if self._builtins:
|
||||
for module in sorted(
|
||||
for module_info in sorted(
|
||||
pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__)
|
||||
):
|
||||
name = f'ocrmypdf.builtin_plugins.{module.name}'
|
||||
name = f'ocrmypdf.builtin_plugins.{module_info.name}'
|
||||
module = importlib.import_module(name)
|
||||
self._pm.register(module)
|
||||
|
||||
@@ -97,17 +100,20 @@ class OcrmypdfPluginManager:
|
||||
self._pm.load_setuptools_entrypoints('ocrmypdf')
|
||||
|
||||
# 3. Register plugins specified on command line
|
||||
for name in self._plugins:
|
||||
if isinstance(name, Path) or name.endswith('.py'):
|
||||
for plugin in self._plugins:
|
||||
if isinstance(plugin, Path) or plugin.endswith('.py'):
|
||||
# Import by filename
|
||||
module_name = Path(name).stem
|
||||
spec = importlib.util.spec_from_file_location(module_name, name)
|
||||
plugin_path = Path(plugin)
|
||||
module_name = plugin_path.stem
|
||||
spec = importlib.util.spec_from_file_location(module_name, plugin_path)
|
||||
if spec is None or spec.loader is None:
|
||||
raise ImportError(f'Could not load plugin from {plugin_path}')
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
sys.modules[module_name] = module
|
||||
spec.loader.exec_module(module)
|
||||
else:
|
||||
# Import by dotted module name
|
||||
module = importlib.import_module(name)
|
||||
module = importlib.import_module(plugin)
|
||||
self._pm.register(module)
|
||||
|
||||
# =========================================================================
|
||||
@@ -172,19 +178,27 @@ class OcrmypdfPluginManager:
|
||||
|
||||
def filter_pdf_page(
|
||||
self, *, page: PageContext, image_filename: Path, output_pdf: Path
|
||||
) -> Path | None:
|
||||
) -> Path:
|
||||
"""Convert a filtered whole page image into a PDF."""
|
||||
return self._pm.hook.filter_pdf_page(
|
||||
result = self._pm.hook.filter_pdf_page(
|
||||
page=page, image_filename=image_filename, output_pdf=output_pdf
|
||||
)
|
||||
if result is None:
|
||||
raise ValueError('No PDF produced')
|
||||
if result != output_pdf:
|
||||
raise ValueError('filter_pdf_page must return output_pdf')
|
||||
return result
|
||||
|
||||
def get_ocr_engine(self, *, options: OcrOptions | None = None) -> OcrEngine | None:
|
||||
def get_ocr_engine(self, *, options: OcrOptions | None = None) -> OcrEngine:
|
||||
"""Returns an OcrEngine to use for processing.
|
||||
|
||||
Args:
|
||||
options: OcrOptions to pass to the hook for engine selection.
|
||||
"""
|
||||
return self._pm.hook.get_ocr_engine(options=options)
|
||||
result = self._pm.hook.get_ocr_engine(options=options)
|
||||
if result is None:
|
||||
raise ValueError('No OCR engine selected')
|
||||
return result
|
||||
|
||||
def generate_pdfa(
|
||||
self,
|
||||
@@ -218,15 +232,18 @@ class OcrmypdfPluginManager:
|
||||
context: PdfContext,
|
||||
executor: Executor,
|
||||
linearize: bool,
|
||||
) -> tuple[Path, Sequence[str]] | None:
|
||||
) -> tuple[Path, Sequence[str]]:
|
||||
"""Optimize a PDF after OCR processing."""
|
||||
return self._pm.hook.optimize_pdf(
|
||||
result = self._pm.hook.optimize_pdf(
|
||||
input_pdf=input_pdf,
|
||||
output_pdf=output_pdf,
|
||||
context=context,
|
||||
executor=executor,
|
||||
linearize=linearize,
|
||||
)
|
||||
if result is None:
|
||||
return input_pdf, []
|
||||
return result
|
||||
|
||||
def is_optimization_enabled(self, *, context: PdfContext) -> bool | None:
|
||||
"""Returns whether optimization is enabled for given context."""
|
||||
|
||||
@@ -21,7 +21,7 @@ class PluginOptionRegistry:
|
||||
compatibility (e.g., options.tesseract_timeout).
|
||||
"""
|
||||
|
||||
def __init__(self):
|
||||
def __init__(self) -> None:
|
||||
self._option_models: dict[str, type[BaseModel]] = {}
|
||||
|
||||
def register_option_model(
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Protect the real standard output from corruption by stray writes.
|
||||
|
||||
When OCRmyPDF writes its final PDF to standard output (``ocrmypdf in.pdf -``),
|
||||
the bytes on stdout must be exactly the PDF and nothing else. Any accidental
|
||||
write to file descriptor 1 anywhere in the process -- from a third-party
|
||||
library, a plugin, or a stray ``print()`` -- would silently corrupt the output.
|
||||
|
||||
This module enforces that guarantee at the operating system level. It saves a
|
||||
private duplicate of the real stdout and points file descriptor 1 at standard
|
||||
error, so that anything that writes to stdout lands harmlessly on stderr. Only
|
||||
OCRmyPDF's final "produce the PDF" step writes to the preserved real stdout, via
|
||||
:func:`get_protected_stdout_fd`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import threading
|
||||
|
||||
_lock = threading.Lock()
|
||||
_saved_fd: int | None = None
|
||||
_active = False
|
||||
|
||||
|
||||
def protect_stdout() -> bool:
|
||||
"""Redirect file descriptor 1 to stderr and preserve the real stdout.
|
||||
|
||||
After this call, any write to file descriptor 1 -- including ``print()`` and
|
||||
writes from third-party C libraries -- is redirected to standard error and
|
||||
cannot corrupt the real standard output. The real stdout is preserved on a
|
||||
private file descriptor available from :func:`get_protected_stdout_fd`.
|
||||
|
||||
This mutates process-global state and affects the whole process. It must be
|
||||
called once, early, before any plugins are loaded or any worker
|
||||
process/thread is started, so that all of them inherit the redirected
|
||||
descriptor.
|
||||
|
||||
Returns:
|
||||
True if protection was installed (or was already active). False if
|
||||
stdout is not backed by a real OS file descriptor -- for example under
|
||||
a test harness that captures stdout -- in which case nothing is changed.
|
||||
"""
|
||||
global _saved_fd, _active
|
||||
with _lock:
|
||||
if _active:
|
||||
return True
|
||||
try:
|
||||
fd1 = sys.stdout.fileno()
|
||||
except (AttributeError, OSError, ValueError):
|
||||
# stdout is not backed by a real file descriptor (e.g. captured by
|
||||
# a test harness or replaced with an in-memory stream).
|
||||
return False
|
||||
try:
|
||||
sys.stdout.flush()
|
||||
saved = os.dup(fd1)
|
||||
os.dup2(2, fd1) # point stdout at stderr
|
||||
except OSError:
|
||||
return False
|
||||
_saved_fd = saved
|
||||
_active = True
|
||||
return True
|
||||
|
||||
|
||||
def get_protected_stdout_fd() -> int | None:
|
||||
"""Return the preserved real stdout file descriptor, or None if inactive."""
|
||||
return _saved_fd if _active else None
|
||||
|
||||
|
||||
def protected_stdout_isatty() -> bool | None:
|
||||
"""Whether the preserved real stdout is a terminal.
|
||||
|
||||
Returns None if protection is not active, in which case the caller should
|
||||
fall back to ``sys.stdout.isatty()``. When protection is active,
|
||||
``sys.stdout`` reports the terminal status of stderr (its descriptor was
|
||||
redirected), so this consults the saved real-stdout descriptor instead.
|
||||
"""
|
||||
if not _active or _saved_fd is None:
|
||||
return None
|
||||
return os.isatty(_saved_fd)
|
||||
+74
-20
@@ -10,15 +10,18 @@ import logging
|
||||
import os
|
||||
import sys
|
||||
from collections.abc import Sequence
|
||||
from collections.abc import Set as AbstractSet
|
||||
from pathlib import Path
|
||||
from shutil import copyfileobj
|
||||
from typing import BinaryIO, cast
|
||||
|
||||
import pikepdf
|
||||
|
||||
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
from ocrmypdf._exec import unpaper
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._options import OcrOptions, ProcessingMode
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager
|
||||
from ocrmypdf._stdoutprotect import protected_stdout_isatty
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
InputFileError,
|
||||
@@ -47,7 +50,7 @@ def check_platform() -> None:
|
||||
|
||||
|
||||
def check_options_languages(
|
||||
options: OcrOptions, ocr_engine_languages: list[str]
|
||||
options: OcrOptions, ocr_engine_languages: AbstractSet[str]
|
||||
) -> None:
|
||||
# Check for blocked languages first, before checking if they're installed
|
||||
DENIED_LANGUAGES = {'equ', 'osd'}
|
||||
@@ -91,7 +94,15 @@ def check_options_sidecar(options: OcrOptions) -> None:
|
||||
raise BadArgsError(
|
||||
"--sidecar filename needed when output file is /dev/null or NUL."
|
||||
)
|
||||
options.sidecar = options.output_file + '.txt'
|
||||
elif not isinstance(options.output_file, str | Path):
|
||||
# The '\0' sentinel is only ever set by the CLI, which always
|
||||
# supplies output_file as a plain path - not a stream. If this
|
||||
# somehow fires, the caller mixed a CLI-only sentinel with the
|
||||
# stream-based API.
|
||||
raise BadArgsError(
|
||||
"--sidecar filename needed when output file is not a path."
|
||||
)
|
||||
options.sidecar = os.fspath(options.output_file) + '.txt'
|
||||
if options.sidecar == options.input_file or options.sidecar == options.output_file:
|
||||
raise BadArgsError(
|
||||
"--sidecar file must be different from the input and output files"
|
||||
@@ -114,19 +125,40 @@ def check_options_preprocessing(options: OcrOptions) -> None:
|
||||
package='unpaper',
|
||||
version_checker=unpaper.version,
|
||||
need_version='6.1',
|
||||
required_for="--clean, --clean-final", # Problem arguments
|
||||
required_for="--clean, --clean-final",
|
||||
)
|
||||
|
||||
|
||||
def check_options_strip(options: OcrOptions) -> None:
|
||||
"""Reject options that cannot apply in strip mode.
|
||||
|
||||
``--mode strip`` removes the OCR text layer in place without rasterizing or
|
||||
running OCR, so image-processing and OCR-output options have no effect.
|
||||
"""
|
||||
if options.mode != ProcessingMode.strip_text:
|
||||
return
|
||||
incompatible = {
|
||||
'--deskew': options.deskew,
|
||||
'--clean': options.clean,
|
||||
'--clean-final': options.clean_final,
|
||||
'--remove-background': options.remove_background,
|
||||
'--rotate-pages': options.rotate_pages,
|
||||
'--oversample': options.oversample,
|
||||
'--remove-vectors': options.remove_vectors,
|
||||
'--sidecar': options.sidecar,
|
||||
}
|
||||
used = sorted(name for name, value in incompatible.items() if value)
|
||||
if used:
|
||||
raise BadArgsError(
|
||||
"--mode strip removes the OCR text layer without rasterizing or "
|
||||
"running OCR, so these options have no effect and are not allowed: "
|
||||
f"{', '.join(used)}"
|
||||
)
|
||||
try:
|
||||
if options.unpaper_args:
|
||||
options.unpaper_args = unpaper.validate_custom_args(
|
||||
options.unpaper_args
|
||||
)
|
||||
except Exception as e:
|
||||
raise BadArgsError("--unpaper-args: " + str(e)) from e
|
||||
|
||||
|
||||
def _check_plugin_invariant_options(options: OcrOptions) -> None:
|
||||
check_platform()
|
||||
check_options_strip(options)
|
||||
check_options_sidecar(options)
|
||||
check_options_preprocessing(options)
|
||||
|
||||
@@ -168,28 +200,32 @@ def create_input_file(options: OcrOptions, work_folder: Path) -> tuple[Path, str
|
||||
# stdin
|
||||
log.info('reading file from standard input')
|
||||
target = work_folder / 'stdin'
|
||||
with open(target, 'wb') as stream_buffer:
|
||||
with target.open('wb') as stream_buffer:
|
||||
copyfileobj(sys.stdin.buffer, stream_buffer)
|
||||
return target, "stdin"
|
||||
elif hasattr(options.input_file, 'readable'):
|
||||
if not options.input_file.readable():
|
||||
input_stream = cast(BinaryIO, options.input_file)
|
||||
if not input_stream.readable():
|
||||
raise InputFileError("Input file stream is not readable")
|
||||
log.info('reading file from input stream')
|
||||
target = work_folder / 'stream'
|
||||
with open(target, 'wb') as stream_buffer:
|
||||
copyfileobj(options.input_file, stream_buffer)
|
||||
with target.open('wb') as stream_buffer:
|
||||
copyfileobj(input_stream, stream_buffer)
|
||||
return target, "stream"
|
||||
else:
|
||||
# The branches above already ruled out the stdin sentinel and
|
||||
# stream-like objects, so this must be a filesystem path.
|
||||
assert isinstance(options.input_file, str | bytes | os.PathLike)
|
||||
try:
|
||||
target = work_folder / 'origin'
|
||||
safe_symlink(options.input_file, target)
|
||||
return target, os.fspath(options.input_file)
|
||||
return target, os.fsdecode(options.input_file)
|
||||
except FileNotFoundError as e:
|
||||
msg = f"File not found - {options.input_file}"
|
||||
msg = f"File not found - {os.fsdecode(options.input_file)}"
|
||||
if running_in_docker(): # pragma: no cover
|
||||
msg += (
|
||||
"\nDocker cannot access your working directory unless you "
|
||||
"explicitly share it with the Docker container and set up"
|
||||
"explicitly share it with the Docker container and set up "
|
||||
"permissions correctly.\n"
|
||||
"You may find it easier to use stdin/stdout:"
|
||||
"\n"
|
||||
@@ -210,7 +246,13 @@ def create_input_file(options: OcrOptions, work_folder: Path) -> tuple[Path, str
|
||||
|
||||
def check_requested_output_file(options: OcrOptions) -> None:
|
||||
if options.output_file == '-':
|
||||
if sys.stdout.isatty():
|
||||
# When stdout protection is active, fd 1 has been redirected to stderr,
|
||||
# so sys.stdout.isatty() would report stderr's status. Consult the
|
||||
# preserved real stdout instead, falling back when protection is off.
|
||||
is_tty = protected_stdout_isatty()
|
||||
if is_tty is None:
|
||||
is_tty = sys.stdout.isatty()
|
||||
if is_tty:
|
||||
raise BadArgsError(
|
||||
"Output was set to stdout '-' but it looks like stdout "
|
||||
"is connected to a terminal. Please redirect stdout to a "
|
||||
@@ -221,7 +263,19 @@ def check_requested_output_file(options: OcrOptions) -> None:
|
||||
raise OutputFileAccessError("Output stream is not writable")
|
||||
elif not is_file_writable(options.output_file):
|
||||
raise OutputFileAccessError(
|
||||
f"Output file location ({options.output_file}) is not a writable file."
|
||||
f"Output file location ({os.fsdecode(options.output_file)}) is not a "
|
||||
"writable file."
|
||||
)
|
||||
|
||||
if (
|
||||
options.no_overwrite
|
||||
and not hasattr(options.output_file, 'writable')
|
||||
and options.output_file != '-'
|
||||
and Path(str(options.output_file)).exists()
|
||||
):
|
||||
raise OutputFileAccessError(
|
||||
f"Output file already exists: {os.fsdecode(options.output_file)}\n"
|
||||
"To overwrite it, omit the --no-overwrite / -n option."
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -10,9 +10,8 @@ import os
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
import pluggy
|
||||
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -20,9 +19,8 @@ log = logging.getLogger(__name__)
|
||||
class ValidationCoordinator:
|
||||
"""Coordinates validation across plugin models and core options."""
|
||||
|
||||
def __init__(self, plugin_manager: pluggy.PluginManager):
|
||||
def __init__(self, plugin_manager: OcrmypdfPluginManager):
|
||||
self.plugin_manager = plugin_manager
|
||||
self.registry = getattr(plugin_manager, '_option_registry', None)
|
||||
|
||||
def validate_all_options(self, options: OcrOptions) -> None:
|
||||
"""Run comprehensive validation on all options.
|
||||
@@ -110,13 +108,18 @@ class ValidationCoordinator:
|
||||
)
|
||||
|
||||
# Validate output type compatibility
|
||||
if options.output_type == 'none' and str(options.output_file) not in (
|
||||
output_file_display = (
|
||||
os.fsdecode(options.output_file)
|
||||
if isinstance(options.output_file, bytes)
|
||||
else str(options.output_file)
|
||||
)
|
||||
if options.output_type == 'none' and output_file_display not in (
|
||||
os.devnull,
|
||||
'-',
|
||||
):
|
||||
raise ValueError(
|
||||
"Since you specified `--output-type none`, the output file "
|
||||
f"{options.output_file} cannot be produced. Set the output file to "
|
||||
f"{output_file_display} cannot be produced. Set the output file to "
|
||||
"`-` to suppress this message."
|
||||
)
|
||||
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
__version__ = "17.9.0"
|
||||
+100
-5
@@ -50,12 +50,15 @@ from pathlib import Path
|
||||
from typing import BinaryIO, overload
|
||||
from warnings import warn
|
||||
|
||||
from pydantic import BaseModel
|
||||
|
||||
from ocrmypdf._logging import PageNumberFilter
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._pipelines.hocr_to_ocr_pdf import run_hocr_to_ocr_pdf_pipeline
|
||||
from ocrmypdf._pipelines.ocr import run_pipeline, run_pipeline_cli
|
||||
from ocrmypdf._pipelines.pdf_to_hocr import run_hocr_pipeline
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager, get_plugin_manager
|
||||
from ocrmypdf._stdoutprotect import protect_stdout
|
||||
from ocrmypdf._validation import check_options
|
||||
from ocrmypdf.cli import ArgumentParser, get_parser
|
||||
from ocrmypdf.exceptions import ExitCode
|
||||
@@ -105,7 +108,7 @@ def setup_plugin_infrastructure(
|
||||
plugin_manager = get_plugin_manager(plugins)
|
||||
|
||||
# Initialize plugins (pass the underlying pluggy manager)
|
||||
plugin_manager.initialize(plugin_manager=plugin_manager.pluggy)
|
||||
plugin_manager.initialize(plugin_manager=plugin_manager.pluggy_manager)
|
||||
|
||||
# Initialize plugin option registry
|
||||
from ocrmypdf._plugin_registry import PluginOptionRegistry
|
||||
@@ -114,7 +117,7 @@ def setup_plugin_infrastructure(
|
||||
|
||||
# Let plugins register their option models
|
||||
option_models = plugin_manager.register_options()
|
||||
all_plugin_models: dict[str, type] = {}
|
||||
all_plugin_models: dict[str, type[BaseModel]] = {}
|
||||
for plugin_options in option_models:
|
||||
if plugin_options: # Skip None returns
|
||||
for namespace, model_class in plugin_options.items():
|
||||
@@ -233,6 +236,37 @@ def configure_logging(
|
||||
return log
|
||||
|
||||
|
||||
def configure_stdout_protection() -> bool:
|
||||
"""Protect the process's real standard output from corruption.
|
||||
|
||||
When OCRmyPDF writes its final PDF to standard output (``output_file='-'``),
|
||||
the bytes on stdout must be exactly the PDF and nothing else. By default
|
||||
OCRmyPDF relies on no in-process code -- third party libraries, plugins, or
|
||||
stray ``print()`` calls -- ever writing to stdout. This function makes that
|
||||
guarantee real: it redirects file descriptor 1 to standard error and
|
||||
preserves a private copy of the real stdout, so that any accidental write to
|
||||
stdout lands harmlessly on stderr while OCRmyPDF still emits its final PDF to
|
||||
the preserved descriptor.
|
||||
|
||||
This is the same protection the ``ocrmypdf`` command line program installs.
|
||||
It is optional for API users and works like :func:`configure_logging`: call
|
||||
it before :func:`ocr` if you want command-line-like behavior. It must be
|
||||
called once, early -- before any plugins are loaded or any worker
|
||||
process/thread is started -- so that they inherit the redirected descriptor.
|
||||
|
||||
Because it mutates process-global file descriptors and affects the entire
|
||||
process, applications that manage their own standard output (for example,
|
||||
a long-lived service that calls :func:`ocr` in-process) should **not** call
|
||||
this function.
|
||||
|
||||
Returns:
|
||||
True if protection was installed (or was already active). False if
|
||||
stdout is not backed by a real operating system file descriptor, in
|
||||
which case nothing is changed.
|
||||
"""
|
||||
return protect_stdout()
|
||||
|
||||
|
||||
def _check_no_conflicting_ocr_params(
|
||||
locals_dict: dict,
|
||||
kwargs: dict,
|
||||
@@ -283,6 +317,51 @@ def _check_no_conflicting_ocr_params(
|
||||
)
|
||||
|
||||
|
||||
def _remap_language_to_languages(options_kwargs: dict) -> None:
|
||||
"""Map the public API 'language' parameter to OcrOptions 'languages' field.
|
||||
|
||||
The public API uses 'language' (matching CLI --language) but OcrOptions
|
||||
uses 'languages' (plural). This also coerces a bare string to a list
|
||||
and splits '+'-separated language codes (e.g. 'eng+deu' -> ['eng', 'deu'])
|
||||
to match the CLI behavior.
|
||||
"""
|
||||
if 'language' in options_kwargs and 'languages' not in options_kwargs:
|
||||
lang = options_kwargs.pop('language')
|
||||
if lang is None:
|
||||
return
|
||||
if isinstance(lang, str):
|
||||
lang = lang.split('+')
|
||||
else:
|
||||
# Flatten any '+'-separated entries in the list
|
||||
expanded: list[str] = []
|
||||
for item in lang:
|
||||
if isinstance(item, str) and '+' in item:
|
||||
expanded.extend(item.split('+'))
|
||||
else:
|
||||
expanded.append(item)
|
||||
lang = expanded
|
||||
options_kwargs['languages'] = lang
|
||||
elif 'language' in options_kwargs:
|
||||
del options_kwargs['language']
|
||||
|
||||
|
||||
def _remap_jpg_quality_to_jpeg_quality(options_kwargs: dict) -> None:
|
||||
"""Map the deprecated 'jpg_quality' parameter to 'jpeg_quality'.
|
||||
|
||||
'jpg_quality' was the original API parameter name. 'jpeg_quality' is the
|
||||
canonical OcrOptions field, matching the primary --jpeg-quality CLI flag.
|
||||
Prefer an explicitly-given 'jpeg_quality' if both are set.
|
||||
"""
|
||||
if 'jpg_quality' not in options_kwargs:
|
||||
return
|
||||
old_value = options_kwargs.pop('jpg_quality')
|
||||
if old_value is None:
|
||||
return
|
||||
warn("ocrmypdf.ocr(jpg_quality=...) is deprecated, use jpeg_quality= instead.")
|
||||
if options_kwargs.get('jpeg_quality') is None:
|
||||
options_kwargs['jpeg_quality'] = old_value
|
||||
|
||||
|
||||
def create_options(
|
||||
*, input_file: PathOrIO, output_file: PathOrIO, parser: ArgumentParser, **kwargs
|
||||
) -> OcrOptions:
|
||||
@@ -304,6 +383,12 @@ def create_options(
|
||||
# Prepare kwargs for direct OcrOptions construction
|
||||
options_kwargs = kwargs.copy()
|
||||
|
||||
# Map API parameter 'language' to OcrOptions field 'languages'
|
||||
_remap_language_to_languages(options_kwargs)
|
||||
|
||||
# Map deprecated 'jpg_quality' parameter to 'jpeg_quality'
|
||||
_remap_jpg_quality_to_jpeg_quality(options_kwargs)
|
||||
|
||||
# Set input and output files
|
||||
options_kwargs['input_file'] = input_file
|
||||
options_kwargs['output_file'] = output_file
|
||||
@@ -383,7 +468,8 @@ def ocr(
|
||||
redo_ocr: bool | None = None,
|
||||
skip_big: float | None = None,
|
||||
optimize: int | None = None,
|
||||
jpg_quality: int | None = None,
|
||||
jpeg_quality: int | None = None,
|
||||
jpg_quality: int | None = None, # Deprecated, use jpeg_quality instead
|
||||
png_quality: int | None = None,
|
||||
jbig2_lossy: bool | None = None,
|
||||
jbig2_page_group_size: int | None = None,
|
||||
@@ -408,6 +494,8 @@ def ocr(
|
||||
fast_web_view: float | None = None,
|
||||
continue_on_soft_render_error: bool | None = None,
|
||||
invalidate_digital_signatures: bool | None = None,
|
||||
tagged_pdf_mode: str | None = None,
|
||||
no_overwrite: bool | None = None,
|
||||
plugins: Iterable[Path | str] | None = None,
|
||||
plugin_manager: OcrmypdfPluginManager | None = None,
|
||||
keep_temporary_files: bool | None = None,
|
||||
@@ -444,7 +532,8 @@ def ocr( # noqa: D417
|
||||
redo_ocr: bool | None = None, # Legacy, use mode='redo' instead
|
||||
skip_big: float | None = None,
|
||||
optimize: int | None = None,
|
||||
jpg_quality: int | None = None,
|
||||
jpeg_quality: int | None = None,
|
||||
jpg_quality: int | None = None, # Deprecated, use jpeg_quality instead
|
||||
png_quality: int | None = None,
|
||||
jbig2_lossy: bool | None = None, # Deprecated, ignored
|
||||
jbig2_page_group_size: int | None = None, # Deprecated, ignored
|
||||
@@ -469,6 +558,8 @@ def ocr( # noqa: D417
|
||||
fast_web_view: float | None = None,
|
||||
continue_on_soft_render_error: bool | None = None,
|
||||
invalidate_digital_signatures: bool | None = None,
|
||||
tagged_pdf_mode: str | None = None,
|
||||
no_overwrite: bool | None = None,
|
||||
plugins: Iterable[Path | str] | None = None,
|
||||
plugin_manager: OcrmypdfPluginManager | None = None,
|
||||
keep_temporary_files: bool | None = None,
|
||||
@@ -760,6 +851,9 @@ def _pdf_to_hocr( # noqa: D417
|
||||
):
|
||||
options_kwargs[param_name] = param_value
|
||||
|
||||
# Map API parameter 'language' to OcrOptions field 'languages'
|
||||
_remap_language_to_languages(options_kwargs)
|
||||
|
||||
# Handle plugins
|
||||
if plugins:
|
||||
options_kwargs['plugins'] = plugins
|
||||
@@ -811,7 +905,7 @@ def _hocr_to_ocr_pdf( # noqa: D417
|
||||
jobs: int | None = None,
|
||||
use_threads: bool | None = None,
|
||||
optimize: int | None = None,
|
||||
jpg_quality: int | None = None,
|
||||
jpeg_quality: int | None = None,
|
||||
png_quality: int | None = None,
|
||||
jbig2_lossy: bool | None = None, # Deprecated, ignored
|
||||
jbig2_page_group_size: int | None = None, # Deprecated, ignored
|
||||
@@ -927,6 +1021,7 @@ __all__ = [
|
||||
'Verbosity',
|
||||
'check_options',
|
||||
'configure_logging',
|
||||
'configure_stdout_protection',
|
||||
'create_options',
|
||||
'get_parser',
|
||||
'get_plugin_manager',
|
||||
|
||||
@@ -27,9 +27,12 @@ from ocrmypdf.exceptions import InputFileError
|
||||
from ocrmypdf.helpers import remove_all_log_handlers
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from logging import LogRecord
|
||||
from typing import TypeAlias
|
||||
|
||||
Queue: TypeAlias = multiprocessing.queues.Queue | queue.Queue
|
||||
Queue: TypeAlias = (
|
||||
multiprocessing.queues.Queue[LogRecord | None] | queue.Queue[LogRecord | None]
|
||||
)
|
||||
UserInit: TypeAlias = Callable[[], None]
|
||||
WorkerInit: TypeAlias = Callable[[Queue, UserInit, int], None]
|
||||
|
||||
@@ -99,7 +102,9 @@ def thread_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
||||
return
|
||||
|
||||
|
||||
def setup_executor(use_threads: bool) -> tuple[Queue, Executor, WorkerInit]:
|
||||
def setup_executor(
|
||||
use_threads: bool,
|
||||
) -> tuple[Queue, FuturesExecutorClass, WorkerInit]:
|
||||
if not use_threads:
|
||||
# Some execution environments like AWS Lambda and Termux do not support
|
||||
# semaphores. Check if semaphore support is available, and if not, fall back
|
||||
@@ -112,6 +117,8 @@ def setup_executor(use_threads: bool) -> tuple[Queue, Executor, WorkerInit]:
|
||||
except ImportError:
|
||||
use_threads = True
|
||||
|
||||
loq_queue: Queue
|
||||
executor_class: FuturesExecutorClass
|
||||
if use_threads:
|
||||
loq_queue = queue.Queue(-1)
|
||||
executor_class = ThreadPoolExecutor
|
||||
|
||||
@@ -44,6 +44,30 @@ class PdfaImageCompression(StrEnum):
|
||||
LOSSLESS = 'lossless'
|
||||
|
||||
|
||||
def _resolve_auto_compression(
|
||||
compression: PdfaImageCompression, optimize_level: int
|
||||
) -> PdfaImageCompression:
|
||||
"""Resolve 'auto' image compression based on the optimization level.
|
||||
|
||||
At ``-O0`` (no optimization) ``auto`` maps to ``lossless`` so Ghostscript
|
||||
will not transcode lossless images to JPEG during PDF/A generation. At all
|
||||
other levels ``auto`` defers to Ghostscript's heuristic, which may
|
||||
recompress images lossily.
|
||||
|
||||
``-O1`` is a historical exception: although it is otherwise a
|
||||
lossless-only optimization level, coercing ``auto`` to ``lossless`` there
|
||||
can bloat output substantially (Ghostscript's heuristic often picks JPEG
|
||||
for photographic content), so the default is left alone for backwards
|
||||
compatibility. Users who want guaranteed lossless image handling at any
|
||||
level can pass ``--pdfa-image-compression=lossless`` explicitly.
|
||||
|
||||
Explicit ``jpeg`` and ``lossless`` choices are always respected.
|
||||
"""
|
||||
if compression == PdfaImageCompression.AUTO and optimize_level == 0:
|
||||
return PdfaImageCompression.LOSSLESS
|
||||
return compression
|
||||
|
||||
|
||||
class GhostscriptOptions(BaseModel):
|
||||
"""Options specific to Ghostscript operations."""
|
||||
|
||||
@@ -54,6 +78,27 @@ class GhostscriptOptions(BaseModel):
|
||||
pdfa_image_compression: Annotated[
|
||||
PdfaImageCompression, Field(description="PDF/A image compression method")
|
||||
] = PdfaImageCompression.AUTO
|
||||
jpeg_quality: Annotated[
|
||||
int | None,
|
||||
Field(
|
||||
ge=0,
|
||||
le=100,
|
||||
description=(
|
||||
"JPEG quality (0-100) for Ghostscript image recompression during "
|
||||
"PDF/A generation; None uses Ghostscript's default."
|
||||
),
|
||||
),
|
||||
] = None
|
||||
jpeg_maxdpi: Annotated[
|
||||
int | None,
|
||||
Field(
|
||||
ge=1,
|
||||
description=(
|
||||
"Maximum DPI for Ghostscript image downsampling during PDF/A "
|
||||
"generation."
|
||||
),
|
||||
),
|
||||
] = None
|
||||
|
||||
@classmethod
|
||||
def add_arguments_to_parser(cls, parser, namespace: str = 'ghostscript'):
|
||||
@@ -78,14 +123,48 @@ class GhostscriptOptions(BaseModel):
|
||||
choices=[pc.value for pc in PdfaImageCompression],
|
||||
default=PdfaImageCompression.AUTO.value,
|
||||
help="Specify how to compress images in the output PDF/A. 'auto' lets "
|
||||
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
|
||||
"OCRmyPDF decide: at -O0 it uses lossless image compression so "
|
||||
"Ghostscript does not transcode lossless images to JPEG; at -O1 and "
|
||||
"above it defers to Ghostscript's heuristic, which may recompress "
|
||||
"images lossily. 'jpeg' changes all grayscale and color images to "
|
||||
"JPEG compression. 'lossless' uses PNG-style lossless compression "
|
||||
"for all images. Monochrome images are always compressed using a "
|
||||
"for non-JPEG images and passes existing JPEGs through unchanged "
|
||||
"(re-encoding them losslessly would only inflate them). Monochrome "
|
||||
"images are always compressed using a "
|
||||
"lossless codec. Compression settings "
|
||||
"are applied to all pages, including those for which OCR was "
|
||||
"skipped. Not supported for --output-type=pdf ; that setting "
|
||||
"preserves the original compression of all images.",
|
||||
)
|
||||
gs.add_argument(
|
||||
'--ghostscript-jpeg-quality',
|
||||
type=int,
|
||||
metavar='Q',
|
||||
default=None,
|
||||
dest=f'{namespace}_jpeg_quality',
|
||||
help=(
|
||||
"Advanced: Set Ghostscript's -dJPEGQ for images that Ghostscript "
|
||||
"transcodes to JPEG during PDF/A generation. 0 is maximum "
|
||||
"compression; 100 is best quality. If omitted, Ghostscript's "
|
||||
"default is used. This only affects images Ghostscript chooses "
|
||||
"to recompress; for general JPEG quality tuning prefer "
|
||||
"--jpeg-quality, which is applied by the OCRmyPDF optimizer."
|
||||
),
|
||||
)
|
||||
gs.add_argument(
|
||||
'--ghostscript-jpeg-maxdpi',
|
||||
type=int,
|
||||
metavar='DPI',
|
||||
default=None,
|
||||
dest=f'{namespace}_jpeg_maxdpi',
|
||||
help=(
|
||||
"Advanced: Force Ghostscript to downsample color, grayscale, "
|
||||
"and monochrome images in PDF/A output to the given maximum DPI. "
|
||||
"Reducing JPEG quality usually gives better results than "
|
||||
"downsampling at the same file size, and can degrade quality "
|
||||
"of high-resolution monochrome masks."
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
@hookimpl
|
||||
@@ -131,10 +210,10 @@ def check_options(options):
|
||||
)
|
||||
if gs_version >= Version('10.6.0'):
|
||||
log.warning(
|
||||
"Ghostscript 10.6.x contains JPEG encoding errors that may corrupt "
|
||||
"images. OCRmyPDF will attempt to mitigate, but this version is "
|
||||
"strongly not recommended. Please upgrade to a newer version. "
|
||||
"As of 2025-12, 10.6.0 is the latest version of Ghostscript."
|
||||
"Ghostscript %s contains JPEG encoding errors that may corrupt "
|
||||
"images. OCRmyPDF will attempt to mitigate, but versions 10.6.0+ "
|
||||
"are strongly not recommended until this is fixed upstream.",
|
||||
gs_version,
|
||||
)
|
||||
if options.output_type == 'pdfa':
|
||||
options.output_type = 'pdfa-2'
|
||||
@@ -177,6 +256,8 @@ def rasterize_pdf_page(
|
||||
# Let pypdfium handle it (it will error in check_options if unavailable)
|
||||
return None
|
||||
|
||||
log.debug("Rasterizing page %d with the Ghostscript rasterizer", pageno)
|
||||
|
||||
ghostscript.rasterize_pdf(
|
||||
input_file,
|
||||
output_file,
|
||||
@@ -347,11 +428,18 @@ def generate_pdfa(
|
||||
if output_type == 'pdfa':
|
||||
output_type = 'pdfa-2'
|
||||
|
||||
compression = _resolve_auto_compression(
|
||||
context.options.ghostscript.pdfa_image_compression,
|
||||
context.options.optimize,
|
||||
)
|
||||
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[pdfmark, *pdf_pages],
|
||||
output_file=output_file,
|
||||
compression=context.options.ghostscript.pdfa_image_compression,
|
||||
compression=compression,
|
||||
color_conversion_strategy=context.options.ghostscript.color_conversion_strategy,
|
||||
jpeg_quality=context.options.ghostscript.jpeg_quality,
|
||||
jpeg_maxdpi=context.options.ghostscript.jpeg_maxdpi,
|
||||
pdf_version=pdf_version,
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=progressbar_class,
|
||||
|
||||
@@ -8,12 +8,15 @@ import logging
|
||||
import threading
|
||||
from contextlib import closing
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Literal
|
||||
|
||||
try:
|
||||
if TYPE_CHECKING:
|
||||
import pypdfium2 as pdfium
|
||||
except ImportError:
|
||||
pdfium = None
|
||||
|
||||
else:
|
||||
try:
|
||||
import pypdfium2 as pdfium
|
||||
except ImportError:
|
||||
pdfium = None
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf import hookimpl
|
||||
@@ -45,36 +48,27 @@ def _open_pdf_document(input_file: Path):
|
||||
return pdfium.PdfDocument(input_file)
|
||||
|
||||
|
||||
def _calculate_mediabox_crop(page) -> tuple[float, float, float, float]:
|
||||
"""Calculate crop values to expand rendering from CropBox to MediaBox.
|
||||
def _expand_cropbox_to_mediabox(page) -> None:
|
||||
"""Set the page's CropBox to its MediaBox so PDFium renders the full page.
|
||||
|
||||
By default pypdfium2 renders to the CropBox. To render the full MediaBox,
|
||||
we need negative crop values to expand the rendering area.
|
||||
|
||||
Returns:
|
||||
Tuple of (left, bottom, right, top) crop values. Negative values
|
||||
expand the rendering area beyond the CropBox to the MediaBox.
|
||||
PDFium renders to the CropBox by default. Negative ``crop`` values to
|
||||
``render()`` are not supported and only pad the output canvas without
|
||||
expanding the rendered area — content outside the CropBox is clipped.
|
||||
The supported approach is to widen the CropBox in memory before rendering.
|
||||
The document is never saved back to disk, so this mutation is local.
|
||||
See https://github.com/ocrmypdf/OCRmyPDF/issues/1685.
|
||||
"""
|
||||
mediabox = page.get_mediabox() # (left, bottom, right, top)
|
||||
cropbox = page.get_cropbox() # (left, bottom, right, top), defaults to mediabox
|
||||
|
||||
# Calculate how much to expand from cropbox to mediabox
|
||||
# Negative values = expand, positive = shrink
|
||||
return (
|
||||
mediabox[0] - cropbox[0], # Expand left
|
||||
mediabox[1] - cropbox[1], # Expand bottom
|
||||
cropbox[2] - mediabox[2], # Expand right
|
||||
cropbox[3] - mediabox[3], # Expand top
|
||||
)
|
||||
page.set_cropbox(*mediabox)
|
||||
|
||||
|
||||
def _render_page_to_bitmap(
|
||||
page,
|
||||
page: pdfium.PdfPage,
|
||||
raster_device: str,
|
||||
raster_dpi: Resolution,
|
||||
rotation: int | None,
|
||||
use_cropbox: bool,
|
||||
):
|
||||
) -> tuple[pdfium.PdfBitmap, int, int]:
|
||||
"""Render a PDF page to a bitmap."""
|
||||
# Round DPI to match Ghostscript's precision
|
||||
raster_dpi = raster_dpi.round(6)
|
||||
@@ -101,16 +95,21 @@ def _render_page_to_bitmap(
|
||||
|
||||
# Render the page to a bitmap
|
||||
# The scale parameter controls the resolution
|
||||
grayscale = raster_device.lower() in ('pnggray', 'jpeggray')
|
||||
# Render in grayscale for mono and gray devices (better input for 1-bit conversion)
|
||||
grayscale = raster_device.lower() in (
|
||||
'pngmono',
|
||||
'pngmonod',
|
||||
'pnggray',
|
||||
'jpeggray',
|
||||
)
|
||||
|
||||
# Calculate crop to render the appropriate box
|
||||
# Default (use_cropbox=False) renders MediaBox for consistency with Ghostscript
|
||||
crop = (0, 0, 0, 0) if use_cropbox else _calculate_mediabox_crop(page)
|
||||
if not use_cropbox:
|
||||
_expand_cropbox_to_mediabox(page)
|
||||
|
||||
bitmap = page.render(
|
||||
scale=scale,
|
||||
rotation=0, # We already set rotation on the page
|
||||
crop=crop,
|
||||
may_draw_forms=True,
|
||||
draw_annots=True,
|
||||
grayscale=grayscale,
|
||||
@@ -121,14 +120,14 @@ def _render_page_to_bitmap(
|
||||
|
||||
|
||||
def _process_image_for_output(
|
||||
pil_image,
|
||||
pil_image: Image.Image,
|
||||
raster_device: str,
|
||||
raster_dpi: Resolution,
|
||||
page_dpi: Resolution | None,
|
||||
stop_on_soft_error: bool,
|
||||
expected_width: int | None = None,
|
||||
expected_height: int | None = None,
|
||||
):
|
||||
) -> tuple[Image.Image, Literal['PNG', 'TIFF', 'JPEG']]:
|
||||
"""Process PIL image for output format and set DPI metadata."""
|
||||
# Correct dimensions if slightly off (within 2 pixels tolerance)
|
||||
if expected_width and expected_height:
|
||||
@@ -146,8 +145,7 @@ def _process_image_for_output(
|
||||
f"{expected_width}x{expected_height}"
|
||||
)
|
||||
pil_image = pil_image.resize(
|
||||
(expected_width, expected_height),
|
||||
Image.Resampling.LANCZOS
|
||||
(expected_width, expected_height), Image.Resampling.LANCZOS
|
||||
)
|
||||
|
||||
# Set the DPI metadata if page_dpi is specified
|
||||
@@ -160,20 +158,52 @@ def _process_image_for_output(
|
||||
dpi_tuple = (float(raster_dpi.x), float(raster_dpi.y))
|
||||
pil_image.info['dpi'] = dpi_tuple
|
||||
|
||||
# Determine output format based on raster_device
|
||||
if raster_device.lower() in ('png', 'pngmono', 'pnggray', 'png16m', 'pngalpha'):
|
||||
format_name = 'PNG'
|
||||
elif raster_device.lower() in ('jpeg', 'jpeggray', 'jpg'):
|
||||
format_name = 'JPEG'
|
||||
# Convert RGBA to RGB for JPEG
|
||||
# Convert image mode to match raster_device
|
||||
# This ensures pypdfium output matches Ghostscript's native device output
|
||||
raster_device_lower = raster_device.lower()
|
||||
|
||||
if raster_device_lower in ('pngmono', 'pngmonod'):
|
||||
# Convert to 1-bit black and white (matches Ghostscript pngmono/pngmonod)
|
||||
if pil_image.mode != '1':
|
||||
if pil_image.mode not in ('L', '1'):
|
||||
pil_image = pil_image.convert('L')
|
||||
pil_image = pil_image.convert('1')
|
||||
elif raster_device_lower in ('pnggray', 'jpeggray'):
|
||||
# Convert to 8-bit grayscale
|
||||
if pil_image.mode not in ('L', '1'):
|
||||
pil_image = pil_image.convert('L')
|
||||
elif raster_device_lower == 'png256':
|
||||
# Convert to 8-bit indexed color (256 colors)
|
||||
if pil_image.mode != 'P':
|
||||
if pil_image.mode not in ('RGB', 'RGBA'):
|
||||
pil_image = pil_image.convert('RGB')
|
||||
pil_image = pil_image.quantize(colors=256)
|
||||
elif raster_device_lower in ('png16m', 'jpeg'):
|
||||
# Convert to RGB
|
||||
if pil_image.mode == 'RGBA':
|
||||
# Create white background
|
||||
background = pil_image.new('RGB', pil_image.size, (255, 255, 255))
|
||||
background.paste(
|
||||
pil_image, mask=pil_image.split()[-1]
|
||||
) # Use alpha channel as mask
|
||||
background = Image.new('RGB', pil_image.size, (255, 255, 255))
|
||||
background.paste(pil_image, mask=pil_image.split()[-1])
|
||||
pil_image = background
|
||||
elif raster_device.lower() in ('tiff', 'tif'):
|
||||
elif pil_image.mode not in ('RGB',):
|
||||
pil_image = pil_image.convert('RGB')
|
||||
# pngalpha: keep RGBA as-is
|
||||
|
||||
# Determine output format based on raster_device
|
||||
png_devices = (
|
||||
'png',
|
||||
'pngmono',
|
||||
'pngmonod',
|
||||
'pnggray',
|
||||
'png256',
|
||||
'png16m',
|
||||
'pngalpha',
|
||||
)
|
||||
format_name: Literal['PNG', 'TIFF', 'JPEG']
|
||||
if raster_device_lower in png_devices:
|
||||
format_name = 'PNG'
|
||||
elif raster_device_lower in ('jpeg', 'jpeggray', 'jpg'):
|
||||
format_name = 'JPEG'
|
||||
elif raster_device_lower in ('tiff', 'tif'):
|
||||
format_name = 'TIFF'
|
||||
else:
|
||||
# Default to PNG for unknown formats
|
||||
@@ -186,7 +216,7 @@ def _process_image_for_output(
|
||||
return pil_image, format_name
|
||||
|
||||
|
||||
def _save_image(pil_image, output_file: Path, format_name: str):
|
||||
def _save_image(pil_image: Image.Image, output_file: Path, format_name: str) -> None:
|
||||
"""Save PIL image to file with appropriate DPI metadata."""
|
||||
save_kwargs = {}
|
||||
if (
|
||||
@@ -226,6 +256,8 @@ def rasterize_pdf_page(
|
||||
if pdfium is None:
|
||||
return None # Fall back to Ghostscript
|
||||
|
||||
log.debug("Rasterizing page %d with the pypdfium2 rasterizer", pageno)
|
||||
|
||||
# Acquire lock to ensure thread-safe access to pypdfium2
|
||||
with (
|
||||
_pdfium_lock,
|
||||
|
||||
@@ -130,7 +130,7 @@ class TesseractOptions(BaseModel):
|
||||
metavar='PSM',
|
||||
choices=range(0, 14),
|
||||
dest=f'{namespace}_pagesegmode',
|
||||
help="Set Tesseract page segmentation mode (see tesseract --help).",
|
||||
help="Set Tesseract page segmentation mode (see tesseract --help-extra).",
|
||||
)
|
||||
|
||||
tess.add_argument(
|
||||
@@ -168,7 +168,7 @@ class TesseractOptions(BaseModel):
|
||||
tess.add_argument(
|
||||
f'--{namespace}-timeout',
|
||||
default=180.0,
|
||||
type=numeric(float, 0),
|
||||
type=numeric(float, 0.0),
|
||||
metavar='SECONDS',
|
||||
dest=f'{namespace}_timeout',
|
||||
help=(
|
||||
@@ -183,7 +183,7 @@ class TesseractOptions(BaseModel):
|
||||
tess.add_argument(
|
||||
f'--{namespace}-non-ocr-timeout',
|
||||
default=180.0,
|
||||
type=numeric(float, 0),
|
||||
type=numeric(float, 0.0),
|
||||
metavar='SECONDS',
|
||||
dest=f'{namespace}_non_ocr_timeout',
|
||||
help=(
|
||||
|
||||
+39
-12
@@ -12,7 +12,7 @@ from typing import Any, TypeVar
|
||||
|
||||
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
from ocrmypdf._defaults import PROGRAM_NAME as _PROGRAM_NAME
|
||||
from ocrmypdf._options import OcrOptions, ProcessingMode
|
||||
from ocrmypdf._options import OcrOptions, ProcessingMode, TaggedPdfMode
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager
|
||||
from ocrmypdf._version import __version__ as _VERSION
|
||||
|
||||
@@ -137,8 +137,9 @@ Online documentation is located at:
|
||||
'output_file',
|
||||
metavar="output_pdf",
|
||||
help="Output searchable PDF file (or '-' to write to standard output). "
|
||||
"Existing files will be overwritten. If same as input file, the "
|
||||
"input file will be updated only if processing is successful.",
|
||||
"Existing files will be overwritten (use --no-overwrite to prevent this). "
|
||||
"If same as input file, the input file will be updated only if "
|
||||
"processing is successful.",
|
||||
)
|
||||
parser.add_argument(
|
||||
'-l',
|
||||
@@ -190,6 +191,15 @@ Online documentation is located at:
|
||||
"may not both use stdout at the same time.",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
'-n',
|
||||
'--no-overwrite',
|
||||
action='store_true',
|
||||
default=False,
|
||||
help="If the output file already exists, exit with an error instead of "
|
||||
"overwriting it.",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
'--version',
|
||||
action='version',
|
||||
@@ -317,7 +327,11 @@ Online documentation is located at:
|
||||
"'default' errors if text is found. "
|
||||
"'force' rasterizes all content and runs OCR (same as --force-ocr). "
|
||||
"'skip' skips pages with existing text (same as --skip-text). "
|
||||
"'redo' re-OCRs pages, replacing old invisible text (same as --redo-ocr).",
|
||||
"'redo' re-OCRs pages, replacing old invisible text (same as --redo-ocr). "
|
||||
"'strip' removes the invisible OCR text layer without rasterizing or "
|
||||
"running OCR, producing a smaller file; only text drawn as invisible "
|
||||
"(render mode 3) is removed, so text from some OCR engines cannot be "
|
||||
"removed this way.",
|
||||
)
|
||||
# Legacy flags for backward compatibility - these set the mode internally
|
||||
ocrsettings.add_argument(
|
||||
@@ -348,7 +362,7 @@ Online documentation is located at:
|
||||
)
|
||||
ocrsettings.add_argument(
|
||||
'--skip-big',
|
||||
type=numeric(float, 0, 5000),
|
||||
type=numeric(float, 0.0, 5000.0),
|
||||
metavar='MPixels',
|
||||
help="Skip OCR on pages larger than the specified amount of megapixels, "
|
||||
"but include skipped pages in final output",
|
||||
@@ -360,6 +374,14 @@ Online documentation is located at:
|
||||
"signature. This option allows OCR to proceed, but the digital signature "
|
||||
"will be invalidated.",
|
||||
)
|
||||
ocrsettings.add_argument(
|
||||
'--tagged-pdf-mode',
|
||||
choices=[mode.value for mode in TaggedPdfMode],
|
||||
default=TaggedPdfMode.default.value,
|
||||
help="Control behavior when a Tagged PDF is encountered. "
|
||||
"'default' errors if --mode is default, otherwise warns. "
|
||||
"'ignore' always warns but continues processing.",
|
||||
)
|
||||
|
||||
advanced = parser.add_argument_group(
|
||||
"Advanced", "Advanced options to control OCRmyPDF"
|
||||
@@ -369,13 +391,14 @@ Online documentation is located at:
|
||||
type=str,
|
||||
help=(
|
||||
"Limit OCR to the specified pages (ranges or comma separated), "
|
||||
"skipping others"
|
||||
"skipping others. The token 'end' is an alias for the last page, "
|
||||
"so e.g. '3-end' OCRs from page 3 to the last page."
|
||||
),
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--max-image-mpixels',
|
||||
action='store',
|
||||
type=numeric(float, 0),
|
||||
type=numeric(float, 0.0),
|
||||
metavar='MPixels',
|
||||
help="Set maximum number of megapixels to unpack before treating an image as a "
|
||||
"decompression bomb",
|
||||
@@ -404,21 +427,25 @@ Online documentation is located at:
|
||||
'--rasterizer',
|
||||
choices=['auto', 'ghostscript', 'pypdfium'],
|
||||
default='auto',
|
||||
help="Choose PDF page rasterizer. 'auto' prefers pypdfium when available, "
|
||||
"falling back to Ghostscript. 'pypdfium' is faster but requires the "
|
||||
"pypdfium2 package. 'ghostscript' uses the traditional Ghostscript rasterizer.",
|
||||
help="Choose PDF page rasterizer. 'auto' (the default) prefers pypdfium2 "
|
||||
"when the pypdfium2 package is installed, falling back to Ghostscript "
|
||||
"otherwise. pypdfium2 anti-aliases page content and generally produces "
|
||||
"better input for OCR than Ghostscript 10.x, which can render aliased "
|
||||
"glyphs that OCR misreads as extra word breaks. 'pypdfium' forces the "
|
||||
"pypdfium2 rasterizer (requires the pypdfium2 package); 'ghostscript' "
|
||||
"forces the traditional Ghostscript rasterizer.",
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--rotate-pages-threshold',
|
||||
default=DEFAULT_ROTATE_PAGES_THRESHOLD,
|
||||
type=numeric(float, 0, 1000),
|
||||
type=numeric(float, 0.0, 1000.0),
|
||||
metavar='CONFIDENCE',
|
||||
help="Only rotate pages when confidence is above this value (arbitrary "
|
||||
"units reported by tesseract)",
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--fast-web-view',
|
||||
type=numeric(float, 0),
|
||||
type=numeric(float, 0.0),
|
||||
default=1.0,
|
||||
metavar="MEGABYTES",
|
||||
help="If the size of file is more than this threshold (in MB), then "
|
||||
|
||||
+70
-10
@@ -139,14 +139,74 @@ class TaggedPDFError(InputFileError):
|
||||
)
|
||||
|
||||
|
||||
class ColorConversionNeededError(BadArgsError):
|
||||
"""PDF needs color conversion."""
|
||||
class NonEmbeddedFontsError(InputFileError):
|
||||
"""Input has non-embedded CID fonts that PDF/A conversion would corrupt.
|
||||
|
||||
message = dedent(
|
||||
"""\
|
||||
The input PDF has an unusual color space. Use
|
||||
--color-conversion-strategy to convert to a common color space
|
||||
such as RGB, or use --output-type pdf to skip PDF/A conversion
|
||||
and retain the original color space.
|
||||
"""
|
||||
)
|
||||
PDF/A requires all fonts to be embedded. Ghostscript substitutes and embeds
|
||||
a replacement for non-embedded CID (CJK) fonts, which corrupts the
|
||||
character-to-Unicode mapping and silently destroys an existing text layer
|
||||
(commonly an Adobe Acrobat CJK OCR layer). OCRmyPDF refuses to produce such
|
||||
output rather than damage the user's data
|
||||
(see https://github.com/ocrmypdf/OCRmyPDF/issues/1561).
|
||||
"""
|
||||
|
||||
def __init__(self, fonts: set[str]):
|
||||
"""Build guidance naming the offending fonts."""
|
||||
super().__init__()
|
||||
font_list = ', '.join(sorted(fonts))
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
The input PDF contains non-embedded CID (character ID) fonts: {font_list}.
|
||||
|
||||
PDF/A requires all fonts to be embedded. Converting to PDF/A would
|
||||
make Ghostscript substitute and embed replacement fonts, which
|
||||
corrupts CID (e.g. CJK/Chinese-Japanese-Korean) text and silently
|
||||
destroys an existing text layer such as one produced by Adobe Acrobat.
|
||||
|
||||
Use --output-type pdf to keep the existing text layer intact without
|
||||
PDF/A conversion, or --force-ocr to discard the existing layer and
|
||||
rebuild it with embedded fonts.
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
class ColorConversionNeededError(BadArgsError):
|
||||
"""PDF needs color conversion to a standard color space.
|
||||
|
||||
Ghostscript reported a DeviceN colorspace with an inappropriate alternate.
|
||||
The resulting PDF/A is liable to render incorrectly (often blank) in some
|
||||
viewers such as Adobe Reader, so the colorspace must be normalized to a
|
||||
common one. RGB, CMYK, and Gray are known to work; LeaveColorUnchanged
|
||||
performs no conversion and UseDeviceIndependentColor does not resolve the
|
||||
problem (see https://github.com/ocrmypdf/OCRmyPDF/issues/1187).
|
||||
"""
|
||||
|
||||
# Strategies that can normalize an unusual DeviceN colorspace into one that
|
||||
# PDF/A viewers render correctly.
|
||||
_effective_strategies = "RGB, CMYK, or Gray"
|
||||
|
||||
def __init__(self, color_conversion_strategy: str = "LeaveColorUnchanged"):
|
||||
"""Build guidance tailored to the conversion strategy that was used."""
|
||||
super().__init__()
|
||||
if color_conversion_strategy == "LeaveColorUnchanged":
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
The input PDF has an unusual DeviceN color space that cannot be
|
||||
represented in PDF/A; the output may appear blank in some viewers
|
||||
such as Adobe Reader. Convert it to a common color space with
|
||||
--color-conversion-strategy ({self._effective_strategies}), or use
|
||||
--output-type pdf to skip PDF/A conversion and retain the original
|
||||
color space.
|
||||
"""
|
||||
)
|
||||
else:
|
||||
self.message = dedent(
|
||||
f"""\
|
||||
Color conversion with --color-conversion-strategy
|
||||
{color_conversion_strategy} did not resolve the input PDF's unusual
|
||||
DeviceN color space; the output may appear blank in some viewers
|
||||
such as Adobe Reader. Try a different --color-conversion-strategy
|
||||
({self._effective_strategies}), or use --output-type pdf to skip
|
||||
PDF/A conversion and retain the original color space.
|
||||
"""
|
||||
)
|
||||
|
||||
@@ -10,6 +10,7 @@ This module provides font infrastructure for the fpdf2 PDF renderer. It includes
|
||||
- MultiFontManager: Automatic font selection for multilingual documents
|
||||
- SystemFontProvider: System font discovery
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ocrmypdf.font.font_manager import FontManager
|
||||
@@ -17,6 +18,7 @@ from ocrmypdf.font.font_provider import (
|
||||
BuiltinFontProvider,
|
||||
ChainedFontProvider,
|
||||
FontProvider,
|
||||
GlyphSearchingFontProvider,
|
||||
)
|
||||
from ocrmypdf.font.multi_font_manager import MultiFontManager
|
||||
from ocrmypdf.font.system_font_provider import SystemFontProvider
|
||||
@@ -24,6 +26,7 @@ from ocrmypdf.font.system_font_provider import SystemFontProvider
|
||||
__all__ = [
|
||||
"FontManager",
|
||||
"FontProvider",
|
||||
"GlyphSearchingFontProvider",
|
||||
"BuiltinFontProvider",
|
||||
"ChainedFontProvider",
|
||||
"MultiFontManager",
|
||||
|
||||
@@ -7,7 +7,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Protocol
|
||||
from typing import Protocol, runtime_checkable
|
||||
|
||||
from ocrmypdf.font.font_manager import FontManager
|
||||
|
||||
@@ -52,6 +52,34 @@ class FontProvider(Protocol):
|
||||
...
|
||||
|
||||
|
||||
@runtime_checkable
|
||||
class GlyphSearchingFontProvider(Protocol):
|
||||
"""Optional capability: find a font by glyph coverage rather than by name.
|
||||
|
||||
A provider only knows a limited set of logical font names, but it may have
|
||||
access to many more fonts than it can name (e.g. the ~100 script-specific
|
||||
Noto faces macOS installs). Implementing this lets MultiFontManager use
|
||||
them as a last resort instead of falling back to glyphless rendering.
|
||||
|
||||
Providers that do not implement this are used as-is; the capability is
|
||||
detected at runtime with ``isinstance``.
|
||||
"""
|
||||
|
||||
def find_font_with_glyphs(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Find a font that has glyphs for every character in text.
|
||||
|
||||
The returned name must subsequently resolve through ``get_font()``, so
|
||||
that callers can cache the selection by name.
|
||||
|
||||
Args:
|
||||
text: Text the font must fully cover
|
||||
|
||||
Returns:
|
||||
(logical font name, FontManager), or None if no font covers text
|
||||
"""
|
||||
...
|
||||
|
||||
|
||||
class BuiltinFontProvider:
|
||||
"""Font provider using builtin fonts from ocrmypdf/data directory."""
|
||||
|
||||
@@ -119,6 +147,18 @@ class BuiltinFontProvider:
|
||||
"""Get the glyphless fallback font."""
|
||||
return self._fonts['Occulta']
|
||||
|
||||
def find_font_with_glyphs(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Find a bundled font that covers text, ignoring glyphless Occulta."""
|
||||
if not text:
|
||||
return None
|
||||
codepoints = {ord(c) for c in text}
|
||||
for name, font in self._fonts.items():
|
||||
if name == 'Occulta':
|
||||
continue
|
||||
if all(font.has_glyph(cp) for cp in codepoints):
|
||||
return name, font
|
||||
return None
|
||||
|
||||
|
||||
class ChainedFontProvider:
|
||||
"""Font provider that tries multiple providers in order.
|
||||
@@ -170,6 +210,25 @@ class ChainedFontProvider:
|
||||
result.append(name)
|
||||
return result
|
||||
|
||||
def find_font_with_glyphs(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Ask each capable provider in turn for a font that covers text.
|
||||
|
||||
Providers that don't implement the search are skipped.
|
||||
|
||||
Args:
|
||||
text: Text the font must fully cover
|
||||
|
||||
Returns:
|
||||
(logical font name, FontManager) from the first provider with a
|
||||
match, or None if no provider found one
|
||||
"""
|
||||
for provider in self.providers:
|
||||
if not isinstance(provider, GlyphSearchingFontProvider):
|
||||
continue
|
||||
if found := provider.find_font_with_glyphs(text):
|
||||
return found
|
||||
return None
|
||||
|
||||
def get_fallback_font(self) -> FontManager:
|
||||
"""Get the glyphless fallback font.
|
||||
|
||||
|
||||
@@ -10,6 +10,7 @@ language hints and glyph coverage analysis.
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
|
||||
from ocrmypdf.font.font_manager import FontManager
|
||||
@@ -17,6 +18,7 @@ from ocrmypdf.font.font_provider import (
|
||||
BuiltinFontProvider,
|
||||
ChainedFontProvider,
|
||||
FontProvider,
|
||||
GlyphSearchingFontProvider,
|
||||
)
|
||||
from ocrmypdf.font.system_font_provider import SystemFontProvider
|
||||
|
||||
@@ -33,9 +35,17 @@ class MultiFontManager:
|
||||
Font selection strategy:
|
||||
1. Try language-preferred font (if language hint available)
|
||||
2. Try fallback fonts in order by glyph coverage
|
||||
3. Fall back to Occulta.ttf (glyphless fallback)
|
||||
3. Ask the provider for any installed font that covers the text
|
||||
4. Fall back to Occulta.ttf (glyphless fallback)
|
||||
"""
|
||||
|
||||
# How many uncoverable characters to name in the missing-font warning
|
||||
MAX_REPORTED_CHARS = 3
|
||||
|
||||
# How many characters of a word to look up individually when composing that
|
||||
# warning; each lookup may scan every font installed on the system
|
||||
MAX_EXAMINED_CHARS = 8
|
||||
|
||||
# Language to font mapping
|
||||
# Keys are ISO 639-2/3 codes or Tesseract language codes
|
||||
LANGUAGE_FONT_MAP = {
|
||||
@@ -54,13 +64,15 @@ class MultiFontManager:
|
||||
'kok': 'NotoSansDevanagari-Regular', # Konkani
|
||||
'bho': 'NotoSansDevanagari-Regular', # Bhojpuri
|
||||
'mai': 'NotoSansDevanagari-Regular', # Maithili
|
||||
# CJK
|
||||
'chi': 'NotoSansCJK-Regular', # Chinese (generic)
|
||||
'zho': 'NotoSansCJK-Regular', # Chinese (ISO 639-3)
|
||||
'chi_sim': 'NotoSansCJK-Regular', # Chinese Simplified (Tesseract)
|
||||
'chi_tra': 'NotoSansCJK-Regular', # Chinese Traditional (Tesseract)
|
||||
'jpn': 'NotoSansCJK-Regular', # Japanese
|
||||
'kor': 'NotoSansCJK-Regular', # Korean
|
||||
# CJK — prefer the family matching the document language, because the
|
||||
# modern per-language Noto fonts are region subsets (e.g. NotoSansSC
|
||||
# lacks Japanese kana). The pan-CJK super font is a shared fallback.
|
||||
'chi': 'NotoSansSC-Regular', # Chinese (generic → Simplified)
|
||||
'zho': 'NotoSansSC-Regular', # Chinese (ISO 639-3)
|
||||
'chi_sim': 'NotoSansSC-Regular', # Chinese Simplified (Tesseract)
|
||||
'chi_tra': 'NotoSansTC-Regular', # Chinese Traditional (Tesseract)
|
||||
'jpn': 'NotoSansJP-Regular', # Japanese
|
||||
'kor': 'NotoSansKR-Regular', # Korean
|
||||
# Thai
|
||||
'tha': 'NotoSansThai-Regular', # Thai
|
||||
# Hebrew
|
||||
@@ -113,7 +125,14 @@ class MultiFontManager:
|
||||
'NotoSans-Regular', # Latin, Greek, Cyrillic
|
||||
'NotoSansArabic-Regular',
|
||||
'NotoSansDevanagari-Regular',
|
||||
# Pan-CJK super font first (full coverage), then the per-language
|
||||
# subsets so a glyph missing from one CJK family is found in another.
|
||||
'NotoSansCJK-Regular',
|
||||
'NotoSansSC-Regular',
|
||||
'NotoSansTC-Regular',
|
||||
'NotoSansHK-Regular',
|
||||
'NotoSansJP-Regular',
|
||||
'NotoSansKR-Regular',
|
||||
'NotoSansThai-Regular',
|
||||
'NotoSansHebrew-Regular',
|
||||
'NotoSansBengali-Regular',
|
||||
@@ -164,6 +183,9 @@ class MultiFontManager:
|
||||
self._selection_cache: dict[tuple[str, str | None], str] = {}
|
||||
# Track whether we've warned about missing fonts (warn once per script)
|
||||
self._warned_scripts: set[str] = set()
|
||||
# Fonts found by glyph coverage rather than by name, tried before
|
||||
# repeating the (expensive) provider search
|
||||
self._discovered_fonts: list[str] = []
|
||||
|
||||
@property
|
||||
def fonts(self) -> dict[str, FontManager]:
|
||||
@@ -199,7 +221,8 @@ class MultiFontManager:
|
||||
Uses a hybrid approach:
|
||||
1. Language-based selection (if language hint available)
|
||||
2. Ordered fallback through available fonts by glyph coverage
|
||||
3. Final fallback to Occulta.ttf (glyphless)
|
||||
3. Provider search over every installed font, by glyph coverage
|
||||
4. Final fallback to Occulta.ttf (glyphless)
|
||||
|
||||
Args:
|
||||
word_text: The text content of the word
|
||||
@@ -224,19 +247,50 @@ class MultiFontManager:
|
||||
if result := self._try_font(preferred, word_text, cache_key):
|
||||
return result
|
||||
|
||||
# Phase 2: Try fallback fonts in order
|
||||
for font_name in self.FALLBACK_FONTS:
|
||||
# Phase 2: Try fallback fonts in order, then anything a previous
|
||||
# coverage search turned up
|
||||
for font_name in [*self.FALLBACK_FONTS, *self._discovered_fonts]:
|
||||
if font_name in tried_fonts:
|
||||
continue
|
||||
tried_fonts.add(font_name)
|
||||
if result := self._try_font(font_name, word_text, cache_key):
|
||||
return result
|
||||
|
||||
# Phase 3: Glyphless fallback (always succeeds)
|
||||
# Phase 3: Ask the provider to search every installed font. The named
|
||||
# families cover common scripts only, but systems ship many more (macOS
|
||||
# installs ~100 Noto faces), and those should be used before giving up
|
||||
# on rendering the text at all. See issue #1722.
|
||||
if found := self._search_font_by_coverage(word_text):
|
||||
font_name, font = found
|
||||
self._selection_cache[cache_key] = font_name
|
||||
return font
|
||||
|
||||
# Phase 4: Glyphless fallback (always succeeds)
|
||||
# Warn if we're falling back for non-ASCII text (likely missing font)
|
||||
self._warn_missing_font(word_text, line_language)
|
||||
self._selection_cache[cache_key] = 'Occulta'
|
||||
return self.font_provider.get_fallback_font()
|
||||
|
||||
def _search_font_by_coverage(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Search the provider for any font covering text, if it supports it.
|
||||
|
||||
Args:
|
||||
text: Text the font must fully cover
|
||||
|
||||
Returns:
|
||||
(font name, FontManager), or None if unsupported or nothing matched
|
||||
"""
|
||||
provider = self.font_provider
|
||||
if not isinstance(provider, GlyphSearchingFontProvider):
|
||||
return None
|
||||
found = provider.find_font_with_glyphs(text)
|
||||
if found is None:
|
||||
return None
|
||||
font_name, _font = found
|
||||
if font_name not in self._discovered_fonts:
|
||||
self._discovered_fonts.append(font_name)
|
||||
return found
|
||||
|
||||
def _warn_missing_font(self, word_text: str, line_language: str | None) -> None:
|
||||
"""Warn user about missing font for non-Latin text.
|
||||
|
||||
@@ -255,22 +309,97 @@ class MultiFontManager:
|
||||
|
||||
self._warned_scripts.add(warn_key)
|
||||
|
||||
if line_language and line_language in self.LANGUAGE_FONT_MAP:
|
||||
font_name = self.LANGUAGE_FONT_MAP[line_language]
|
||||
uncoverable = self._uncoverable_characters(word_text)
|
||||
if not uncoverable:
|
||||
# Every character has a font, but no single font has them all.
|
||||
# Telling the user to install fonts would be wrong advice here.
|
||||
log.warning(
|
||||
"No font found with glyphs for '%s' text. "
|
||||
"Install %s for better rendering. "
|
||||
"See https://fonts.google.com/noto",
|
||||
"Text mixing scripts that no single installed font covers (%r) "
|
||||
"was added as an invisible text layer: it stays searchable and "
|
||||
"copyable, but appears blank when highlighted in a PDF viewer. "
|
||||
"Installing more fonts will not help; OCRmyPDF uses one font "
|
||||
"per word.",
|
||||
word_text,
|
||||
)
|
||||
return
|
||||
|
||||
missing = self._describe_characters(uncoverable)
|
||||
if line_language and line_language in self.LANGUAGE_FONT_MAP:
|
||||
font_family = self.LANGUAGE_FONT_MAP[line_language].removesuffix('-Regular')
|
||||
log.warning(
|
||||
"No installed font has glyphs for the detected '%s' text (%s), "
|
||||
"so it was added as an invisible text layer: it stays searchable "
|
||||
"and copyable, but appears blank when highlighted in a PDF "
|
||||
"viewer. Install the %s font family (via your OS package "
|
||||
"manager or https://fonts.google.com/noto) for full rendering.",
|
||||
line_language,
|
||||
font_name,
|
||||
missing,
|
||||
font_family,
|
||||
)
|
||||
else:
|
||||
log.warning(
|
||||
"No font found with glyphs for some text. "
|
||||
"Install Noto fonts for better rendering. "
|
||||
"See https://fonts.google.com/noto"
|
||||
"No installed font has glyphs for some of the detected text "
|
||||
"(%s), so it was added as an invisible text layer: it stays "
|
||||
"searchable and copyable, but appears blank when highlighted "
|
||||
"in a PDF viewer. Install a Noto font covering that script "
|
||||
"(https://fonts.google.com/noto) for full rendering.",
|
||||
missing,
|
||||
)
|
||||
|
||||
def _uncoverable_characters(self, word_text: str) -> list[str]:
|
||||
"""Find the characters of word_text that no installed font can render.
|
||||
|
||||
Args:
|
||||
word_text: The word that fell back to glyphless rendering
|
||||
|
||||
Returns:
|
||||
The distinct uncoverable characters, in order of first appearance,
|
||||
considering at most MAX_EXAMINED_CHARS of them
|
||||
"""
|
||||
candidates = [
|
||||
char
|
||||
for char in dict.fromkeys(word_text) # de-duplicate, keep order
|
||||
if not char.isspace() and not self._is_char_renderable(char)
|
||||
]
|
||||
# The named fonts missed these, but the provider may still have a font
|
||||
# for them, so confirm before telling the user to install anything. The
|
||||
# search walks every installed font, hence the cap on how many
|
||||
# characters we are willing to look up for one warning.
|
||||
return [
|
||||
char
|
||||
for char in candidates[: self.MAX_EXAMINED_CHARS]
|
||||
if self._search_font_by_coverage(char) is None
|
||||
]
|
||||
|
||||
def _describe_characters(self, chars: list[str]) -> str:
|
||||
"""Describe characters by codepoint and Unicode name.
|
||||
|
||||
Naming the codepoints tells the user which font to install even for
|
||||
scripts OCRmyPDF has no language mapping for, which the generic
|
||||
"install the matching Noto fonts" advice did not. See issue #1722.
|
||||
|
||||
Args:
|
||||
chars: Characters to describe
|
||||
|
||||
Returns:
|
||||
Human-readable description, truncated to MAX_REPORTED_CHARS
|
||||
"""
|
||||
described = ", ".join(
|
||||
f"{char!r} U+{ord(char):04X} {unicodedata.name(char, 'unnamed character')}"
|
||||
for char in chars[: self.MAX_REPORTED_CHARS]
|
||||
)
|
||||
if len(chars) > self.MAX_REPORTED_CHARS:
|
||||
described += f", and {len(chars) - self.MAX_REPORTED_CHARS} more"
|
||||
return described
|
||||
|
||||
def _is_char_renderable(self, char: str) -> bool:
|
||||
"""Check whether any font already known to us has a glyph for char."""
|
||||
for font_name in [*self.FALLBACK_FONTS, *self._discovered_fonts]:
|
||||
font = self.font_provider.get_font(font_name)
|
||||
if font is not None and self._has_all_glyphs(font, char):
|
||||
return True
|
||||
return False
|
||||
|
||||
def _has_all_glyphs(self, font: FontManager, text: str) -> bool:
|
||||
"""Check if a font has glyphs for all characters in text.
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ Linux, macOS, and Windows platforms.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import glob
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
@@ -75,6 +76,35 @@ class SystemFontProvider:
|
||||
# Variable fonts
|
||||
'NotoSansCJKsc-VF.otf',
|
||||
],
|
||||
# Per-language CJK families. Modern Google Fonts / Homebrew ship these
|
||||
# as region subset variable fonts ('NotoSansJP[wght].ttf'), matched by
|
||||
# the flexible base search; the legacy per-region super OTFs (full
|
||||
# coverage) are listed here so they also satisfy the logical name.
|
||||
'NotoSansSC-Regular': [
|
||||
'NotoSansSC-Regular.otf',
|
||||
'NotoSansSC-Regular.ttf',
|
||||
'NotoSansCJKsc-Regular.otf',
|
||||
],
|
||||
'NotoSansTC-Regular': [
|
||||
'NotoSansTC-Regular.otf',
|
||||
'NotoSansTC-Regular.ttf',
|
||||
'NotoSansCJKtc-Regular.otf',
|
||||
],
|
||||
'NotoSansHK-Regular': [
|
||||
'NotoSansHK-Regular.otf',
|
||||
'NotoSansHK-Regular.ttf',
|
||||
'NotoSansCJKhk-Regular.otf',
|
||||
],
|
||||
'NotoSansJP-Regular': [
|
||||
'NotoSansJP-Regular.otf',
|
||||
'NotoSansJP-Regular.ttf',
|
||||
'NotoSansCJKjp-Regular.otf',
|
||||
],
|
||||
'NotoSansKR-Regular': [
|
||||
'NotoSansKR-Regular.otf',
|
||||
'NotoSansKR-Regular.ttf',
|
||||
'NotoSansCJKkr-Regular.otf',
|
||||
],
|
||||
'NotoSansThai-Regular': [
|
||||
'NotoSansThai-Regular.ttf',
|
||||
'NotoSansThai-Regular.otf',
|
||||
@@ -149,6 +179,28 @@ class SystemFontProvider:
|
||||
],
|
||||
}
|
||||
|
||||
# Font file extensions we know how to load.
|
||||
_FONT_EXTENSIONS = ('.ttf', '.otf', '.ttc')
|
||||
|
||||
# Acceptable filename variants for a font family, ranked best-first.
|
||||
# Lower rank wins when multiple variants of the same family are present.
|
||||
_VARIANT_RANK = {'regular': 0, 'variable': 1, 'vf': 2, 'plain': 3}
|
||||
|
||||
# Extra family bases that can satisfy a logical font, tried after its own
|
||||
# base (so the listed order is the preference). CJK is the case that needs
|
||||
# this: the legacy Adobe-style 'NotoSansCJKsc-Regular.otf' is handled by
|
||||
# NOTO_FONT_PATTERNS, but Homebrew casks and current Google Fonts ship the
|
||||
# per-language families as variable fonts (e.g. 'NotoSansSC[wght].ttf').
|
||||
_ALTERNATE_BASES: dict[str, list[str]] = {
|
||||
'NotoSansCJK-Regular': [
|
||||
'NotoSansSC', # Simplified Chinese
|
||||
'NotoSansTC', # Traditional Chinese
|
||||
'NotoSansHK', # Hong Kong
|
||||
'NotoSansJP', # Japanese
|
||||
'NotoSansKR', # Korean
|
||||
],
|
||||
}
|
||||
|
||||
def __init__(self) -> None:
|
||||
"""Initialize system font provider with empty caches."""
|
||||
# Cache: font_name -> FontManager (successfully loaded fonts)
|
||||
@@ -157,6 +209,13 @@ class SystemFontProvider:
|
||||
self._not_found: set[str] = set()
|
||||
# Cached font directories (computed lazily)
|
||||
self._font_dirs: list[Path] | None = None
|
||||
# Cached (logical name, path) of every Noto face on the system, in the
|
||||
# order the coverage search should try them (computed lazily)
|
||||
self._noto_candidates: list[tuple[str, Path]] | None = None
|
||||
# Memoized results of find_font_with_glyphs(), keyed by codepoint set
|
||||
self._coverage_cache: dict[frozenset[int], str | None] = {}
|
||||
# Font files that failed to load, so we only complain about them once
|
||||
self._unloadable: set[Path] = set()
|
||||
|
||||
def _get_platform(self) -> str:
|
||||
"""Get the current platform identifier.
|
||||
@@ -222,14 +281,221 @@ class SystemFontProvider:
|
||||
try:
|
||||
matches = list(font_dir.rglob(pattern))
|
||||
if matches:
|
||||
log.debug(
|
||||
"Found system font %s at %s", font_name, matches[0]
|
||||
)
|
||||
log.debug("Found system font %s at %s", font_name, matches[0])
|
||||
return matches[0]
|
||||
except PermissionError:
|
||||
# Skip directories we can't read
|
||||
continue
|
||||
|
||||
# No exact static '-Regular' file. Many distributors (Homebrew casks,
|
||||
# current Google Fonts releases) ship Noto fonts as variable fonts with
|
||||
# bracketed axis filenames such as 'NotoSansArabic[wdth,wght].ttf'.
|
||||
# Fall back to a flexible search that also accepts those. See #1652.
|
||||
return self._find_variant_font_file(font_name)
|
||||
|
||||
@staticmethod
|
||||
def _classify_variant(stem: str, base: str) -> str | None:
|
||||
"""Classify a font filename stem as a usable variant of ``base``.
|
||||
|
||||
Args:
|
||||
stem: Filename without extension (e.g. 'NotoSansArabic[wdth,wght]')
|
||||
base: Family base name (e.g. 'NotoSansArabic')
|
||||
|
||||
Returns:
|
||||
The variant kind ('regular', 'variable', 'vf', 'plain') or None if
|
||||
the stem is not an acceptable representative of the family. The
|
||||
boundary after ``base`` is required so that 'NotoSans' does not
|
||||
match 'NotoSansArabic', and 'NotoSansArabicUI'/'NotoSansArabic-Bold'
|
||||
do not match a request for 'NotoSansArabic'.
|
||||
"""
|
||||
if stem == f'{base}-Regular':
|
||||
return 'regular'
|
||||
if stem.startswith(f'{base}['): # variable font, e.g. Base[wdth,wght]
|
||||
return 'variable'
|
||||
if stem == f'{base}-VF': # alternate variable-font naming
|
||||
return 'vf'
|
||||
if stem == base: # bare family name
|
||||
return 'plain'
|
||||
return None
|
||||
|
||||
def _find_variant_font_file(self, font_name: str) -> Path | None:
|
||||
"""Search for a variable font or other acceptable filename variant.
|
||||
|
||||
Tries the font's own family base first, then any alternate bases (used
|
||||
for the modern per-language CJK families). Within that, a static Regular
|
||||
is preferred over a variable font. See issue #1652.
|
||||
|
||||
Args:
|
||||
font_name: Logical font name (e.g. 'NotoSansArabic-Regular')
|
||||
|
||||
Returns:
|
||||
Path to the best-ranked matching font file, or None.
|
||||
"""
|
||||
bases = [font_name.removesuffix('-Regular')]
|
||||
bases.extend(self._ALTERNATE_BASES.get(font_name, []))
|
||||
|
||||
# Selection key (base_index, variant_rank): earlier base wins, then the
|
||||
# better variant. Path is carried along but not part of the comparison.
|
||||
best: tuple[tuple[int, int], Path] | None = None
|
||||
for base_index, base in enumerate(bases):
|
||||
for font_dir in self._get_font_dirs():
|
||||
if not font_dir.exists():
|
||||
continue
|
||||
try:
|
||||
for path in font_dir.rglob(glob.escape(base) + '*'):
|
||||
if path.suffix.lower() not in self._FONT_EXTENSIONS:
|
||||
continue
|
||||
kind = self._classify_variant(path.stem, base)
|
||||
if kind is None:
|
||||
continue
|
||||
key = (base_index, self._VARIANT_RANK[kind])
|
||||
if best is None or key < best[0]:
|
||||
best = (key, path)
|
||||
except PermissionError:
|
||||
# Skip directories we can't read
|
||||
continue
|
||||
if best is not None:
|
||||
log.debug("Found system font %s at %s (variant match)", font_name, best[1])
|
||||
return best[1]
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _family_base(stem: str) -> str | None:
|
||||
"""Get the Noto family base a filename stem is the Regular face of.
|
||||
|
||||
Args:
|
||||
stem: Filename without extension, e.g. 'NotoSansCherokee-Regular'
|
||||
|
||||
Returns:
|
||||
The family base ('NotoSansCherokee') or None if the stem is not a
|
||||
Noto font, or is a weight/slope variant such as '-Bold' or
|
||||
'-Italic' that should not stand in for the family.
|
||||
"""
|
||||
head = stem.split('[', 1)[0] # drop variable-font axes, e.g. '[wght]'
|
||||
if head.endswith('-Regular'):
|
||||
head = head[: -len('-Regular')]
|
||||
elif head.endswith('-VF'):
|
||||
head = head[: -len('-VF')]
|
||||
elif '-' in head:
|
||||
return None
|
||||
return head if head.startswith('Noto') else None
|
||||
|
||||
@classmethod
|
||||
def _candidate_sort_key(cls, base: str) -> tuple[int, int, str]:
|
||||
"""Rank a family base for the coverage search.
|
||||
|
||||
Sans comes before serif before everything else, and plain families come
|
||||
ahead of their narrower UI and Mono cousins.
|
||||
"""
|
||||
if base.startswith('NotoSans'):
|
||||
family_rank = 0
|
||||
elif base.startswith('NotoSerif'):
|
||||
family_rank = 1
|
||||
else:
|
||||
family_rank = 2
|
||||
narrow_use = base.endswith('UI') or base.startswith('NotoSansMono')
|
||||
return (family_rank, int(narrow_use), base)
|
||||
|
||||
def _get_noto_candidates(self) -> list[tuple[str, Path]]:
|
||||
"""Enumerate every Noto family installed on the system.
|
||||
|
||||
Scans each font directory once and keeps the best-ranked file per
|
||||
family, so a family present in several directories or in several
|
||||
variants contributes a single candidate.
|
||||
|
||||
Returns:
|
||||
List of (logical font name, path) in the order to try them.
|
||||
"""
|
||||
if self._noto_candidates is not None:
|
||||
return self._noto_candidates
|
||||
|
||||
best: dict[str, tuple[int, Path]] = {}
|
||||
for font_dir in self._get_font_dirs():
|
||||
if not font_dir.exists():
|
||||
continue
|
||||
try:
|
||||
paths = sorted(font_dir.rglob('Noto*'))
|
||||
except OSError:
|
||||
# Skip directories we can't read
|
||||
continue
|
||||
for path in paths:
|
||||
if path.suffix.lower() not in self._FONT_EXTENSIONS:
|
||||
continue
|
||||
base = self._family_base(path.stem)
|
||||
if base is None:
|
||||
continue
|
||||
kind = self._classify_variant(path.stem, base)
|
||||
if kind is None:
|
||||
continue
|
||||
rank = self._VARIANT_RANK[kind]
|
||||
if base not in best or rank < best[base][0]:
|
||||
best[base] = (rank, path)
|
||||
|
||||
self._noto_candidates = [
|
||||
(f'{base}-Regular', path)
|
||||
for base, (_rank, path) in sorted(
|
||||
best.items(), key=lambda item: self._candidate_sort_key(item[0])
|
||||
)
|
||||
]
|
||||
return self._noto_candidates
|
||||
|
||||
def find_font_with_glyphs(self, text: str) -> tuple[str, FontManager] | None:
|
||||
"""Find any installed Noto font that covers every character in text.
|
||||
|
||||
``NOTO_FONT_PATTERNS`` enumerates the couple dozen scripts OCRmyPDF
|
||||
knows by name, but systems ship far more: macOS alone installs around a
|
||||
hundred script-specific Noto faces in
|
||||
``/System/Library/Fonts/Supplemental``. This is the last resort that
|
||||
makes those usable, so a document is only rendered glyphless when no
|
||||
installed font can actually cover it. See issue #1722.
|
||||
|
||||
This walks every Noto face on the system and is therefore expensive;
|
||||
results are memoized, and callers should only reach it after the named
|
||||
fonts have failed.
|
||||
|
||||
Args:
|
||||
text: Text that the returned font must fully cover
|
||||
|
||||
Returns:
|
||||
(logical font name, FontManager) of the first covering font, or
|
||||
None if nothing installed covers the text.
|
||||
"""
|
||||
if not text:
|
||||
return None
|
||||
needed = frozenset(ord(c) for c in text)
|
||||
|
||||
if needed in self._coverage_cache:
|
||||
cached_name = self._coverage_cache[needed]
|
||||
if cached_name is None:
|
||||
return None
|
||||
if cached := self._font_cache.get(cached_name):
|
||||
return cached_name, cached
|
||||
|
||||
for font_name, path in self._get_noto_candidates():
|
||||
font = self._font_cache.get(font_name)
|
||||
if font is None:
|
||||
if path in self._unloadable:
|
||||
continue
|
||||
try:
|
||||
font = FontManager(path)
|
||||
except Exception as e:
|
||||
log.debug("Skipping unreadable font %s: %s", path, e)
|
||||
self._unloadable.add(path)
|
||||
continue
|
||||
if all(font.has_glyph(cp) for cp in needed):
|
||||
# Keep only fonts we actually use; the rest are released so a
|
||||
# full scan doesn't retain every font file on the system.
|
||||
self._font_cache[font_name] = font
|
||||
self._not_found.discard(font_name)
|
||||
self._coverage_cache[needed] = font_name
|
||||
log.debug(
|
||||
"Found system font %s at %s (glyph coverage match)",
|
||||
font_name,
|
||||
path,
|
||||
)
|
||||
return font_name, font
|
||||
|
||||
self._coverage_cache[needed] = None
|
||||
return None
|
||||
|
||||
def get_font(self, font_name: str) -> FontManager | None:
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
This module provides the PDF renderer using fpdf2 for creating
|
||||
searchable OCR text layers.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ocrmypdf.fpdf_renderer.renderer import (
|
||||
|
||||
@@ -10,13 +10,15 @@ OCR text layers.
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import unicodedata
|
||||
from dataclasses import dataclass
|
||||
from itertools import pairwise
|
||||
from math import atan, degrees
|
||||
from math import atan, cos, degrees, radians, sin, sqrt
|
||||
from pathlib import Path
|
||||
from typing import cast
|
||||
|
||||
from fpdf import FPDF
|
||||
from fpdf.enums import TextMode
|
||||
from fpdf.enums import PDFResourceType, TextMode
|
||||
from fpdf.fonts import TTFFont
|
||||
from pikepdf import Matrix, Rectangle
|
||||
|
||||
from ocrmypdf.font import FontManager, MultiFontManager
|
||||
@@ -25,6 +27,21 @@ from ocrmypdf.models.ocr_element import OcrClass, OcrElement
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _is_rtl_text(text: str) -> bool:
|
||||
"""Check if text is right-to-left based on Unicode bidi properties.
|
||||
|
||||
Looks for the first character with a strong directional type
|
||||
(R, AL, or L) to determine the text's base direction.
|
||||
"""
|
||||
for char in text:
|
||||
bidi = unicodedata.bidirectional(char)
|
||||
if bidi in ('R', 'AL'):
|
||||
return True
|
||||
if bidi == 'L':
|
||||
return False
|
||||
return False
|
||||
|
||||
|
||||
def transform_point(matrix: Matrix, x: float, y: float) -> tuple[float, float]:
|
||||
"""Transform a point (x, y) by a matrix.
|
||||
|
||||
@@ -67,6 +84,17 @@ def transform_box(
|
||||
)
|
||||
|
||||
|
||||
@dataclass
|
||||
class WordRenderData:
|
||||
"""Rendering parameters for a single word on a line."""
|
||||
|
||||
text: str
|
||||
x_baseline: float
|
||||
font_family: str
|
||||
word_tz: float
|
||||
is_rtl: bool
|
||||
|
||||
|
||||
@dataclass
|
||||
class DebugRenderOptions:
|
||||
"""Options for debug visualization during rendering.
|
||||
@@ -129,6 +157,7 @@ class Fpdf2PdfRenderer:
|
||||
dpi: float,
|
||||
multi_font_manager: MultiFontManager,
|
||||
invisible_text: bool = True,
|
||||
image: Path | None = None,
|
||||
debug_render_options: DebugRenderOptions | None = None,
|
||||
):
|
||||
"""Initialize renderer.
|
||||
@@ -138,6 +167,8 @@ class Fpdf2PdfRenderer:
|
||||
dpi: Source image DPI
|
||||
multi_font_manager: MultiFontManager instance
|
||||
invisible_text: If True, render text as invisible (text mode 3)
|
||||
image: Optional path to image to overlay on top of the text layer,
|
||||
creating a sandwich PDF (text underneath, image on top)
|
||||
debug_render_options: Options for debug visualization
|
||||
|
||||
Raises:
|
||||
@@ -152,6 +183,7 @@ class Fpdf2PdfRenderer:
|
||||
self.dpi = dpi
|
||||
self.multi_font_manager = multi_font_manager
|
||||
self.invisible_text = invisible_text
|
||||
self.image = image
|
||||
self.debug_options = debug_render_options or DebugRenderOptions()
|
||||
|
||||
# Setup coordinate transform
|
||||
@@ -163,6 +195,8 @@ class Fpdf2PdfRenderer:
|
||||
|
||||
# Registered fonts: font_path -> fpdf_family_name
|
||||
self._registered_fonts: dict[str, str] = {}
|
||||
# Track whether we've already logged the info-level suppression message
|
||||
self._logged_aspect_ratio_suppression = False
|
||||
|
||||
def render(self, output_path: Path) -> None:
|
||||
"""Render page to PDF file.
|
||||
@@ -189,9 +223,9 @@ class Fpdf2PdfRenderer:
|
||||
|
||||
# Set text mode for invisible text
|
||||
if self.invisible_text:
|
||||
pdf.text_rendering_mode = TextMode.INVISIBLE
|
||||
pdf.text_mode = TextMode.INVISIBLE
|
||||
else:
|
||||
pdf.text_rendering_mode = TextMode.FILL
|
||||
pdf.text_mode = TextMode.FILL
|
||||
|
||||
# Render content to PDF
|
||||
self.render_to_pdf(pdf)
|
||||
@@ -209,10 +243,16 @@ class Fpdf2PdfRenderer:
|
||||
pdf: FPDF instance to render into
|
||||
"""
|
||||
# Add page with correct dimensions
|
||||
# fpdf2's add_page() stub says format: str, but its docstring and
|
||||
# get_page_format() helper confirm a (width, height) tuple is
|
||||
# supported too - the annotation on add_page() itself is just wrong.
|
||||
pdf.add_page(
|
||||
format=(
|
||||
self.coord_transform.page_width_pt,
|
||||
self.coord_transform.page_height_pt,
|
||||
format=cast(
|
||||
'str',
|
||||
(
|
||||
self.coord_transform.page_width_pt,
|
||||
self.coord_transform.page_height_pt,
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
@@ -225,6 +265,16 @@ class Fpdf2PdfRenderer:
|
||||
for line in self.page.lines:
|
||||
self._render_line(pdf, line)
|
||||
|
||||
# Place image on top of text layer (sandwich mode)
|
||||
if self.image is not None:
|
||||
pdf.image(
|
||||
str(self.image),
|
||||
x=0,
|
||||
y=0,
|
||||
w=self.coord_transform.page_width_pt,
|
||||
h=self.coord_transform.page_height_pt,
|
||||
)
|
||||
|
||||
def _register_font(self, pdf: FPDF, font_manager: FontManager) -> str:
|
||||
"""Register font with fpdf2 if not already registered.
|
||||
|
||||
@@ -299,6 +349,30 @@ class Fpdf2PdfRenderer:
|
||||
# Get textangle (rotation of the entire line)
|
||||
textangle = line.textangle or 0.0
|
||||
|
||||
# Read baseline early so we can detect rotation from steep slopes.
|
||||
# When Tesseract doesn't report textangle for rotated text, the
|
||||
# rotation gets encoded as a very steep baseline slope instead.
|
||||
slope = 0.0
|
||||
intercept_pt = 0.0
|
||||
has_meaningful_baseline = False
|
||||
if line.baseline is not None:
|
||||
slope = line.baseline.slope
|
||||
intercept_pt = self.coord_transform.px_to_pt(line.baseline.intercept)
|
||||
if abs(slope) < 0.005:
|
||||
slope = 0.0
|
||||
has_meaningful_baseline = True
|
||||
|
||||
# Detect text rotation from steep baseline slope.
|
||||
# A slope magnitude > 1.0 corresponds to > 45° from horizontal,
|
||||
# which indicates the line is rotated, not merely skewed.
|
||||
if textangle == 0.0 and abs(slope) > 1.0:
|
||||
textangle = degrees(atan(slope))
|
||||
# The original baseline slope and intercept are not meaningful
|
||||
# after extracting rotation; recalculate intercept from font
|
||||
# metrics below.
|
||||
slope = 0.0
|
||||
has_meaningful_baseline = False
|
||||
|
||||
# Build line_size_aabb_matrix: transforms from page coords to un-rotated
|
||||
# line coords. The hOCR bbox is the minimum axis-aligned bounding box
|
||||
# enclosing the rotated text.
|
||||
@@ -317,16 +391,10 @@ class Fpdf2PdfRenderer:
|
||||
inv_line_matrix, line_left_pt, line_top_pt, line_right_pt, line_bottom_pt
|
||||
)
|
||||
|
||||
# Get baseline information (slope and intercept)
|
||||
slope = 0.0
|
||||
intercept_pt = 0.0
|
||||
if line.baseline is not None:
|
||||
slope = line.baseline.slope
|
||||
intercept_pt = self.coord_transform.px_to_pt(line.baseline.intercept)
|
||||
if abs(slope) < 0.005:
|
||||
slope = 0.0
|
||||
else:
|
||||
# No baseline provided: calculate from font metrics
|
||||
# Get baseline intercept
|
||||
if not has_meaningful_baseline:
|
||||
# No baseline provided or baseline was used for rotation detection:
|
||||
# calculate intercept from font metrics
|
||||
default_font_manager = self.multi_font_manager.fonts['NotoSans-Regular']
|
||||
ascent, descent, units_per_em = default_font_manager.get_font_metrics()
|
||||
ascent_norm = ascent / units_per_em
|
||||
@@ -383,143 +451,364 @@ class Fpdf2PdfRenderer:
|
||||
w for w in line.children if w.ocr_class == OcrClass.WORD and w.text
|
||||
]
|
||||
|
||||
# Render each word followed by space (except last)
|
||||
# Use pairwise to iterate over consecutive word pairs, pairing the last
|
||||
# word with a None to signal the end of the line.
|
||||
for current_word, next_word in pairwise(words + [None]):
|
||||
if current_word: # Don't render EOL sentinel
|
||||
# Render the current word
|
||||
self._render_word(
|
||||
pdf,
|
||||
current_word,
|
||||
baseline_matrix,
|
||||
inv_baseline_matrix,
|
||||
font_size,
|
||||
total_rotation_deg,
|
||||
line_language,
|
||||
)
|
||||
if next_word: # Don't render EOL sentinel
|
||||
self._maybe_render_space(
|
||||
pdf,
|
||||
current_word,
|
||||
next_word,
|
||||
baseline_matrix,
|
||||
inv_baseline_matrix,
|
||||
font_size,
|
||||
total_rotation_deg,
|
||||
line_language,
|
||||
line.direction,
|
||||
# Suppress lines where the text aspect ratio is implausible.
|
||||
# This catches cases where Tesseract failed to detect rotation
|
||||
# entirely (slope=0, no textangle) and produced garbage text in a
|
||||
# bounding box whose shape doesn't match the text content at all.
|
||||
if not self._check_aspect_ratio_plausible(
|
||||
pdf,
|
||||
words,
|
||||
font_size,
|
||||
slope_angle_deg,
|
||||
line_size_width,
|
||||
line_size_height,
|
||||
line_language,
|
||||
):
|
||||
return
|
||||
|
||||
word_render_data: list[WordRenderData] = []
|
||||
for word in words:
|
||||
if word is None or not word.text or word.bbox is None:
|
||||
continue
|
||||
|
||||
word_left_pt = self.coord_transform.px_to_pt(word.bbox.left)
|
||||
word_top_pt = self.coord_transform.px_to_pt(word.bbox.top)
|
||||
word_right_pt = self.coord_transform.px_to_pt(word.bbox.right)
|
||||
word_bottom_pt = self.coord_transform.px_to_pt(word.bbox.bottom)
|
||||
word_width_pt = word_right_pt - word_left_pt
|
||||
|
||||
# Debug rendering: draw word bbox (in page coordinates)
|
||||
if self.debug_options.render_word_bbox:
|
||||
self._render_debug_word_bbox(
|
||||
pdf, word_left_pt, word_top_pt, word_right_pt, word_bottom_pt
|
||||
)
|
||||
|
||||
def _render_word(
|
||||
# Get x position in baseline coordinate system
|
||||
box_llx, _, _, _ = transform_box(
|
||||
inv_baseline_matrix,
|
||||
word_left_pt,
|
||||
word_top_pt,
|
||||
word_right_pt,
|
||||
word_bottom_pt,
|
||||
)
|
||||
|
||||
# Select font and compute word-only Tz
|
||||
font_manager = self.multi_font_manager.select_font_for_word(
|
||||
word.text, line_language
|
||||
)
|
||||
font_family = self._register_font(pdf, font_manager)
|
||||
pdf.set_font(font_family, size=font_size)
|
||||
|
||||
# For RTL words with invisible text, we use encode_text()
|
||||
# (which maps characters 1:1 in logical order) combined with
|
||||
# a -1 x-scale text matrix. This avoids an fpdf2 issue where
|
||||
# shaped RTL ligature glyphs (e.g. lam-alef) get multi-char
|
||||
# CMap entries whose character order is reversed by the bidi
|
||||
# algorithm during text extraction.
|
||||
# Since the text is invisible, glyph mirroring is harmless.
|
||||
# Compute Tz using unshaped widths to match encode_text().
|
||||
word_is_rtl = self.invisible_text and _is_rtl_text(word.text)
|
||||
if word_is_rtl:
|
||||
saved_shaping = pdf.text_shaping
|
||||
pdf.text_shaping = None
|
||||
natural_width = pdf.get_string_width(word.text)
|
||||
pdf.text_shaping = saved_shaping
|
||||
else:
|
||||
natural_width = pdf.get_string_width(word.text)
|
||||
if natural_width > 0 and word_width_pt > 0:
|
||||
word_tz = (word_width_pt / natural_width) * 100
|
||||
else:
|
||||
word_tz = 100.0
|
||||
|
||||
word_render_data.append(
|
||||
WordRenderData(
|
||||
text=word.text,
|
||||
x_baseline=box_llx,
|
||||
font_family=font_family,
|
||||
word_tz=word_tz,
|
||||
is_rtl=word_is_rtl,
|
||||
)
|
||||
)
|
||||
|
||||
if not word_render_data:
|
||||
return
|
||||
|
||||
# Emit single BT block for the entire line using raw PDF operators.
|
||||
# This avoids a poppler bug where Tz (horizontal scaling) is not
|
||||
# carried across BT/ET boundaries, affecting all poppler-based tools
|
||||
# and viewers (Evince, pdftotext, etc.). By keeping all words in a
|
||||
# single BT block with relative Td positioning and per-word Tz, we
|
||||
# ensure correct inter-word spacing.
|
||||
self._emit_line_bt_block(
|
||||
pdf,
|
||||
word_render_data,
|
||||
baseline_matrix,
|
||||
font_size,
|
||||
total_rotation_deg,
|
||||
)
|
||||
|
||||
def _check_aspect_ratio_plausible(
|
||||
self,
|
||||
pdf: FPDF,
|
||||
word: OcrElement,
|
||||
baseline_matrix: Matrix,
|
||||
inv_baseline_matrix: Matrix,
|
||||
words: list[OcrElement | None],
|
||||
font_size: float,
|
||||
rotation_deg: float,
|
||||
slope_angle_deg: float,
|
||||
line_size_width: float,
|
||||
line_size_height: float,
|
||||
line_language: str | None,
|
||||
) -> None:
|
||||
"""Render a word using word bbox positioning.
|
||||
) -> bool:
|
||||
"""Check whether the line's aspect ratio is plausible for its text.
|
||||
|
||||
Position text so its visual bounding box matches the hOCR word bbox.
|
||||
This provides more accurate placement than baseline-relative positioning
|
||||
because we match the actual glyph bounds rather than relying on font
|
||||
metrics which may not exactly match the OCR'd text appearance.
|
||||
Compares the aspect ratio of the OCR bounding box to the aspect ratio
|
||||
the text would have if rendered normally (accounting for baseline
|
||||
slope). A large mismatch indicates Tesseract misread rotated text
|
||||
without detecting the rotation.
|
||||
|
||||
Returns:
|
||||
True if plausible (rendering should proceed), False to suppress.
|
||||
"""
|
||||
if line_size_width <= 0 or line_size_height <= 0 or font_size <= 0:
|
||||
return True
|
||||
|
||||
# Fast path: most lines are wider than they are tall, which is
|
||||
# the normal shape for horizontal text. Only tall-narrow boxes
|
||||
# (height > width) need the expensive font measurement check.
|
||||
if line_size_width >= line_size_height:
|
||||
return True
|
||||
|
||||
line_text = ' '.join(w.text for w in words if w is not None and w.text)
|
||||
if not line_text:
|
||||
return True
|
||||
|
||||
# Measure the natural rendered width of the line text
|
||||
font_manager = self.multi_font_manager.select_font_for_word(
|
||||
line_text, line_language
|
||||
)
|
||||
font_family = self._register_font(pdf, font_manager)
|
||||
pdf.set_font(font_family, size=round(font_size))
|
||||
natural_width = pdf.get_string_width(line_text)
|
||||
|
||||
if natural_width <= 0:
|
||||
return True
|
||||
|
||||
# Compute the AABB the text would occupy considering baseline slope
|
||||
theta = radians(abs(slope_angle_deg))
|
||||
expected_w = natural_width * cos(theta) + font_size * sin(theta)
|
||||
expected_h = natural_width * sin(theta) + font_size * cos(theta)
|
||||
|
||||
if expected_h <= 0:
|
||||
return True
|
||||
|
||||
actual_aspect = line_size_width / line_size_height
|
||||
expected_aspect = expected_w / expected_h
|
||||
ratio = actual_aspect / expected_aspect
|
||||
|
||||
if ratio >= 0.1:
|
||||
return True
|
||||
|
||||
# Implausible aspect ratio — suppress this line
|
||||
log.debug(
|
||||
"Suppressing text with improbable aspect ratio: "
|
||||
"actual=%.3f expected=%.3f ratio=%.4f text=%r",
|
||||
actual_aspect,
|
||||
expected_aspect,
|
||||
ratio,
|
||||
line_text[:80],
|
||||
)
|
||||
if not self._logged_aspect_ratio_suppression:
|
||||
log.info("Suppressing OCR output text with improbable aspect ratio")
|
||||
self._logged_aspect_ratio_suppression = True
|
||||
return False
|
||||
|
||||
def _emit_line_bt_block(
|
||||
self,
|
||||
pdf: FPDF,
|
||||
word_render_data: list[WordRenderData],
|
||||
baseline_matrix: Matrix,
|
||||
font_size: float,
|
||||
total_rotation_deg: float,
|
||||
) -> None:
|
||||
"""Emit a single BT block for the entire line using raw PDF operators.
|
||||
|
||||
Writes all words in a single BT..ET block with relative Td positioning
|
||||
and per-word Tz. Each non-last word gets a trailing space appended, with
|
||||
Tz calculated so the rendered width of "word " spans from the current
|
||||
word's start to the next word's start. This works around a poppler bug
|
||||
where Tz is not carried across BT/ET boundaries, which affects all
|
||||
poppler-based viewers and tools (Evince, pdftotext, etc.).
|
||||
|
||||
Args:
|
||||
pdf: FPDF instance
|
||||
word: Word OCR element
|
||||
word_render_data: List of WordRenderData, one per word on this line
|
||||
baseline_matrix: Transform from baseline coords to page coords
|
||||
inv_baseline_matrix: Transform from page coords to baseline coords
|
||||
font_size: Font size in points (from line calculation)
|
||||
rotation_deg: Total rotation angle for text
|
||||
line_language: Language code from line for font selection
|
||||
font_size: Font size in points
|
||||
total_rotation_deg: Total rotation angle (textangle + slope)
|
||||
"""
|
||||
if not word.text or word.bbox is None:
|
||||
return
|
||||
page_height = self.coord_transform.page_height_pt
|
||||
|
||||
# Select appropriate font for this word
|
||||
font_manager = self.multi_font_manager.select_font_for_word(
|
||||
word.text, line_language
|
||||
)
|
||||
# Compute baseline direction in PDF coordinates for rotation
|
||||
has_rotation = abs(total_rotation_deg) > 0.01
|
||||
bx0, by0_fpdf = transform_point(baseline_matrix, 0, 0)
|
||||
by0_pdf = page_height - by0_fpdf
|
||||
|
||||
# Register font with fpdf2
|
||||
font_family = self._register_font(pdf, font_manager)
|
||||
ops: list[str] = []
|
||||
|
||||
# Convert word bbox to PDF points
|
||||
word_left_pt = self.coord_transform.px_to_pt(word.bbox.left)
|
||||
word_top_pt = self.coord_transform.px_to_pt(word.bbox.top)
|
||||
word_right_pt = self.coord_transform.px_to_pt(word.bbox.right)
|
||||
word_bottom_pt = self.coord_transform.px_to_pt(word.bbox.bottom)
|
||||
word_width_pt = word_right_pt - word_left_pt
|
||||
if has_rotation:
|
||||
# Compute direction vector along the baseline in PDF coordinates
|
||||
bx1, by1_fpdf = transform_point(baseline_matrix, 100, 0)
|
||||
by1_pdf = page_height - by1_fpdf
|
||||
dx = bx1 - bx0
|
||||
dy = by1_pdf - by0_pdf
|
||||
length = sqrt(dx * dx + dy * dy)
|
||||
if length > 0:
|
||||
cos_a = dx / length
|
||||
sin_a = dy / length
|
||||
else:
|
||||
cos_a = 1.0
|
||||
sin_a = 0.0
|
||||
|
||||
# Transform word bbox into baseline coordinate system to get x position
|
||||
box_llx, _, _, _ = transform_box(
|
||||
inv_baseline_matrix,
|
||||
word_left_pt,
|
||||
word_top_pt,
|
||||
word_right_pt,
|
||||
word_bottom_pt,
|
||||
)
|
||||
|
||||
# Debug rendering: draw word bbox (in page coordinates)
|
||||
if self.debug_options.render_word_bbox:
|
||||
self._render_debug_word_bbox(
|
||||
pdf, word_left_pt, word_top_pt, word_right_pt, word_bottom_pt
|
||||
# Save graphics state, apply rotation+translation via cm.
|
||||
# The cm maps local coordinates (baseline-aligned, x along text)
|
||||
# to PDF page coordinates.
|
||||
ops.append('q')
|
||||
ops.append(
|
||||
f'{cos_a:.6f} {sin_a:.6f} {-sin_a:.6f} {cos_a:.6f} '
|
||||
f'{bx0:.2f} {by0_pdf:.2f} cm'
|
||||
)
|
||||
|
||||
# Use line-based font_size for consistent vertical sizing
|
||||
word_font_size = font_size
|
||||
# Begin text object
|
||||
ops.append('BT')
|
||||
|
||||
# Set font
|
||||
pdf.set_font(font_family, size=word_font_size)
|
||||
# Text render mode: 3 = invisible, 0 = fill
|
||||
tr = 3 if self.invisible_text else 0
|
||||
ops.append(f'{tr} Tr')
|
||||
|
||||
# Calculate natural text width at this font size
|
||||
natural_width = pdf.get_string_width(word.text)
|
||||
|
||||
# Calculate horizontal scale to fit word bbox width
|
||||
if natural_width > 0 and word_width_pt > 0:
|
||||
scale_x = (word_width_pt / natural_width) * 100
|
||||
# Initial text position
|
||||
first_x_baseline = word_render_data[0].x_baseline
|
||||
if has_rotation:
|
||||
# In the cm-transformed space, origin is at the baseline start
|
||||
ops.append(f'{first_x_baseline:.2f} 0 Td')
|
||||
else:
|
||||
scale_x = 100
|
||||
# Direct PDF coordinates
|
||||
page_x, page_y_fpdf = transform_point(baseline_matrix, first_x_baseline, 0)
|
||||
page_y_pdf = page_height - page_y_fpdf
|
||||
ops.append(f'{page_x:.2f} {page_y_pdf:.2f} Td')
|
||||
|
||||
# Apply horizontal stretching
|
||||
pdf.set_stretching(scale_x)
|
||||
prev_font_family: str | None = None
|
||||
prev_x_baseline = first_x_baseline
|
||||
|
||||
# Get left side bearing of first character to compensate for glyph offset
|
||||
lsb_pt = font_manager.get_left_side_bearing(word.text[0], word_font_size)
|
||||
for i, word in enumerate(word_render_data):
|
||||
is_last = i == len(word_render_data) - 1
|
||||
|
||||
# Transform the baseline-relative x position back to page coordinates
|
||||
# The word sits at (box_llx, 0) in baseline coords (on the baseline)
|
||||
page_x, page_y = transform_point(baseline_matrix, box_llx, 0)
|
||||
# Set font if changed
|
||||
if word.font_family != prev_font_family:
|
||||
pdf.set_font(word.font_family, size=font_size)
|
||||
# We only ever register fonts via add_font() with a TTF file
|
||||
# (see _register_font), so set_font() always resolves to a
|
||||
# TTFFont, never a built-in CoreFont or leaves it unset.
|
||||
assert pdf.current_font is not None
|
||||
# Register font resource on this page
|
||||
pdf._resource_catalog.add(
|
||||
PDFResourceType.FONT, pdf.current_font.i, pdf.page
|
||||
)
|
||||
ops.append(f'/F{pdf.current_font.i} {pdf.font_size_pt:.2f} Tf')
|
||||
prev_font_family = word.font_family
|
||||
|
||||
# Adjust x position to account for lsb (scaled by horizontal stretch)
|
||||
adjusted_x = page_x - lsb_pt * (scale_x / 100)
|
||||
# Relative positioning (for words after the first)
|
||||
if i > 0:
|
||||
if has_rotation:
|
||||
# In rotated space, advance is purely along x-axis
|
||||
dx_baseline = word.x_baseline - prev_x_baseline
|
||||
ops.append(f'{dx_baseline:.2f} 0 Td')
|
||||
else:
|
||||
# Non-rotated: compute delta in PDF coordinates
|
||||
px_prev, py_prev_f = transform_point(
|
||||
baseline_matrix, prev_x_baseline, 0
|
||||
)
|
||||
px_curr, py_curr_f = transform_point(
|
||||
baseline_matrix, word.x_baseline, 0
|
||||
)
|
||||
dx_pdf = px_curr - px_prev
|
||||
# Flip y delta for PDF coordinates (y-up)
|
||||
dy_pdf = -(py_curr_f - py_prev_f)
|
||||
ops.append(f'{dx_pdf:.2f} {dy_pdf:.2f} Td')
|
||||
|
||||
# Calculate y position based on baseline
|
||||
# In fpdf2, set_xy(x, y) positions text such that the baseline is at:
|
||||
# baseline_y = set_y + font_size * (ascent / (ascent + |descent|))
|
||||
# We want baseline at page_y, so:
|
||||
# page_y = set_y + font_size * (ascent / (ascent + |descent|))
|
||||
# set_y = page_y - font_size * (ascent / (ascent + |descent|))
|
||||
ascent, descent, _ = font_manager.get_font_metrics()
|
||||
total_height = ascent + abs(descent)
|
||||
baseline_offset_ratio = ascent / total_height
|
||||
adjusted_y = page_y - word_font_size * baseline_offset_ratio
|
||||
# Determine text to render
|
||||
if not is_last:
|
||||
next_word = word_render_data[i + 1]
|
||||
advance = next_word.x_baseline - word.x_baseline
|
||||
|
||||
# Position and draw text with rotation
|
||||
if abs(rotation_deg) > 0.1:
|
||||
with pdf.rotation(-rotation_deg, x=page_x, y=page_y):
|
||||
pdf.set_xy(adjusted_x, adjusted_y)
|
||||
pdf.cell(text=word.text)
|
||||
else:
|
||||
pdf.set_xy(adjusted_x, adjusted_y)
|
||||
pdf.cell(text=word.text)
|
||||
# Add trailing space for text extraction unless both are CJK
|
||||
if advance > 0 and not (
|
||||
self._is_cjk_only(word.text) and self._is_cjk_only(next_word.text)
|
||||
):
|
||||
text_to_render = word.text + ' '
|
||||
else:
|
||||
text_to_render = word.text
|
||||
else:
|
||||
text_to_render = word.text
|
||||
|
||||
# Reset stretching
|
||||
pdf.set_stretching(100)
|
||||
# Use word_tz (fits word into its hOCR bbox) — Td handles
|
||||
# inter-word gaps, so Tz should not stretch to fill them.
|
||||
ops.append(f'{word.word_tz:.2f} Tz')
|
||||
ops.append(self._encode_shaped_text(pdf, text_to_render, word.is_rtl))
|
||||
|
||||
prev_x_baseline = word.x_baseline
|
||||
|
||||
# End text object
|
||||
ops.append('ET')
|
||||
|
||||
if has_rotation:
|
||||
ops.append('Q')
|
||||
|
||||
pdf._out('\n'.join(ops))
|
||||
|
||||
# Reset fpdf2's internal stretching tracking so subsequent API calls
|
||||
# don't think Tz is still set from our raw operators
|
||||
pdf.font_stretching = 100
|
||||
|
||||
def _encode_shaped_text(self, pdf: FPDF, text: str, is_rtl: bool = False) -> str:
|
||||
"""Encode text using HarfBuzz text shaping for complex script support.
|
||||
|
||||
Unlike font.encode_text() which maps unicode characters one-by-one to
|
||||
glyph IDs, this uses HarfBuzz to handle BiDi reordering, Arabic joining
|
||||
forms, Devanagari conjuncts, and other complex script shaping. Falls
|
||||
back to encode_text() when text shaping is not enabled.
|
||||
|
||||
For RTL words with invisible text, we use encode_text() instead of
|
||||
shape_text(). fpdf2's shape_text() produces RTL ligature glyphs
|
||||
(e.g. lam-alef) with multi-character CMap entries whose character
|
||||
order gets reversed by the bidi algorithm during text extraction,
|
||||
producing garbled output (e.g. "سالح" instead of "سلاح").
|
||||
encode_text() maps characters 1:1 in logical order, giving correct
|
||||
extraction. Since the text is invisible (Tr=3), the lack of proper
|
||||
joining forms and ligature shaping is harmless.
|
||||
"""
|
||||
font = pdf.current_font
|
||||
# We only ever register fonts via add_font() with a TTF file (see
|
||||
# _register_font), so current_font is always a TTFFont - never the
|
||||
# built-in CoreFont (which lacks shape_text()/escape_text()) or None.
|
||||
assert isinstance(font, TTFFont)
|
||||
if is_rtl:
|
||||
# Reverse the text so that after bidi reversal by the text
|
||||
# extractor, the characters end up in correct logical order.
|
||||
# The text cursor advances left-to-right from the word's left
|
||||
# edge (set by Td), so characters are positioned left-to-right
|
||||
# in the PDF. The extractor sees RTL characters in L-to-R
|
||||
# positions and applies bidi reversal, which reverses them.
|
||||
# By pre-reversing, the double reversal yields the original.
|
||||
return font.encode_text(text[::-1])
|
||||
if pdf.text_shaping and pdf.text_shaping.get("use_shaping_engine"):
|
||||
shaped = font.shape_text(text, pdf.font_size_pt, pdf.text_shaping)
|
||||
if shaped:
|
||||
mapped = "".join(
|
||||
chr(ti["mapped_char"])
|
||||
for ti in shaped
|
||||
if ti["mapped_char"] is not None
|
||||
)
|
||||
if mapped:
|
||||
return f"({font.escape_text(mapped)}) Tj"
|
||||
return font.encode_text(text)
|
||||
|
||||
def _is_cjk_only(self, text: str) -> bool:
|
||||
"""Check if text contains only CJK characters.
|
||||
@@ -559,157 +848,6 @@ class Fpdf2PdfRenderer:
|
||||
return False
|
||||
return True
|
||||
|
||||
def _maybe_render_space(
|
||||
self,
|
||||
pdf: FPDF,
|
||||
current_word: OcrElement,
|
||||
next_word: OcrElement,
|
||||
baseline_matrix: Matrix,
|
||||
inv_baseline_matrix: Matrix,
|
||||
font_size: float,
|
||||
rotation_deg: float,
|
||||
line_language: str | None,
|
||||
direction: str | None,
|
||||
) -> None:
|
||||
"""Render a space character between two words if a gap exists.
|
||||
|
||||
This ensures that PDF readers like pdfminer.six can properly segment
|
||||
words during text extraction. Some PDF readers rely on explicit space
|
||||
characters rather than inferring word boundaries from positioning.
|
||||
|
||||
Args:
|
||||
pdf: FPDF instance
|
||||
current_word: The word that was just rendered
|
||||
next_word: The next word to be rendered
|
||||
baseline_matrix: Transform from baseline coords to page coords
|
||||
inv_baseline_matrix: Transform from page coords to baseline coords
|
||||
font_size: Font size in points
|
||||
rotation_deg: Total rotation angle for text
|
||||
line_language: Language code from line for font selection
|
||||
direction: Text direction ("ltr" or "rtl")
|
||||
"""
|
||||
if current_word.bbox is None or next_word.bbox is None:
|
||||
return
|
||||
|
||||
# Skip if both words are CJK-only (no spaces in CJK text)
|
||||
if self._is_cjk_only(current_word.text) and self._is_cjk_only(next_word.text):
|
||||
return
|
||||
|
||||
# Calculate gap between words
|
||||
if direction == "rtl":
|
||||
gap_left = next_word.bbox.right
|
||||
gap_right = current_word.bbox.left
|
||||
else:
|
||||
gap_left = current_word.bbox.right
|
||||
gap_right = next_word.bbox.left
|
||||
|
||||
gap_width_px = gap_right - gap_left
|
||||
|
||||
# Use word height as proxy for line height
|
||||
line_height_px = current_word.bbox.height
|
||||
|
||||
# Skip if gap is too small (noise) or words are overlapping
|
||||
if gap_width_px <= line_height_px * 0.05:
|
||||
return
|
||||
|
||||
# Render space in the gap
|
||||
self._render_space(
|
||||
pdf,
|
||||
gap_left,
|
||||
gap_right,
|
||||
current_word.bbox.top,
|
||||
current_word.bbox.bottom,
|
||||
baseline_matrix,
|
||||
inv_baseline_matrix,
|
||||
font_size,
|
||||
rotation_deg,
|
||||
line_language,
|
||||
)
|
||||
|
||||
def _render_space(
|
||||
self,
|
||||
pdf: FPDF,
|
||||
gap_left_px: float,
|
||||
gap_right_px: float,
|
||||
gap_top_px: float,
|
||||
gap_bottom_px: float,
|
||||
baseline_matrix: Matrix,
|
||||
inv_baseline_matrix: Matrix,
|
||||
font_size: float,
|
||||
rotation_deg: float,
|
||||
line_language: str | None,
|
||||
) -> None:
|
||||
"""Render a space character in a gap between words.
|
||||
|
||||
Uses the same baseline transformation logic as word rendering to ensure
|
||||
proper alignment on rotated or sloped baselines.
|
||||
|
||||
Args:
|
||||
pdf: FPDF instance
|
||||
gap_left_px: Left edge of gap in pixels
|
||||
gap_right_px: Right edge of gap in pixels
|
||||
gap_top_px: Top edge of gap in pixels
|
||||
gap_bottom_px: Bottom edge of gap in pixels
|
||||
baseline_matrix: Transform from baseline coords to page coords
|
||||
inv_baseline_matrix: Transform from page coords to baseline coords
|
||||
font_size: Font size in points
|
||||
rotation_deg: Total rotation angle for text
|
||||
line_language: Language code from line for font selection
|
||||
"""
|
||||
# Convert gap to PDF points
|
||||
gap_left_pt = self.coord_transform.px_to_pt(gap_left_px)
|
||||
gap_top_pt = self.coord_transform.px_to_pt(gap_top_px)
|
||||
gap_right_pt = self.coord_transform.px_to_pt(gap_right_px)
|
||||
gap_bottom_pt = self.coord_transform.px_to_pt(gap_bottom_px)
|
||||
gap_width_pt = gap_right_pt - gap_left_pt
|
||||
|
||||
# Transform gap bbox into baseline coordinate system to get x position
|
||||
box_llx, _, _, _ = transform_box(
|
||||
inv_baseline_matrix,
|
||||
gap_left_pt,
|
||||
gap_top_pt,
|
||||
gap_right_pt,
|
||||
gap_bottom_pt,
|
||||
)
|
||||
|
||||
# Select font (use default font for space)
|
||||
font_manager = self.multi_font_manager.select_font_for_word(" ", line_language)
|
||||
font_family = self._register_font(pdf, font_manager)
|
||||
|
||||
# Set font
|
||||
pdf.set_font(font_family, size=font_size)
|
||||
|
||||
# Calculate natural space width and scaling
|
||||
natural_width = pdf.get_string_width(" ")
|
||||
if natural_width > 0 and gap_width_pt > 0:
|
||||
scale_x = (gap_width_pt / natural_width) * 100
|
||||
else:
|
||||
scale_x = 100
|
||||
|
||||
# Apply horizontal stretching
|
||||
pdf.set_stretching(scale_x)
|
||||
|
||||
# Transform the baseline-relative x position back to page coordinates
|
||||
page_x, page_y = transform_point(baseline_matrix, box_llx, 0)
|
||||
|
||||
# Calculate y position based on baseline (same as _render_word)
|
||||
ascent, descent, _ = font_manager.get_font_metrics()
|
||||
total_height = ascent + abs(descent)
|
||||
baseline_offset_ratio = ascent / total_height
|
||||
adjusted_y = page_y - font_size * baseline_offset_ratio
|
||||
|
||||
# Position and draw space with rotation
|
||||
if abs(rotation_deg) > 0.1:
|
||||
with pdf.rotation(-rotation_deg, x=page_x, y=page_y):
|
||||
pdf.set_xy(page_x, adjusted_y)
|
||||
pdf.cell(text=" ")
|
||||
else:
|
||||
pdf.set_xy(page_x, adjusted_y)
|
||||
pdf.cell(text=" ")
|
||||
|
||||
# Reset stretching
|
||||
pdf.set_stretching(100)
|
||||
|
||||
def _render_debug_line_bbox(
|
||||
self,
|
||||
pdf: FPDF,
|
||||
@@ -802,9 +940,9 @@ class Fpdf2MultiPageRenderer:
|
||||
|
||||
# Set text mode for invisible text
|
||||
if self.invisible_text:
|
||||
pdf.text_rendering_mode = TextMode.INVISIBLE
|
||||
pdf.text_mode = TextMode.INVISIBLE
|
||||
else:
|
||||
pdf.text_rendering_mode = TextMode.FILL
|
||||
pdf.text_mode = TextMode.FILL
|
||||
|
||||
# Shared font registration across all pages
|
||||
shared_registered_fonts: dict[str, str] = {}
|
||||
|
||||
+52
-17
@@ -18,6 +18,7 @@ from math import isclose, isfinite
|
||||
from pathlib import Path
|
||||
from statistics import harmonic_mean
|
||||
from typing import (
|
||||
TYPE_CHECKING,
|
||||
Any,
|
||||
Generic,
|
||||
TypeVar,
|
||||
@@ -26,6 +27,9 @@ from typing import (
|
||||
import img2pdf
|
||||
import pikepdf
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from _typeshed import StrOrBytesPath
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
IMG2PDF_KWARGS = dict(engine=img2pdf.Engine.pikepdf, rotation=img2pdf.Rotation.ifvalid)
|
||||
@@ -135,7 +139,7 @@ class Resolution(Generic[T]):
|
||||
return self._isclose(self.x, other.x) and self._isclose(self.y, other.y)
|
||||
|
||||
|
||||
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike) -> None:
|
||||
def safe_symlink(input_file: StrOrBytesPath, soft_link_name: StrOrBytesPath) -> None:
|
||||
"""Create a symbolic link at ``soft_link_name``, which references ``input_file``.
|
||||
|
||||
Think of this as copying ``input_file`` to ``soft_link_name`` with less overhead.
|
||||
@@ -144,11 +148,11 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike) -> None:
|
||||
used since symlinks may require administrator privileges. An existing link at the
|
||||
destination is removed.
|
||||
"""
|
||||
input_file = os.fspath(input_file)
|
||||
soft_link_name = os.fspath(soft_link_name)
|
||||
input_path = Path(os.fsdecode(input_file))
|
||||
soft_link_path = Path(os.fsdecode(soft_link_name))
|
||||
|
||||
# Guard against soft linking to oneself
|
||||
if input_file == soft_link_name:
|
||||
if input_path == soft_link_path:
|
||||
log.warning(
|
||||
"No symbolic link created. You are using the original data directory "
|
||||
"as the working directory."
|
||||
@@ -156,24 +160,24 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike) -> None:
|
||||
return
|
||||
|
||||
# Soft link already exists: delete for relink?
|
||||
if os.path.lexists(soft_link_name):
|
||||
if os.path.lexists(soft_link_path):
|
||||
# do not delete or overwrite real (non-soft link) file
|
||||
if not os.path.islink(soft_link_name):
|
||||
raise FileExistsError(f"{soft_link_name} exists and is not a link")
|
||||
os.unlink(soft_link_name)
|
||||
if not soft_link_path.is_symlink():
|
||||
raise FileExistsError(f"{soft_link_path} exists and is not a link")
|
||||
soft_link_path.unlink()
|
||||
|
||||
if not os.path.exists(input_file):
|
||||
raise FileNotFoundError(f"trying to create a broken symlink to {input_file}")
|
||||
if not input_path.exists():
|
||||
raise FileNotFoundError(f"trying to create a broken symlink to {input_path}")
|
||||
|
||||
if os.name == 'nt':
|
||||
# Don't actually use symlinks on Windows due to permission issues
|
||||
shutil.copyfile(input_file, soft_link_name)
|
||||
shutil.copyfile(input_path, soft_link_path)
|
||||
return
|
||||
|
||||
log.debug("os.symlink(%s, %s)", input_file, soft_link_name)
|
||||
log.debug("os.symlink(%s, %s)", input_path, soft_link_path)
|
||||
|
||||
# Create symbolic link using absolute path
|
||||
os.symlink(os.path.abspath(input_file), soft_link_name)
|
||||
soft_link_path.symlink_to(input_path.resolve())
|
||||
|
||||
|
||||
def samefile(file1: os.PathLike, file2: os.PathLike) -> bool:
|
||||
@@ -184,7 +188,7 @@ def samefile(file1: os.PathLike, file2: os.PathLike) -> bool:
|
||||
if os.name == 'nt':
|
||||
return file1 == file2
|
||||
else:
|
||||
return os.path.samefile(file1, file2)
|
||||
return Path(file1).samefile(file2)
|
||||
|
||||
|
||||
def is_iterable_notstr(thing: Any) -> bool:
|
||||
@@ -199,7 +203,7 @@ def monotonic(seq: Sequence) -> bool:
|
||||
|
||||
def page_number(input_file: os.PathLike) -> int:
|
||||
"""Get one-based page number implied by filename (000002.pdf -> 2)."""
|
||||
return int(os.path.basename(os.fspath(input_file))[0:6])
|
||||
return int(Path(input_file).name[0:6])
|
||||
|
||||
|
||||
def available_cpu_count() -> int:
|
||||
@@ -214,7 +218,7 @@ def available_cpu_count() -> int:
|
||||
return 1
|
||||
|
||||
|
||||
def is_file_writable(test_file: os.PathLike) -> bool:
|
||||
def is_file_writable(test_file: StrOrBytesPath) -> bool:
|
||||
"""Intentionally racy test if target is writable.
|
||||
|
||||
We intend to write to the output file if and only if we succeed and
|
||||
@@ -222,7 +226,7 @@ def is_file_writable(test_file: os.PathLike) -> bool:
|
||||
the location is writable.
|
||||
"""
|
||||
try:
|
||||
p = Path(test_file)
|
||||
p = Path(os.fsdecode(test_file))
|
||||
if p.is_symlink():
|
||||
p = p.resolve(strict=False)
|
||||
|
||||
@@ -329,6 +333,37 @@ def pikepdf_enable_mmap() -> None:
|
||||
log.debug("pikepdf mmap not available")
|
||||
|
||||
|
||||
def pikepdf_get_int(obj: pikepdf.Object, key: pikepdf.Name, default: int = 0) -> int:
|
||||
"""Look up a key on a pikepdf dictionary/stream, returning a plain int.
|
||||
|
||||
``.get(key, default)``'s return type is the ambiguous ``Object | int``,
|
||||
which does not support arithmetic or comparison against a plain int. In
|
||||
pikepdf's default (implicit) conversion mode, a PDF Integer is already
|
||||
unboxed to a native ``int`` by the time we see it here; under explicit
|
||||
conversion mode it would instead be a ``pikepdf.Object``. ``int()``
|
||||
handles both, since ``Object`` implements ``__int__``.
|
||||
"""
|
||||
value = obj.get(key)
|
||||
return int(value) if value is not None else default
|
||||
|
||||
|
||||
def pikepdf_get_bool(
|
||||
obj: pikepdf.Object, key: pikepdf.Name, default: bool = False
|
||||
) -> bool:
|
||||
"""Look up a key on a pikepdf dictionary/stream, returning a plain bool.
|
||||
|
||||
Unlike ``int()``/``float()``, ``bool()`` is not supported on
|
||||
``pikepdf.Object`` (it raises), so both conversion modes must be
|
||||
handled explicitly. See :func:`pikepdf_get_int` for background.
|
||||
"""
|
||||
value = obj.get(key)
|
||||
if value is None:
|
||||
return default
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
return value.as_bool(default)
|
||||
|
||||
|
||||
def running_in_docker() -> bool:
|
||||
"""Returns True if we seem to be running in a Docker container."""
|
||||
return Path('/.dockerenv').exists()
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Simple CLI for testing HOCR to PDF conversion using fpdf2 renderer."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
@@ -31,7 +32,7 @@ if __name__ == "__main__":
|
||||
'-i',
|
||||
'--image',
|
||||
default=None,
|
||||
help='Path to the image to be placed above the text (not yet supported)',
|
||||
help='Path to the image to overlay on top of the text layer',
|
||||
)
|
||||
parser.add_argument('hocrfile', help='Path to the hocr file to be parsed')
|
||||
parser.add_argument('outputfile', help='Path to the PDF file to be generated')
|
||||
@@ -58,17 +59,13 @@ if __name__ == "__main__":
|
||||
multi_font_manager = MultiFontManager(font_dir)
|
||||
|
||||
# Render to PDF using fpdf2
|
||||
image_path = Path(args.image) if args.image else None
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=ocr_page,
|
||||
dpi=dpi,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=not args.boundingboxes, # Visible text in debug mode
|
||||
invisible_text=bool(args.image),
|
||||
image=image_path,
|
||||
debug_render_options=debug_options,
|
||||
)
|
||||
renderer.render(Path(args.outputfile))
|
||||
|
||||
if args.image:
|
||||
print(
|
||||
f"Warning: Image overlay (--image {args.image}) is not yet supported "
|
||||
"with the fpdf2 renderer."
|
||||
)
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
Derived from
|
||||
https://www.loc.gov/standards/iso639-2/ascii_8bits.html
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import NamedTuple
|
||||
|
||||
+43
-19
@@ -12,7 +12,7 @@ import threading
|
||||
from collections.abc import Callable, Iterator, MutableSet, Sequence
|
||||
from os import fspath
|
||||
from pathlib import Path
|
||||
from typing import Any, NamedTuple, NewType
|
||||
from typing import Any, NamedTuple, NewType, cast
|
||||
from zlib import compress
|
||||
|
||||
import img2pdf
|
||||
@@ -37,7 +37,7 @@ from ocrmypdf._exec import ghostscript, jbig2enc, pngquant
|
||||
from ocrmypdf._jobcontext import PdfContext
|
||||
from ocrmypdf._progressbar import ProgressBar
|
||||
from ocrmypdf.exceptions import OutputFileAccessError
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, safe_symlink
|
||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, pikepdf_get_int, safe_symlink
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -260,10 +260,19 @@ def _find_image_xrefs_container(
|
||||
exclude_xrefs: MutableSet[Xref],
|
||||
pageno_for_xref: dict[Xref, int],
|
||||
depth: int = 0,
|
||||
visited_forms: MutableSet[Xref] | None = None,
|
||||
):
|
||||
"""Find all image XRefs or Form XObject and add to the include/exclude sets."""
|
||||
# Form XObjects are not added to include/exclude_xrefs, so the dedup
|
||||
# check below doesn't catch Form-XObject cycles or DAGs. Track them in
|
||||
# a shared set so each Form is only descended into once per document
|
||||
# (issue #1321).
|
||||
if visited_forms is None:
|
||||
visited_forms = set()
|
||||
if depth > 10:
|
||||
log.warning("Recursion depth exceeded in _find_image_xrefs_page")
|
||||
# With visited_forms memoization, this is a soft DAG-height guard
|
||||
# rather than a cycle defense, so a debug log is sufficient.
|
||||
log.debug("Recursion depth exceeded in _find_image_xrefs_page")
|
||||
return
|
||||
try:
|
||||
xobjs = container.Resources.XObject
|
||||
@@ -276,7 +285,9 @@ def _find_image_xrefs_container(
|
||||
if xref in include_xrefs or xref in exclude_xrefs:
|
||||
continue # Already processed
|
||||
if Name.Subtype in image and image.Subtype == Name.Form:
|
||||
# Recurse into Form XObjects
|
||||
if xref in visited_forms:
|
||||
continue
|
||||
visited_forms.add(xref)
|
||||
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
||||
_find_image_xrefs_container(
|
||||
pdf,
|
||||
@@ -286,6 +297,7 @@ def _find_image_xrefs_container(
|
||||
exclude_xrefs,
|
||||
pageno_for_xref,
|
||||
depth + 1,
|
||||
visited_forms,
|
||||
)
|
||||
continue
|
||||
if Name.SMask in image:
|
||||
@@ -342,9 +354,16 @@ def extract_images(
|
||||
pdf=pdf, root=root, image=image, xref=xref, options=options
|
||||
)
|
||||
except Exception: # pylint: disable=broad-except
|
||||
log.exception(
|
||||
f"xref {xref}: While extracting this image, an error occurred"
|
||||
# Optimization is best-effort: an image we cannot process is simply
|
||||
# left unchanged in the output, which remains valid. Report this as
|
||||
# a concise warning rather than an alarming traceback (issue #846);
|
||||
# the full detail is still available at debug verbosity.
|
||||
log.warning(
|
||||
f"xref {xref}: this image could not be processed by the "
|
||||
"optimizer and was left unchanged. The output file is still "
|
||||
"valid."
|
||||
)
|
||||
log.debug(f"xref {xref}: image optimization error detail", exc_info=True)
|
||||
errors += 1
|
||||
else:
|
||||
if result:
|
||||
@@ -430,12 +449,12 @@ def convert_to_jbig2(
|
||||
|
||||
|
||||
def _optimize_jpeg(
|
||||
xref: Xref, in_jpg: Path, opt_jpg: Path, jpg_quality: int
|
||||
xref: Xref, in_jpg: Path, opt_jpg: Path, jpeg_quality: int
|
||||
) -> tuple[Xref, Path | None]:
|
||||
with Image.open(in_jpg) as im:
|
||||
save_kwargs: dict[str, Any] = {'optimize': True}
|
||||
if isinstance(jpg_quality, int) and 0 < jpg_quality <= 100:
|
||||
save_kwargs['quality'] = jpg_quality
|
||||
if isinstance(jpeg_quality, int) and 0 < jpeg_quality <= 100:
|
||||
save_kwargs['quality'] = jpeg_quality
|
||||
im.save(opt_jpg, **save_kwargs)
|
||||
|
||||
if opt_jpg.stat().st_size > in_jpg.stat().st_size:
|
||||
@@ -454,7 +473,7 @@ def transcode_jpegs(
|
||||
for xref in jpegs:
|
||||
in_jpg = jpg_name(root, xref)
|
||||
opt_jpg = in_jpg.with_suffix('.opt.jpg')
|
||||
yield xref, in_jpg, opt_jpg, options.jpg_quality
|
||||
yield xref, in_jpg, opt_jpg, options.jpeg_quality
|
||||
|
||||
def finish_jpeg(result: tuple[Xref, Path | None], pbar: ProgressBar):
|
||||
xref, opt_jpg = result
|
||||
@@ -508,8 +527,8 @@ def _find_deflatable_jpeg(
|
||||
(
|
||||
# Don't flate very large images because it will slow down PDF viewers
|
||||
1 <= options.optimize <= 2
|
||||
and image.get(Name.Width, 0) < FLATE_JPEG_THRESHOLD
|
||||
and image.get(Name.Height, 0) < FLATE_JPEG_THRESHOLD
|
||||
and pikepdf_get_int(image, Name.Width) < FLATE_JPEG_THRESHOLD
|
||||
and pikepdf_get_int(image, Name.Height) < FLATE_JPEG_THRESHOLD
|
||||
)
|
||||
or options.optimize == 3
|
||||
)
|
||||
@@ -589,10 +608,13 @@ def _transcode_png(pdf: Pdf, filename: Path, xref: Xref) -> bool:
|
||||
local_image = pdf.copy_foreign(foreign_image)
|
||||
|
||||
im_obj = pdf.get_object(xref, 0)
|
||||
# pikepdf's Object attribute access can't statically know Filter/
|
||||
# DecodeParms hold these specific subtypes, but a copied image's
|
||||
# stream dictionary always does per the PDF spec.
|
||||
im_obj.write(
|
||||
local_image.read_raw_bytes(),
|
||||
filter=local_image.Filter,
|
||||
decode_parms=local_image.DecodeParms,
|
||||
filter=cast('Name | Array | list[Name] | None', local_image.Filter),
|
||||
decode_parms=cast('Dictionary | Array | None', local_image.DecodeParms),
|
||||
)
|
||||
|
||||
# Don't copy keys from the new image...
|
||||
@@ -681,9 +703,9 @@ def optimize(
|
||||
safe_symlink(input_file, output_file)
|
||||
return output_file
|
||||
|
||||
if options.jpg_quality == 0:
|
||||
options.jpg_quality = DEFAULT_JPEG_QUALITY if options.optimize < 3 else 40
|
||||
if options.png_quality == 0:
|
||||
if not options.jpeg_quality:
|
||||
options.jpeg_quality = DEFAULT_JPEG_QUALITY if options.optimize < 3 else 40
|
||||
if not options.png_quality:
|
||||
options.png_quality = DEFAULT_PNG_QUALITY if options.optimize < 3 else 30
|
||||
|
||||
with Pdf.open(input_file) as pdf:
|
||||
@@ -744,7 +766,7 @@ def main(infile, outfile, level, jobs=1):
|
||||
output_file=outfile, # Required field
|
||||
jobs=jobs,
|
||||
optimize=int(level),
|
||||
jpg_quality=0, # Use default
|
||||
jpeg_quality=0, # Use default
|
||||
png_quality=0,
|
||||
jbig2_threshold=0.85,
|
||||
quiet=True,
|
||||
@@ -752,7 +774,9 @@ def main(infile, outfile, level, jobs=1):
|
||||
)
|
||||
|
||||
with TemporaryDirectory() as tmpdir:
|
||||
context = PdfContext(options, Path(tmpdir), infile, None, None)
|
||||
# optimize() only reads context.options on this standalone path, so
|
||||
# pdfinfo and plugin_manager are not needed.
|
||||
context = PdfContext(options, Path(tmpdir), infile, None, None) # type: ignore[arg-type]
|
||||
tmpout = Path(tmpdir) / 'out.pdf'
|
||||
optimize(
|
||||
infile,
|
||||
|
||||
+74
-7
@@ -12,7 +12,7 @@ from importlib.resources import files as package_files
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
from pikepdf import Array, Dictionary, Name, Pdf, Stream
|
||||
from pikepdf import Array, Dictionary, Name, Object, Pdf, Stream
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -137,6 +137,71 @@ def file_claims_pdfa(filename: Path):
|
||||
return pdfa_dict
|
||||
|
||||
|
||||
def _cid_font_is_embedded(type0_font: Object) -> bool:
|
||||
"""Return True if a Type0 font's CID descendant carries embedded glyphs."""
|
||||
for descendant in type0_font.get(Name.DescendantFonts, []):
|
||||
descriptor = descendant.get(Name.FontDescriptor, None)
|
||||
# A malformed PDF may store a non-dictionary here; `key in descriptor`
|
||||
# raises on those, so require a real dictionary before probing it.
|
||||
if isinstance(descriptor, Dictionary) and any(
|
||||
key in descriptor for key in (Name.FontFile, Name.FontFile2, Name.FontFile3)
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def find_nonembedded_cid_fonts(pdf: Pdf) -> set[str]:
|
||||
"""Find CID-keyed (Type0) fonts that lack embedded glyph data.
|
||||
|
||||
PDF/A requires every font to be embedded. When Ghostscript converts a PDF
|
||||
to PDF/A it must substitute and embed a replacement for any non-embedded
|
||||
font. For CID-keyed fonts -- which is how CJK text is encoded, including the
|
||||
OCR text layers produced by Adobe Acrobat -- this substitution routinely
|
||||
corrupts the character-to-Unicode mapping, silently destroying the
|
||||
searchable text. Detecting these fonts lets the caller refuse PDF/A
|
||||
conversion rather than emit corrupted output.
|
||||
|
||||
Simple (non-CID) non-embedded fonts are not reported: Ghostscript
|
||||
substitutes standard encodings for them without corrupting the text, and
|
||||
they are far too common to treat as conversion blockers.
|
||||
|
||||
Args:
|
||||
pdf: An open ``pikepdf.Pdf`` to scan.
|
||||
|
||||
Returns:
|
||||
The set of ``BaseFont`` names of non-embedded CID fonts found.
|
||||
"""
|
||||
found: set[str] = set()
|
||||
|
||||
def scan_resources(resources, depth: int = 0) -> None:
|
||||
if resources is None or depth > 10:
|
||||
return
|
||||
# A well-formed PDF stores dictionaries under /Font and /XObject, but a
|
||||
# malformed one (common in OCR workloads) may store an array, a name, or
|
||||
# another non-dictionary object. Only such dictionaries have .values(),
|
||||
# so guard with isinstance rather than let the scan crash (issue #1713).
|
||||
fonts = resources.get(Name.Font, None)
|
||||
if isinstance(fonts, Dictionary):
|
||||
for font in fonts.as_dict().values():
|
||||
try:
|
||||
if font.get(Name.Subtype) != Name.Type0:
|
||||
continue
|
||||
if not _cid_font_is_embedded(font):
|
||||
basefont = str(font.get(Name.BaseFont, '/(unnamed)'))
|
||||
found.add(basefont.lstrip('/'))
|
||||
except (AttributeError, TypeError, KeyError):
|
||||
continue
|
||||
xobjects = resources.get(Name.XObject, None)
|
||||
if isinstance(xobjects, Dictionary):
|
||||
for xobj in xobjects.as_dict().values():
|
||||
if xobj.get(Name.Subtype) == Name.Form and Name.Resources in xobj:
|
||||
scan_resources(xobj[Name.Resources], depth + 1)
|
||||
|
||||
for page in pdf.pages:
|
||||
scan_resources(page.get(Name.Resources, None))
|
||||
return found
|
||||
|
||||
|
||||
def _load_srgb_icc_profile() -> bytes:
|
||||
"""Load the sRGB ICC profile from package data."""
|
||||
return (package_files('ocrmypdf.data') / SRGB_ICC_PROFILE_NAME).read_bytes()
|
||||
@@ -191,12 +256,14 @@ def add_srgb_output_intent(pdf: Pdf) -> None:
|
||||
icc_stream[Name.N] = 3 # RGB has 3 components
|
||||
|
||||
# Create OutputIntent dictionary
|
||||
output_intent = Dictionary({
|
||||
'/Type': Name.OutputIntent,
|
||||
'/S': Name('/GTS_PDFA1'),
|
||||
'/OutputConditionIdentifier': 'sRGB',
|
||||
'/DestOutputProfile': icc_stream,
|
||||
})
|
||||
output_intent = Dictionary(
|
||||
{
|
||||
'/Type': Name.OutputIntent,
|
||||
'/S': Name('/GTS_PDFA1'),
|
||||
'/OutputConditionIdentifier': 'sRGB',
|
||||
'/DestOutputProfile': icc_stream,
|
||||
}
|
||||
)
|
||||
|
||||
# Add to catalog's OutputIntents array
|
||||
if Name.OutputIntents not in pdf.Root:
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from ocrmypdf.pdfinfo._types import Colorspace, Encoding, FloatRect
|
||||
from ocrmypdf.pdfinfo._types import Colorspace, Encoding, FloatRect, Ink
|
||||
from ocrmypdf.pdfinfo.info import PageInfo, PdfInfo
|
||||
|
||||
__all__ = ["Colorspace", "Encoding", "FloatRect", "PageInfo", "PdfInfo"]
|
||||
__all__ = ["Colorspace", "Encoding", "FloatRect", "Ink", "PageInfo", "PdfInfo"]
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user