Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
250615561d | ||
|
|
a659f83d67 | ||
|
|
08f95c0b13 | ||
|
|
dbd3c93757 | ||
|
|
5d128a91d2 | ||
|
|
a1b8113d56 | ||
|
|
f052e910c9 | ||
|
|
116e2692d0 | ||
|
|
b2669c7d71 | ||
|
|
c8c53d38a3 | ||
|
|
d303b42c86 | ||
|
|
f77f701a50 | ||
|
|
1c3b7d1507 | ||
|
|
bf62562787 | ||
|
|
6c6cbfd4d6 | ||
|
|
ee5acbe94e | ||
|
|
5e478a7774 | ||
|
|
92c5200ad2 | ||
|
|
86a102f8e6 | ||
|
|
2463b91051 | ||
|
|
07f7c6b812 | ||
|
|
8138664287 | ||
|
|
120ca72393 | ||
|
|
f9b3e9a97b | ||
|
|
1e87930bbb | ||
|
|
fe4725658e | ||
|
|
9d042767cc | ||
|
|
23bc247b9c | ||
|
|
e44bf46d77 | ||
|
|
f50620c244 | ||
|
|
6f755321b8 | ||
|
|
706681deb8 | ||
|
|
c283cf0a0d | ||
|
|
0f82d7223e | ||
|
|
9a6150ae53 | ||
|
|
fec0948a13 | ||
|
|
18b59c57b4 | ||
|
|
a67a11e61c | ||
|
|
6ca4940a32 | ||
|
|
0e4cce2642 | ||
|
|
8fca0c71dc | ||
|
|
944d99bdc1 | ||
|
|
5bb6e1c5d7 | ||
|
|
8d7a8f0f98 | ||
|
|
b9dd0a5e3c | ||
|
|
6949ad2c5d | ||
|
|
b3324c3b4e | ||
|
|
b38cac6931 | ||
|
|
bb4c47e707 | ||
|
|
5e1e2497ab | ||
|
|
cd910fbf21 | ||
|
|
1225269a4b | ||
|
|
3a75b20740 | ||
|
|
d35d008806 | ||
|
|
f5662d5eb0 | ||
|
|
39010dd255 | ||
|
|
fbaad570c7 | ||
|
|
f974e3b3c1 | ||
|
|
46b49cc176 | ||
|
|
5256e74d0c | ||
|
|
621d6a0b89 | ||
|
|
08be7c8bbe | ||
|
|
980a5472b6 | ||
|
|
51c618e357 | ||
|
|
4dde3786c2 | ||
|
|
d544342602 | ||
|
|
fac91fca2a | ||
|
|
6edf756849 | ||
|
|
4fb1bb4de6 | ||
|
|
6a8eb7daaa | ||
|
|
0544d06c3d | ||
|
|
34c285c9ac | ||
|
|
2f53b27651 | ||
|
|
772677746b | ||
|
|
f0bad87ea6 | ||
|
|
44e71f8c14 | ||
|
|
964b30ca26 | ||
|
|
214a333e2d | ||
|
|
ec6401ab57 | ||
|
|
cbc5e8ce8d | ||
|
|
a1c4cfe8f1 | ||
|
|
3a721e6578 | ||
|
|
e6b716cdde | ||
|
|
02c39998b8 | ||
|
|
0774bc7f14 | ||
|
|
c6a98b3d0b | ||
|
|
981bbf1105 | ||
|
|
2b0c6cfd40 | ||
|
|
59f6bc8306 | ||
|
|
653c4ffb45 | ||
|
|
d947ca258e | ||
|
|
d5ff7f7db9 | ||
|
|
579cef3649 | ||
|
|
cb2f090c60 | ||
|
|
f3d6387bca | ||
|
|
abf9729c61 | ||
|
|
442e9c9f0d | ||
|
|
397fad249d | ||
|
|
9a3c5a3f7c | ||
|
|
950c700274 | ||
|
|
26432c38a9 | ||
|
|
28be50136c | ||
|
|
0c62f2de5d | ||
|
|
5caf654f22 | ||
|
|
205593445e | ||
|
|
f25fb8c63a | ||
|
|
99c78650b6 | ||
|
|
69355886a8 | ||
|
|
08e89e2dbe | ||
|
|
0e013df161 | ||
|
|
9ba4e3ab46 | ||
|
|
5fdcb7602b | ||
|
|
b4db1b741f | ||
|
|
7a8cc21e31 | ||
|
|
0674829d8f | ||
|
|
315aa0474b | ||
|
|
df3451e779 | ||
|
|
3ba42802d1 | ||
|
|
d6342cb8c2 | ||
|
|
065bddbc6c | ||
|
|
067f429dde | ||
|
|
6895c2d70f | ||
|
|
686481982a | ||
|
|
a9e1d19b78 | ||
|
|
f95aa63718 | ||
|
|
855de287b2 | ||
|
|
feeb9f213f | ||
|
|
e7eb8fa805 | ||
|
|
8a747f005a | ||
|
|
16ab4a8b4e | ||
|
|
8d30cff4ef | ||
|
|
59d5b0d1bd | ||
|
|
9ec0745ab8 | ||
|
|
3a3635f7f9 | ||
|
|
6a746a1cbb | ||
|
|
906c130f96 | ||
|
|
4a78458821 | ||
|
|
fddf3ce2f4 | ||
|
|
353b34e695 | ||
|
|
7d63355c3c | ||
|
|
42ff7fc842 | ||
|
|
26470fe16a | ||
|
|
3b9d4b7f0a | ||
|
|
11f53fe9a9 | ||
|
|
123c0c766f | ||
|
|
6a9be2142e | ||
|
|
0bc350f55e | ||
|
|
7a6edf62ba | ||
|
|
07b6f06f11 | ||
|
|
2005f622bb | ||
|
|
cca04fd799 | ||
|
|
75bf8e4ba2 | ||
|
|
daabb5b100 | ||
|
|
035ebea72f | ||
|
|
a499956462 | ||
|
|
74d2a156c4 | ||
|
|
f87fc7b12d | ||
|
|
602f5632cb | ||
|
|
9fbbcf7599 | ||
|
|
9498f01f59 | ||
|
|
2c59aca5a1 | ||
|
|
51301d69c9 | ||
|
|
7e608fd1df | ||
|
|
ecc79315df | ||
|
|
14365d10b8 | ||
|
|
5e5320020f | ||
|
|
103c3e0cd6 | ||
|
|
7a1c89edd9 | ||
|
|
a5ff3d2f42 | ||
|
|
b71d16dd96 | ||
|
|
fd593eb5e9 | ||
|
|
a0b98abb94 | ||
|
|
18353e1e94 | ||
|
|
9adcad84da | ||
|
|
f2714586d8 | ||
|
|
0b6fb62967 | ||
|
|
1db8b0b943 | ||
|
|
f38aebb3d5 | ||
|
|
7162c36d37 | ||
|
|
f4d4ea46c8 | ||
|
|
2fd1a0f178 | ||
|
|
73ed33a086 | ||
|
|
e6095a9949 | ||
|
|
16f05af401 | ||
|
|
1631afc878 | ||
|
|
63d87fc440 | ||
|
|
9489c01259 | ||
|
|
30d92ad83f | ||
|
|
a4987733c4 | ||
|
|
39eee05230 | ||
|
|
5b2f2e6290 | ||
|
|
445617a1a5 | ||
|
|
f6e90a5934 | ||
|
|
43618e6b3f | ||
|
|
e97f89de3b | ||
|
|
11d3e32f1e | ||
|
|
2affa83efe | ||
|
|
c90d5cd84b | ||
|
|
aacaba3d26 | ||
|
|
fec53be841 | ||
|
|
3f7b540f76 | ||
|
|
d217856166 | ||
|
|
e2be457e9b | ||
|
|
4850f486d2 | ||
|
|
729c7febd9 | ||
|
|
6c6aca2f1e | ||
|
|
c69823f496 | ||
|
|
73f8f6aac8 | ||
|
|
d944254e45 | ||
|
|
f7ddffe554 | ||
|
|
8a73ed5d5a | ||
|
|
03669183d7 | ||
|
|
74e101a2fa | ||
|
|
532cf18ad3 | ||
|
|
0b90b697e2 | ||
|
|
6be7c5f7c8 | ||
|
|
db2e5132e6 | ||
|
|
b14f6f778a | ||
|
|
415de77457 | ||
|
|
a9466c4f58 | ||
|
|
d9ae453a63 | ||
|
|
9841e09233 | ||
|
|
0ca314e066 | ||
|
|
d7680cae27 | ||
|
|
491b6bdb1f | ||
|
|
c591f9601a | ||
|
|
8d1e75017e | ||
|
|
94615f7ad4 | ||
|
|
e5df8e1315 | ||
|
|
d739b91aef | ||
|
|
686cfb2539 | ||
|
|
2633716bb7 | ||
|
|
0a07c0a44e | ||
|
|
2ca6e110ca | ||
|
|
334a07c839 | ||
|
|
a57c39358d | ||
|
|
30a0c315fb | ||
|
|
b860f0d94c | ||
|
|
14f4c19f5a | ||
|
|
7ab5c55d46 | ||
|
|
8b6ecd5971 | ||
|
|
7b0871ae4c | ||
|
|
b73af7ce10 | ||
|
|
60645717e2 | ||
|
|
1cbf578538 | ||
|
|
e966c1fceb | ||
|
|
d0133f8641 | ||
|
|
6d30b497dc | ||
|
|
f3b89e66eb | ||
|
|
04154e207c | ||
|
|
9898904be7 | ||
|
|
27d5229842 | ||
|
|
4a9a575ef0 | ||
|
|
52fd9a630d | ||
|
|
a596ccf844 | ||
|
|
e7fa97731f | ||
|
|
290aa28108 | ||
|
|
a95640ed9e | ||
|
|
f69267bb67 | ||
|
|
e36d5a309f | ||
|
|
55566d9830 | ||
|
|
f02ea20678 | ||
|
|
372c22d42b | ||
|
|
949265bbd0 |
+22
-26
@@ -1,7 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
FROM ubuntu:22.04 as base
|
FROM ubuntu:22.04 AS base
|
||||||
|
|
||||||
ENV LANG=C.UTF-8
|
ENV LANG=C.UTF-8
|
||||||
ENV TZ=UTC
|
ENV TZ=UTC
|
||||||
@@ -9,19 +9,15 @@ RUN echo 'debconf debconf/frontend select Noninteractive' | debconf-set-selectio
|
|||||||
|
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
python3 \
|
python3 \
|
||||||
libqpdf-dev \
|
python-is-python3
|
||||||
zlib1g \
|
|
||||||
liblept5
|
|
||||||
|
|
||||||
FROM base as builder
|
FROM base AS builder
|
||||||
|
|
||||||
# Note we need leptonica here to build jbig2
|
# Note we need leptonica here to build jbig2
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
build-essential autoconf automake libtool \
|
build-essential autoconf automake libtool \
|
||||||
libleptonica-dev \
|
libleptonica-dev \
|
||||||
zlib1g-dev \
|
zlib1g-dev \
|
||||||
python3-dev \
|
|
||||||
python3-distutils \
|
|
||||||
libffi-dev \
|
libffi-dev \
|
||||||
ca-certificates \
|
ca-certificates \
|
||||||
curl \
|
curl \
|
||||||
@@ -29,15 +25,11 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
|||||||
libcairo2-dev \
|
libcairo2-dev \
|
||||||
pkg-config
|
pkg-config
|
||||||
|
|
||||||
# Get the latest pip (Ubuntu version doesn't support manylinux2010)
|
|
||||||
RUN \
|
|
||||||
curl https://bootstrap.pypa.io/get-pip.py | python3
|
|
||||||
|
|
||||||
# Compile and install jbig2
|
# Compile and install jbig2
|
||||||
# Needs libleptonica-dev, zlib1g-dev
|
# Needs libleptonica-dev, zlib1g-dev
|
||||||
RUN \
|
RUN \
|
||||||
mkdir jbig2 \
|
mkdir jbig2 \
|
||||||
&& curl -L https://github.com/agl/jbig2enc/archive/ea6a40a.tar.gz | \
|
&& curl -L https://github.com/agl/jbig2enc/archive/c0141bf.tar.gz | \
|
||||||
tar xz -C jbig2 --strip-components=1 \
|
tar xz -C jbig2 --strip-components=1 \
|
||||||
&& cd jbig2 \
|
&& cd jbig2 \
|
||||||
&& ./autogen.sh && ./configure && make && make install \
|
&& ./autogen.sh && ./configure && make && make install \
|
||||||
@@ -48,21 +40,23 @@ COPY . /app
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
RUN pip3 install --no-cache-dir .[test,webservice,watcher]
|
RUN curl -LsSf https://astral.sh/uv/0.4.27/install.sh | sh
|
||||||
|
|
||||||
|
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||||
|
|
||||||
|
# Instead of restarting the shell, use uv directly from its installed location.
|
||||||
|
RUN /root/.cargo/bin/uv sync --extra test --extra webservice --extra watcher
|
||||||
|
|
||||||
FROM base
|
FROM base
|
||||||
|
|
||||||
# For Tesseract 5
|
RUN apt-get update && apt-get install -y software-properties-common
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
|
||||||
software-properties-common gpg-agent
|
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr5
|
||||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr-devel
|
|
||||||
|
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
fonts-droid-fallback \
|
fonts-droid-fallback \
|
||||||
jbig2dec \
|
jbig2dec \
|
||||||
img2pdf \
|
|
||||||
libsm6 libxext6 libxrender-dev \
|
|
||||||
pngquant \
|
pngquant \
|
||||||
tesseract-ocr \
|
tesseract-ocr \
|
||||||
tesseract-ocr-chi-sim \
|
tesseract-ocr-chi-sim \
|
||||||
@@ -79,11 +73,13 @@ WORKDIR /app
|
|||||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
||||||
COPY --from=builder /usr/local/bin/ /usr/local/bin/
|
COPY --from=builder /usr/local/bin/ /usr/local/bin/
|
||||||
|
|
||||||
COPY --from=builder /app/misc/webservice.py /app/
|
COPY --from=builder --chown=app:app /app /app
|
||||||
COPY --from=builder /app/misc/watcher.py /app/
|
|
||||||
|
|
||||||
# Copy minimal project files to get the test suite.
|
RUN rm -rf /app/.git && \
|
||||||
COPY --from=builder /app/pyproject.toml /app/README.md /app/
|
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||||
COPY --from=builder /app/tests /app/tests
|
ln -s /app/misc/watcher.py /app/watcher.py
|
||||||
|
|
||||||
|
ENV PATH="/app/.venv/bin:${PATH}"
|
||||||
|
|
||||||
|
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||||
|
|
||||||
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
|
||||||
|
|||||||
+19
-34
@@ -1,7 +1,14 @@
|
|||||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
FROM alpine:3.18 as base
|
# Note: Alpine 3.20 builds tesseract with --enable-opencl, which is not
|
||||||
|
# supported by anyone. OCRmyPDF is not compatible with Alpine 3.20.0
|
||||||
|
# through 3.20.3. The Alpine issue should be fixed in 3.21.0. It is
|
||||||
|
# not clear if 3.20.4+ will have the fix.
|
||||||
|
# Details
|
||||||
|
# https://gitlab.alpinelinux.org/alpine/aports/-/issues/16143
|
||||||
|
# https://github.com/ocrmypdf/OCRmyPDF/issues/1395
|
||||||
|
FROM alpine:3.19 AS base
|
||||||
|
|
||||||
ENV LANG=C.UTF-8
|
ENV LANG=C.UTF-8
|
||||||
ENV TZ=UTC
|
ENV TZ=UTC
|
||||||
@@ -10,40 +17,24 @@ RUN apk add --no-cache \
|
|||||||
python3 \
|
python3 \
|
||||||
zlib
|
zlib
|
||||||
|
|
||||||
FROM base as builder
|
FROM base AS builder
|
||||||
|
|
||||||
RUN apk add --no-cache \
|
RUN apk add --no-cache \
|
||||||
ca-certificates \
|
ca-certificates \
|
||||||
git \
|
git \
|
||||||
python3-dev \
|
python3-dev \
|
||||||
py3-pip
|
curl
|
||||||
|
|
||||||
# On arm64, we need to build cffi from source.
|
|
||||||
ARG TARGETPLATFORM
|
|
||||||
|
|
||||||
RUN if [ "${TARGETPLATFORM}" == "linux/arm64" ]; then \
|
|
||||||
apk add --no-cache \
|
|
||||||
build-base \
|
|
||||||
autoconf \
|
|
||||||
automake \
|
|
||||||
libtool \
|
|
||||||
zlib-dev \
|
|
||||||
libffi-dev \
|
|
||||||
cairo-dev \
|
|
||||||
pkgconfig \
|
|
||||||
; \
|
|
||||||
fi
|
|
||||||
|
|
||||||
COPY . /app
|
COPY . /app
|
||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
RUN python3 -m venv .venv
|
RUN curl -LsSf https://astral.sh/uv/0.4.27/install.sh | sh
|
||||||
|
|
||||||
RUN source .venv/bin/activate \
|
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||||
&& python3 -m pip install --no-cache-dir --upgrade pip \
|
|
||||||
&& python3 -m pip install --no-cache-dir wheel \
|
# Instead of restarting the shell, use uv directly from its installed location.
|
||||||
&& python3 -m pip install --no-cache-dir .[test,webservice,watcher]
|
RUN /root/.cargo/bin/uv sync --extra test --extra webservice --extra watcher
|
||||||
|
|
||||||
FROM base
|
FROM base
|
||||||
|
|
||||||
@@ -66,17 +57,11 @@ RUN apk add --no-cache \
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
COPY --from=builder --chown=app:app /app /app
|
||||||
COPY --from=builder /usr/local/bin/ /usr/local/bin/
|
|
||||||
|
|
||||||
COPY --from=builder /app/.venv/ /app/.venv/
|
RUN rm -rf /app/.git && \
|
||||||
|
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||||
COPY --from=builder /app/misc/webservice.py /app/
|
ln -s /app/misc/watcher.py /app/watcher.py
|
||||||
COPY --from=builder /app/misc/watcher.py /app/
|
|
||||||
|
|
||||||
# Copy minimal project files to get the test suite.
|
|
||||||
COPY --from=builder /app/pyproject.toml /app/README.md /app/
|
|
||||||
COPY --from=builder /app/tests /app/tests
|
|
||||||
|
|
||||||
ENV PATH="/app/.venv/bin:${PATH}"
|
ENV PATH="/app/.venv/bin:${PATH}"
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
name: General issues
|
name: Installation, packaging, dependencies
|
||||||
description: Installation, packages, dependencies, "nothing works", test suite failures...
|
description: Installation, packages, dependencies, "nothing works", test suite failures...
|
||||||
title: "[Bug]: "
|
title: "[Bug]: "
|
||||||
labels: ["bug", "triage"]
|
labels: ["triage"]
|
||||||
assignees:
|
assignees:
|
||||||
- jbarlow83
|
- jbarlow83
|
||||||
body:
|
body:
|
||||||
@@ -9,6 +9,10 @@ body:
|
|||||||
attributes:
|
attributes:
|
||||||
value: |
|
value: |
|
||||||
Thanks for taking the time to fill out this bug report!
|
Thanks for taking the time to fill out this bug report!
|
||||||
|
|
||||||
|
If your issue involves using OCRmyPDF on specific file(s) and not getting
|
||||||
|
good results, this is the *wrong* issue template. Please use the recommended
|
||||||
|
template to ensure we have enough information to help.
|
||||||
- type: textarea
|
- type: textarea
|
||||||
id: what-happened
|
id: what-happened
|
||||||
attributes:
|
attributes:
|
||||||
@@ -20,7 +24,7 @@ body:
|
|||||||
- type: dropdown
|
- type: dropdown
|
||||||
id: packaging-system
|
id: packaging-system
|
||||||
attributes:
|
attributes:
|
||||||
label: Where are you installing from?
|
label: Where are you installing/running from?
|
||||||
multiple: true
|
multiple: true
|
||||||
options:
|
options:
|
||||||
- PyPI (pip, poetry, pipx, etc.)
|
- PyPI (pip, poetry, pipx, etc.)
|
||||||
@@ -33,6 +37,11 @@ body:
|
|||||||
- source build
|
- source build
|
||||||
validations:
|
validations:
|
||||||
required: true
|
required: true
|
||||||
|
- type: input
|
||||||
|
id: version
|
||||||
|
attributes:
|
||||||
|
label: OCRmyPDF version
|
||||||
|
description: Paste "ocrmypdf --version" here
|
||||||
- type: dropdown
|
- type: dropdown
|
||||||
id: operating-system
|
id: operating-system
|
||||||
attributes:
|
attributes:
|
||||||
@@ -43,6 +52,18 @@ body:
|
|||||||
- Windows
|
- Windows
|
||||||
- macOS
|
- macOS
|
||||||
- BSD
|
- BSD
|
||||||
|
- type: input
|
||||||
|
id: os_version
|
||||||
|
attributes:
|
||||||
|
label: Operating system details and version
|
||||||
|
- type: checkboxes
|
||||||
|
attributes:
|
||||||
|
label: Simple sanity checks
|
||||||
|
description: Select all that apply
|
||||||
|
options:
|
||||||
|
- label: Operating system is currently supported by its vendor (not end of life)
|
||||||
|
- label: Python version is compatible with OCRmyPDF
|
||||||
|
- label: This issue is not about a specific input file
|
||||||
- type: textarea
|
- type: textarea
|
||||||
id: logs
|
id: logs
|
||||||
attributes:
|
attributes:
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
name: Problem with specific file
|
name: Problem with specific file
|
||||||
description: Something went wrong while trying to OCR a specific file
|
description: Something went wrong while trying to OCR a specific file
|
||||||
title: "[Bug]: "
|
title: "[Bug]: "
|
||||||
labels: ["bug", "triage"]
|
labels: ["triage"]
|
||||||
assignees:
|
assignees:
|
||||||
- jbarlow83
|
- jbarlow83
|
||||||
body:
|
body:
|
||||||
@@ -21,7 +21,7 @@ body:
|
|||||||
id: reproduce
|
id: reproduce
|
||||||
attributes:
|
attributes:
|
||||||
label: Steps to reproduce
|
label: Steps to reproduce
|
||||||
description: Please include steps to reproduce
|
description: Please include steps to reproduce.
|
||||||
value: |
|
value: |
|
||||||
1. Run ocrmypdf -v1 ...arguments... input.pdf output.pdf
|
1. Run ocrmypdf -v1 ...arguments... input.pdf output.pdf
|
||||||
2. Open output.pdf
|
2. Open output.pdf
|
||||||
@@ -31,8 +31,21 @@ body:
|
|||||||
id: files
|
id: files
|
||||||
attributes:
|
attributes:
|
||||||
label: Files
|
label: Files
|
||||||
description: Please attach the input and output files, or any screenshots that may be helpful.
|
description: |
|
||||||
placeholder: Drag and drop files here
|
Please attach the input and output files, or any screenshots that may be helpful.
|
||||||
|
|
||||||
|
If you cannot provide a test file, we probably won't be able to help with the issue.
|
||||||
|
PDF is a complex file format, and there may be technical details in the PDF that are
|
||||||
|
causing the issue. There's really no substitute for a test file.
|
||||||
|
|
||||||
|
We understand files may contain personal or sensitive information. Here are some options:
|
||||||
|
- Try reproducing the issue with a file from the OCRmyPDF test suite. (See tests/resources)
|
||||||
|
- Try to create another file in the same way as your private file.
|
||||||
|
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
||||||
|
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
||||||
|
omits personal information.
|
||||||
|
placeholder: |
|
||||||
|
Drag and drop files here.
|
||||||
- type: dropdown
|
- type: dropdown
|
||||||
id: packaging-system
|
id: packaging-system
|
||||||
attributes:
|
attributes:
|
||||||
|
|||||||
@@ -0,0 +1,83 @@
|
|||||||
|
name: Problem with third party app that uses OCRmyPDF
|
||||||
|
description: |
|
||||||
|
For PDF generation issues with third party software such as Paperless-ngx that
|
||||||
|
uses OCRmyPDF to perform OCR or generate PDFs.
|
||||||
|
title: "[3rdparty]: "
|
||||||
|
labels: ["triage"]
|
||||||
|
assignees:
|
||||||
|
- jbarlow83
|
||||||
|
body:
|
||||||
|
- type: markdown
|
||||||
|
attributes:
|
||||||
|
value: |
|
||||||
|
Thanks for taking the time to describe this issue with a particular file
|
||||||
|
and third party app.
|
||||||
|
|
||||||
|
If you are comfortable using OCRmyPDF, please trying to install OCRmyPDF,
|
||||||
|
run it on your file, and see if it works. It's easier for everyone
|
||||||
|
if you can confirm that the issue occurs with OCRmyPDF and not with
|
||||||
|
the third party app.
|
||||||
|
- type: checkboxes
|
||||||
|
attributes:
|
||||||
|
label: Simple sanity checks
|
||||||
|
description: Select all that apply
|
||||||
|
options:
|
||||||
|
- label: This is an issue with an app that uses OCRmyPDF for OCR
|
||||||
|
- label: I am using a recent version of the third party app
|
||||||
|
- label: I will include a file that reproduces the issuse
|
||||||
|
- type: input
|
||||||
|
id: thirdparty-app-name-version
|
||||||
|
attributes:
|
||||||
|
label: Third party app name and version
|
||||||
|
description: e.g. Paperless-ngx 2.9.0
|
||||||
|
- type: textarea
|
||||||
|
id: what-happened
|
||||||
|
attributes:
|
||||||
|
label: Describe the bug
|
||||||
|
description: A clear and concise description of what the bug is.
|
||||||
|
placeholder: Tell us what you see!
|
||||||
|
validations:
|
||||||
|
required: true
|
||||||
|
- type: textarea
|
||||||
|
id: reproduce
|
||||||
|
attributes:
|
||||||
|
label: Steps to reproduce
|
||||||
|
description: Please include steps to reproduce.
|
||||||
|
value: |
|
||||||
|
1. Import attached file into Paperless-ngx
|
||||||
|
2. Trigger OCR
|
||||||
|
3. Check log file
|
||||||
|
4. ...
|
||||||
|
render: plain text
|
||||||
|
- type: textarea
|
||||||
|
id: files
|
||||||
|
attributes:
|
||||||
|
label: Files
|
||||||
|
description: |
|
||||||
|
Please attach the input and output files, or any screenshots that may be helpful.
|
||||||
|
|
||||||
|
If you cannot provide a test file, we probably won't be able to help with the issue.
|
||||||
|
PDF is a complex file format, and there may be technical details in the PDF that are
|
||||||
|
causing the issue. There's really no substitute for a test file.
|
||||||
|
|
||||||
|
We understand files may contain personal or sensitive information. Here are some options:
|
||||||
|
- Try reproducing the issue with a file from the test suite. (See tests/resources)
|
||||||
|
- Try to create another file in the same way as your private file.
|
||||||
|
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
||||||
|
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
||||||
|
omits personal information.
|
||||||
|
placeholder: |
|
||||||
|
Drag and drop files here.
|
||||||
|
- type: input
|
||||||
|
id: version
|
||||||
|
attributes:
|
||||||
|
label: OCRmyPDF version
|
||||||
|
description: Paste "ocrmypdf --version" here
|
||||||
|
placeholder: ocrmypdf --version
|
||||||
|
- type: textarea
|
||||||
|
id: logs
|
||||||
|
attributes:
|
||||||
|
label: Relevant log output
|
||||||
|
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||||
|
placeholder: Run OCRmyPDF with verbosity `-v1` to get more detailed logging output.
|
||||||
|
render: plain text
|
||||||
+86
-66
@@ -21,21 +21,14 @@ jobs:
|
|||||||
runs-on: ${{ matrix.os }}
|
runs-on: ${{ matrix.os }}
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
|
os: [ubuntu-22.04, ubuntu-24.04]
|
||||||
|
python: ["3.10", "3.11", "3.12", "3.13"]
|
||||||
include:
|
include:
|
||||||
- os: ubuntu-22.04
|
- os: ubuntu-22.04
|
||||||
python: "3.9"
|
tesseract_ppa: "ppa"
|
||||||
- os: ubuntu-22.04
|
|
||||||
python: "3.10"
|
python: "3.10"
|
||||||
- os: ubuntu-22.04
|
- os: ubuntu-24.04
|
||||||
python: "3.11"
|
python: "pypy3.10"
|
||||||
- os: ubuntu-22.04
|
|
||||||
python: "3.9"
|
|
||||||
tesseract5: true
|
|
||||||
- os: ubuntu-latest
|
|
||||||
python: "3.12"
|
|
||||||
tesseract5: true
|
|
||||||
#- os: ubuntu-latest
|
|
||||||
# python: "pypy3.9"
|
|
||||||
|
|
||||||
env:
|
env:
|
||||||
OS: ${{ matrix.os }}
|
OS: ${{ matrix.os }}
|
||||||
@@ -46,16 +39,20 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v4
|
- name: Install uv
|
||||||
name: Setup Python
|
uses: astral-sh/setup-uv@v3
|
||||||
|
with:
|
||||||
|
version: "0.4.27"
|
||||||
|
|
||||||
|
- name: "Set up Python"
|
||||||
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python }}
|
python-version: ${{ matrix.python }}
|
||||||
cache: "pip"
|
|
||||||
|
|
||||||
- name: Install Tesseract 5
|
- name: Install Tesseract from PPA
|
||||||
if: matrix.tesseract5
|
if: matrix.tesseract_ppa == 'ppa'
|
||||||
run: |
|
run: |
|
||||||
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr-devel
|
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr5.3
|
||||||
|
|
||||||
- name: Install common packages
|
- name: Install common packages
|
||||||
run: |
|
run: |
|
||||||
@@ -63,6 +60,7 @@ jobs:
|
|||||||
sudo apt-get install -y --no-install-recommends \
|
sudo apt-get install -y --no-install-recommends \
|
||||||
curl \
|
curl \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
|
jbig2dec \
|
||||||
img2pdf \
|
img2pdf \
|
||||||
libexempi8 \
|
libexempi8 \
|
||||||
libffi-dev \
|
libffi-dev \
|
||||||
@@ -86,8 +84,7 @@ jobs:
|
|||||||
|
|
||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip wheel
|
uv sync --extra test
|
||||||
python -m pip install --prefer-binary .[test]
|
|
||||||
|
|
||||||
- name: Report versions
|
- name: Report versions
|
||||||
run: |
|
run: |
|
||||||
@@ -95,14 +92,16 @@ jobs:
|
|||||||
gs --version
|
gs --version
|
||||||
pngquant --version
|
pngquant --version
|
||||||
unpaper --version
|
unpaper --version
|
||||||
img2pdf --version
|
uv run img2pdf --version
|
||||||
|
|
||||||
- name: Test
|
- name: Test
|
||||||
run: |
|
run: |
|
||||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
uv run pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v3
|
uses: codecov/codecov-action@v4
|
||||||
|
env:
|
||||||
|
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
@@ -112,8 +111,8 @@ jobs:
|
|||||||
runs-on: ${{ matrix.os }}
|
runs-on: ${{ matrix.os }}
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
os: [macos-latest]
|
os: [macos-latest, macos-13] # macos-latest is arm64, macos-13 is x86_64
|
||||||
python: ["3.10", "3.11", "3.12"]
|
python: ["3.10", "3.11", "3.12", "3.13"]
|
||||||
|
|
||||||
env:
|
env:
|
||||||
OS: ${{ matrix.os }}
|
OS: ${{ matrix.os }}
|
||||||
@@ -133,34 +132,38 @@ jobs:
|
|||||||
ghostscript \
|
ghostscript \
|
||||||
jbig2enc \
|
jbig2enc \
|
||||||
openjpeg \
|
openjpeg \
|
||||||
openssl \
|
|
||||||
pngquant \
|
pngquant \
|
||||||
tesseract
|
tesseract
|
||||||
|
|
||||||
- uses: actions/setup-python@v4
|
- name: Install uv
|
||||||
name: Setup Python
|
uses: astral-sh/setup-uv@v3
|
||||||
|
with:
|
||||||
|
version: "0.4.27"
|
||||||
|
|
||||||
|
- name: "Set up Python"
|
||||||
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python }}
|
python-version: ${{ matrix.python }}
|
||||||
cache: "pip"
|
|
||||||
|
|
||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip wheel
|
uv sync --extra test
|
||||||
python -m pip install --prefer-binary .[test]
|
|
||||||
|
|
||||||
- name: Report versions
|
- name: Report versions
|
||||||
run: |
|
run: |
|
||||||
tesseract --version
|
tesseract --version
|
||||||
gs --version
|
gs --version
|
||||||
pngquant --version
|
pngquant --version
|
||||||
img2pdf --version
|
uv run img2pdf --version
|
||||||
|
|
||||||
- name: Test
|
- name: Test
|
||||||
run: |
|
run: |
|
||||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
uv run pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v3
|
uses: codecov/codecov-action@v4
|
||||||
|
env:
|
||||||
|
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
@@ -171,7 +174,7 @@ jobs:
|
|||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
os: [windows-latest]
|
os: [windows-latest]
|
||||||
python: ["3.10", "3.11", "3.12"]
|
python: ["3.10", "3.11", "3.12", "3.13"]
|
||||||
|
|
||||||
env:
|
env:
|
||||||
OS: ${{ matrix.os }}
|
OS: ${{ matrix.os }}
|
||||||
@@ -182,28 +185,33 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v4
|
- name: Install uv
|
||||||
name: Setup Python
|
uses: astral-sh/setup-uv@v3
|
||||||
|
with:
|
||||||
|
version: "0.4.27"
|
||||||
|
|
||||||
|
- name: "Set up Python"
|
||||||
|
uses: actions/setup-python@v5
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python }}
|
python-version: ${{ matrix.python }}
|
||||||
cache: "pip"
|
|
||||||
|
|
||||||
- name: Install system packages
|
- name: Install system packages
|
||||||
run: |
|
run: |
|
||||||
choco install --yes --no-progress --pre tesseract
|
choco install --yes --no-progress --pre tesseract
|
||||||
choco install --yes --no-progress --ignore-checksums ghostscript
|
choco install --yes --no-progress --ignore-checksums ghostscript --version 9.56.1
|
||||||
|
|
||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip wheel
|
uv sync --extra test
|
||||||
python -m pip install --prefer-binary .[test]
|
|
||||||
|
|
||||||
- name: Test
|
- name: Test
|
||||||
run: |
|
run: |
|
||||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
uv run pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v3
|
uses: codecov/codecov-action@v4
|
||||||
|
env:
|
||||||
|
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
@@ -216,19 +224,18 @@ jobs:
|
|||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v4
|
- name: Install uv
|
||||||
name: Setup Python
|
uses: astral-sh/setup-uv@v3
|
||||||
with:
|
with:
|
||||||
python-version: "3.9"
|
version: "0.4.27"
|
||||||
cache: "pip"
|
|
||||||
|
|
||||||
- name: Make wheels and sdist
|
- name: Make wheels and sdist
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip wheel build
|
uv build --sdist --wheel
|
||||||
python -m build --sdist --wheel
|
|
||||||
|
|
||||||
- uses: actions/upload-artifact@v3
|
- uses: actions/upload-artifact@v4
|
||||||
with:
|
with:
|
||||||
|
name: artifact
|
||||||
path: |
|
path: |
|
||||||
./dist/*.whl
|
./dist/*.whl
|
||||||
./dist/*.tar.gz
|
./dist/*.tar.gz
|
||||||
@@ -242,7 +249,7 @@ jobs:
|
|||||||
id-token: write # mandatory for PyPI publishing
|
id-token: write # mandatory for PyPI publishing
|
||||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/download-artifact@v3
|
- uses: actions/download-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: artifact
|
name: artifact
|
||||||
path: dist
|
path: dist
|
||||||
@@ -252,29 +259,45 @@ jobs:
|
|||||||
|
|
||||||
create_release:
|
create_release:
|
||||||
name: Create GitHub release
|
name: Create GitHub release
|
||||||
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
needs: [upload_pypi]
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||||
permissions:
|
permissions:
|
||||||
# Required to create a release
|
# Required to create a release
|
||||||
contents: write
|
contents: write
|
||||||
|
id-token: write
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/download-artifact@v3
|
- uses: actions/download-artifact@v4
|
||||||
with:
|
with:
|
||||||
name: artifact
|
name: artifact
|
||||||
path: dist
|
path: dist
|
||||||
|
|
||||||
- name: Create Release
|
- name: Sign the dists with Sigstore
|
||||||
id: create-release
|
uses: sigstore/gh-action-sigstore-python@v3.0.0
|
||||||
uses: shogo82148/actions-create-release@v1
|
|
||||||
|
|
||||||
- name: Upload Assets
|
|
||||||
uses: shogo82148/actions-upload-release-asset@v1
|
|
||||||
with:
|
with:
|
||||||
upload_url: ${{ steps.create-release.outputs.upload_url }}
|
inputs: >-
|
||||||
asset_path: |
|
|
||||||
./dist/*.whl
|
|
||||||
./dist/*.tar.gz
|
./dist/*.tar.gz
|
||||||
|
./dist/*.whl
|
||||||
|
|
||||||
|
- name: Create GitHub Release
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ github.token }}
|
||||||
|
run: >-
|
||||||
|
gh release create
|
||||||
|
'${{ github.ref_name }}'
|
||||||
|
--repo '${{ github.repository }}'
|
||||||
|
--notes ""
|
||||||
|
|
||||||
|
- name: Upload artifact signatures to GitHub Release
|
||||||
|
env:
|
||||||
|
GITHUB_TOKEN: ${{ github.token }}
|
||||||
|
# Upload to GitHub Release using the `gh` CLI.
|
||||||
|
# `dist/` contains the built packages, and the
|
||||||
|
# sigstore-produced signatures and certificates.
|
||||||
|
run: >-
|
||||||
|
gh release upload
|
||||||
|
'${{ github.ref_name }}' dist/**
|
||||||
|
--repo '${{ github.repository }}'
|
||||||
|
|
||||||
docker_ubuntu:
|
docker_ubuntu:
|
||||||
name: Build Ubuntu-based Docker image
|
name: Build Ubuntu-based Docker image
|
||||||
@@ -353,9 +376,6 @@ jobs:
|
|||||||
username: jbarlow83
|
username: jbarlow83
|
||||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||||
|
|
||||||
- name: Set up QEMU
|
|
||||||
uses: docker/setup-qemu-action@v3
|
|
||||||
|
|
||||||
- name: Set up Docker Buildx
|
- name: Set up Docker Buildx
|
||||||
id: buildx
|
id: buildx
|
||||||
uses: docker/setup-buildx-action@v3
|
uses: docker/setup-buildx-action@v3
|
||||||
@@ -367,6 +387,6 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
docker buildx build \
|
docker buildx build \
|
||||||
--push \
|
--push \
|
||||||
--platform linux/amd64 \
|
--platform linux/amd64,linux/arm64 \
|
||||||
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \
|
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \
|
||||||
--file .docker/Dockerfile.alpine .
|
--file .docker/Dockerfile.alpine .
|
||||||
|
|||||||
@@ -44,3 +44,4 @@ docs/_build/
|
|||||||
docs/_static/
|
docs/_static/
|
||||||
docs/_templates/
|
docs/_templates/
|
||||||
docs/Makefile
|
docs/Makefile
|
||||||
|
src/ocrmypdf/_version.py
|
||||||
+1
-1
@@ -19,7 +19,7 @@ formats:
|
|||||||
build:
|
build:
|
||||||
os: ubuntu-22.04
|
os: ubuntu-22.04
|
||||||
tools:
|
tools:
|
||||||
python: "3.9"
|
python: "3.10"
|
||||||
|
|
||||||
python:
|
python:
|
||||||
install:
|
install:
|
||||||
|
|||||||
@@ -70,6 +70,7 @@ Files: tests/resources/linn.png
|
|||||||
tests/resources/ccitt.pdf
|
tests/resources/ccitt.pdf
|
||||||
tests/resources/cardinal.pdf
|
tests/resources/cardinal.pdf
|
||||||
tests/resources/jbig2.pdf
|
tests/resources/jbig2.pdf
|
||||||
|
tests/resources/jbig2_baddevicen.pdf
|
||||||
tests/resources/skew.pdf
|
tests/resources/skew.pdf
|
||||||
tests/resources/rotated_skew.pdf
|
tests/resources/rotated_skew.pdf
|
||||||
tests/resources/poster.pdf
|
tests/resources/poster.pdf
|
||||||
@@ -123,6 +124,14 @@ Copyright: Kai-Uwe Behrmann <www.behrmann.name>
|
|||||||
ColorSolutions <www.basICColor.com>
|
ColorSolutions <www.basICColor.com>
|
||||||
License: Zlib
|
License: Zlib
|
||||||
|
|
||||||
|
Files: src/ocrmypdf/data/pdf.ttf
|
||||||
|
Copyright: (C) 2014 Ray Smith
|
||||||
|
(C) 2015 Ken Sharp
|
||||||
|
(C) 2016 James R. Barlow
|
||||||
|
(C) 2016 Jeff Breidenbach
|
||||||
|
(C) 2017 Zdenko Podobný
|
||||||
|
License: Apache-2.0
|
||||||
|
|
||||||
Files: tests/resources/3small.pdf
|
Files: tests/resources/3small.pdf
|
||||||
Copyright: (C) 2014 Euskaldunaa
|
Copyright: (C) 2014 Euskaldunaa
|
||||||
(C) 2017 James R. Barlow
|
(C) 2017 James R. Barlow
|
||||||
|
|||||||
@@ -70,6 +70,7 @@ Linux, Windows, macOS and FreeBSD are supported. Docker images are also availabl
|
|||||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||||
| Fedora | ``dnf install ocrmypdf`` |
|
| Fedora | ``dnf install ocrmypdf`` |
|
||||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||||
|
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||||
|
|||||||
+99
-41
@@ -118,15 +118,16 @@ OCR for huge images
|
|||||||
-------------------
|
-------------------
|
||||||
|
|
||||||
Tesseract has internal limits on the size
|
Tesseract has internal limits on the size
|
||||||
of images it will process. If you issue
|
of images it will process. By default,
|
||||||
``--tesseract-downsample-large-images``, OCRmyPDF will downsample images
|
``--tesseract-downsample-large-images`` is enabled, and OCRmyPDF will
|
||||||
to fit Tesseract limits. (The limits are usually entered only for scanned
|
downsample images to fit Tesseract limits. (The limits are usually encountered
|
||||||
images of oversized media, such as large maps or blueprints exceeding
|
only for scanned images of oversized media, such as large maps or blueprints exceeding
|
||||||
110 cm or 43 inches in either dimension, and at high DPI.)
|
110 cm or 43 inches in either dimension, and at high DPI.) This feature can disabled
|
||||||
|
using ``--no-tesseract-downsample-large-images``.
|
||||||
|
|
||||||
``--tesseract-downsample-above Npixels`` adjusts the threshold at which images
|
``--tesseract-downsample-above Npixels`` adjusts the threshold at which images
|
||||||
will be downsampled. By default, only images that exceed any of Tesseract's
|
will be downsampled. By default, only images that exceed any of Tesseract's
|
||||||
internal limits are downsampled.
|
internal limits are downsampled (32767 pixels on either dimension).
|
||||||
|
|
||||||
You will also need to set ``--tesseract-timeout`` high enough to allow
|
You will also need to set ``--tesseract-timeout`` high enough to allow
|
||||||
for processing.
|
for processing.
|
||||||
@@ -227,6 +228,59 @@ then run ocrmypdf as follows (along with any other desired arguments):
|
|||||||
Some combinations of control parameters will break Tesseract or break
|
Some combinations of control parameters will break Tesseract or break
|
||||||
assumptions that OCRmyPDF makes about Tesseract's output.
|
assumptions that OCRmyPDF makes about Tesseract's output.
|
||||||
|
|
||||||
|
Changing page segmentation mode
|
||||||
|
-------------------------------
|
||||||
|
|
||||||
|
The directive ``--tesseract-pagesegmode Nmode`` forwards the desired page segmentation
|
||||||
|
mode to Tesseract OCR. The default is 3.
|
||||||
|
|
||||||
|
Page segmentation can improve OCR results when you know that a PDF ought to be
|
||||||
|
analyzed a particular way, such as PDFs whose pages contain only a single line of
|
||||||
|
text. For the vast majority of users, changing the page segmentation mode will only
|
||||||
|
make things worse.
|
||||||
|
|
||||||
|
As of June 2024, the Tesseract page segmentation modes are:
|
||||||
|
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| ID | Description |
|
||||||
|
+=====+==================================================================================+
|
||||||
|
| 0 | Orientation and script detection (OSD) only. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 1 | Automatic page segmentation with OSD. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 2 | Automatic page segmentation, but no OSD, or OCR. (not implemented) |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 3 | Fully automatic page segmentation, but no OSD. (Default) |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 4 | Assume a single column of text of variable sizes. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 5 | Assume a single uniform block of vertically aligned text. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 6 | Assume a single uniform block of text. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 7 | Treat the image as a single text line. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 8 | Treat the image as a single word. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 9 | Treat the image as a single word in a circle. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 10 | Treat the image as a single character. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 11 | Sparse text. Find as much text as possible in no particular order. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 12 | Sparse text with OSD. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 13 | Raw line. Treat the image as a single text line, bypassing hacks that are |
|
||||||
|
| | Tesseract-specific. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
|
||||||
|
Modes 0, 1, 2, and 12 (all of those that enable orientation and script detection)
|
||||||
|
are not compatible with OCRmyPDF, which performs OSD in a separate step from OCR.
|
||||||
|
Their use may interfere with ``--rotate-pages`` and other features.
|
||||||
|
|
||||||
|
It is currently not possible to use advanced Tesseract OCR features, such as creating
|
||||||
|
OCR information, when using Tesseract through OCRmyPDF.
|
||||||
|
|
||||||
Changing the PDF renderer
|
Changing the PDF renderer
|
||||||
=========================
|
=========================
|
||||||
|
|
||||||
@@ -239,46 +293,46 @@ rendering
|
|||||||
OCRmyPDF has these PDF renderers: ``sandwich`` and ``hocr``. The
|
OCRmyPDF has these PDF renderers: ``sandwich`` and ``hocr``. The
|
||||||
renderer may be selected using ``--pdf-renderer``. The default is
|
renderer may be selected using ``--pdf-renderer``. The default is
|
||||||
``auto`` which lets OCRmyPDF select the renderer to use. Currently,
|
``auto`` which lets OCRmyPDF select the renderer to use. Currently,
|
||||||
``auto`` always selects ``sandwich``.
|
``auto`` always selects ``hocr``.
|
||||||
|
|
||||||
The ``sandwich`` renderer
|
|
||||||
-------------------------
|
|
||||||
|
|
||||||
The ``sandwich`` renderer uses Tesseract's new text-only PDF feature,
|
|
||||||
which produces a PDF page that lays out the OCR in invisible text. This
|
|
||||||
page is then "sandwiched" onto the original PDF page, allowing lossless
|
|
||||||
application of OCR even to PDF pages that contain other vector objects.
|
|
||||||
|
|
||||||
Currently this is the best renderer for most uses, however it is
|
|
||||||
implemented in Tesseract so OCRmyPDF cannot influence it. Currently some
|
|
||||||
problematic PDF viewers like Mozilla PDF.js and macOS Preview have
|
|
||||||
problems with segmenting its text output, and
|
|
||||||
mightrunseveralwordstogether.
|
|
||||||
|
|
||||||
When image preprocessing features like ``--deskew`` are used, the
|
|
||||||
original PDF will be rendered as a full page and the OCR layer will be
|
|
||||||
placed on top.
|
|
||||||
|
|
||||||
The ``hocr`` renderer
|
The ``hocr`` renderer
|
||||||
---------------------
|
---------------------
|
||||||
|
|
||||||
The ``hocr`` renderer works with older versions of Tesseract. The image
|
.. versionchanged:: 16.0.0
|
||||||
layer is copied from the original PDF page if possible, avoiding
|
|
||||||
potentially lossy transcoding or loss of other PDF information. If
|
|
||||||
preprocessing is specified, then the image layer is a new PDF. (You may
|
|
||||||
need to disable PDF/A conversion nad optimization to eliminate all
|
|
||||||
lossy transformations.)
|
|
||||||
|
|
||||||
Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone
|
In both renderers, a text-only layer is rendered and sandwiched (overlaid)
|
||||||
looking to customize how OCR is presented should look here. A major
|
on to either the original PDF page, or newly rasterized version of the
|
||||||
disadvantage of this renderer is it not capable of correctly handling
|
original PDF page (when ``--force-ocr`` is used). In this way, loss
|
||||||
text outside the Latin alphabet (specifically, it supports the ISO 8859-1
|
of PDF information is generally avoided. (You may need to disable PDF/A
|
||||||
character set). Pull requests to improve the situation are welcome.
|
conversion and optimization to eliminate all lossy transformations.)
|
||||||
|
|
||||||
Currently, this renderer has the best compatibility with Mozilla's
|
The current approach used by the new hOCR renderer is a re-implementation
|
||||||
PDF.js viewer.
|
of Tesseract's PDF renderer, using the same Glyphless font and general
|
||||||
|
ideas, but fixing many technical issues that impeded it. The new hocr
|
||||||
|
provides better text placement accuracy, avoids issues with word
|
||||||
|
segmentation, and provides better positioning of skewed text.
|
||||||
|
|
||||||
This works in all versions of Tesseract.
|
Using the experimental API, it is also possible to edit the OCR output
|
||||||
|
from Tesseract, using any tool that is capable of editing hOCR files.
|
||||||
|
|
||||||
|
Older versions of this renderer did not support non-Latin languages, but
|
||||||
|
it is now universal.
|
||||||
|
|
||||||
|
The ``sandwich`` renderer
|
||||||
|
-------------------------
|
||||||
|
|
||||||
|
The ``sandwich`` renderer uses Tesseract's text-only PDF feature,
|
||||||
|
which produces a PDF page that lays out the OCR in invisible text.
|
||||||
|
|
||||||
|
Currently some problematic PDF viewers like Mozilla PDF.js and macOS
|
||||||
|
Preview have problems with segmenting its text output, and
|
||||||
|
mightrunseveralwordstogether. It also does not implement right to left
|
||||||
|
fonts (Arabic, Hebrew, Persian). The output of this renderer cannot
|
||||||
|
be edited. The sandwich renderer is retained for testing.
|
||||||
|
|
||||||
|
When image preprocessing features like ``--deskew`` are used, the
|
||||||
|
original PDF will be rendered as a full page and the OCR layer will be
|
||||||
|
placed on top.
|
||||||
|
|
||||||
Rendering and rasterizing options
|
Rendering and rasterizing options
|
||||||
=================================
|
=================================
|
||||||
@@ -400,6 +454,10 @@ whether it succeeded or failed. An example message is:
|
|||||||
Temporary working files retained at:
|
Temporary working files retained at:
|
||||||
/tmp/ocrmypdf.io.u20wpz07
|
/tmp/ocrmypdf.io.u20wpz07
|
||||||
|
|
||||||
|
When OCRmyPDF is launched as a snap, this corresponds to the snap filesystem, for instance:
|
||||||
|
|
||||||
|
/tmp/snap-private-tmp/snap.ocrmypdf/tmp/ocrmypdf.io.u20wpz07
|
||||||
|
|
||||||
The organization of this folder is an implementation detail and subject
|
The organization of this folder is an implementation detail and subject
|
||||||
to change between releases. However the general organization is that
|
to change between releases. However the general organization is that
|
||||||
working files on a per page basis have the page number as a prefix
|
working files on a per page basis have the page number as a prefix
|
||||||
@@ -411,9 +469,9 @@ suffix indicates the file type. Some important files include:
|
|||||||
on arguments this may differ from the presentation image
|
on arguments this may differ from the presentation image
|
||||||
- ``_pp_deskew.png`` - the image, after deskewing
|
- ``_pp_deskew.png`` - the image, after deskewing
|
||||||
- ``_pp_clean.png`` - the image, after cleaning with unpaper
|
- ``_pp_clean.png`` - the image, after cleaning with unpaper
|
||||||
- ``_ocr_tess.pdf`` - the OCR file; appears as a blank page with invisible
|
- ``_ocr_hocr.pdf`` - the OCR file; appears as a blank page with invisible
|
||||||
text embedded
|
text embedded
|
||||||
- ``_ocr_tess.txt`` - the OCR text (not necessarily all text on the page,
|
- ``_ocr_hocr.txt`` - the OCR text (not necessarily all text on the page,
|
||||||
if the page is mixed format)
|
if the page is mixed format)
|
||||||
- ``fix_docinfo.pdf`` - a temporary file created to fix the PDF DocumentInfo
|
- ``fix_docinfo.pdf`` - a temporary file created to fix the PDF DocumentInfo
|
||||||
data structure
|
data structure
|
||||||
|
|||||||
+3
-1
@@ -112,7 +112,9 @@ OCRmyPDF is strict about not writing to standard output so that
|
|||||||
users can safely use it in a pipeline and produce a valid output
|
users can safely use it in a pipeline and produce a valid output
|
||||||
file. A caller application will have to ensure it does not write to
|
file. A caller application will have to ensure it does not write to
|
||||||
standard output either, if it wants to be compatible with this
|
standard output either, if it wants to be compatible with this
|
||||||
behavior and support piping to a file.
|
behavior and support piping to a file. Another benefit of running
|
||||||
|
OCRmyPDF in a child process, as recommended above, is that it will
|
||||||
|
not interfere with the parent process's standard output.
|
||||||
|
|
||||||
Exceptions
|
Exceptions
|
||||||
----------
|
----------
|
||||||
|
|||||||
+2
-2
@@ -44,7 +44,7 @@ place, and printing each filename in between runs:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
find . -printf '%p\n' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
|
find . -name '*.pdf' -printf '%p\n' -exec ocrmypdf '{}' '{}' \;
|
||||||
|
|
||||||
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
||||||
``find`` to create a directory list and ``parallel`` to parallelize runs
|
``find`` to create a directory list and ``parallel`` to parallelize runs
|
||||||
@@ -135,7 +135,7 @@ Users may need to customize the script to meet their requirements.
|
|||||||
"OCR_ON_SUCCESS_ARCHIVE", "This will move the processed original file to ``OCR_ARCHIVE_DIRECTORY`` if the exit code is 0 (OK). Note that ``OCR_ON_SUCCESS_DELETE`` takes precedence over this option, i.e. if both options are set, the input file will be deleted."
|
"OCR_ON_SUCCESS_ARCHIVE", "This will move the processed original file to ``OCR_ARCHIVE_DIRECTORY`` if the exit code is 0 (OK). Note that ``OCR_ON_SUCCESS_DELETE`` takes precedence over this option, i.e. if both options are set, the input file will be deleted."
|
||||||
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||||
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
||||||
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
|
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true, ""optimize"": "3"}'``."
|
||||||
"OCR_POLL_NEW_FILE_SECONDS", "Polling interval"
|
"OCR_POLL_NEW_FILE_SECONDS", "Polling interval"
|
||||||
"OCR_LOGLEVEL", "Level of log messages to report"
|
"OCR_LOGLEVEL", "Level of log messages to report"
|
||||||
|
|
||||||
|
|||||||
+16
-2
@@ -187,7 +187,7 @@ might remove desirable content, especially from poor quality scans.
|
|||||||
background from grayscale or color images. Monochrome images are
|
background from grayscale or color images. Monochrome images are
|
||||||
ignored. This should not be used on documents that contain color
|
ignored. This should not be used on documents that contain color
|
||||||
photos as it may remove them.
|
photos as it may remove them.
|
||||||
- ``--deskew`` will correct pages were scanned at a skewed angle by
|
- ``--deskew`` will correct pages that were scanned at a skewed angle by
|
||||||
rotating them back into place.
|
rotating them back into place.
|
||||||
- ``--clean`` uses
|
- ``--clean`` uses
|
||||||
`unpaper <https://www.flameeyes.eu/projects/unpaper>`__ to clean up
|
`unpaper <https://www.flameeyes.eu/projects/unpaper>`__ to clean up
|
||||||
@@ -245,6 +245,20 @@ if all you want to is to apply image processing or PDF/A conversion.
|
|||||||
the case. Use ``--tesseract-non-ocr-timeout`` to control the timeout
|
the case. Use ``--tesseract-non-ocr-timeout`` to control the timeout
|
||||||
for non-OCR operations, if needed.
|
for non-OCR operations, if needed.
|
||||||
|
|
||||||
|
Remove all text or OCR from my PDF
|
||||||
|
----------------------------------
|
||||||
|
|
||||||
|
This is getting ridiculous, but OCRmyPDF can complete strip all textual
|
||||||
|
information from a PDF and reconstruct it as a "bag of images" PDF.
|
||||||
|
|
||||||
|
.. code-block::
|
||||||
|
|
||||||
|
ocrmypdf --tesseract-timeout 0 --force-ocr input.pdf output.pdf
|
||||||
|
|
||||||
|
Why would you want to do this? Perhaps you have a PDF where OCR
|
||||||
|
fails to produce useful results, and just want to get rid of all OCR information.
|
||||||
|
This command also removes OCR generated by third party tools.
|
||||||
|
|
||||||
Optimize images without performing OCR
|
Optimize images without performing OCR
|
||||||
--------------------------------------
|
--------------------------------------
|
||||||
|
|
||||||
@@ -393,4 +407,4 @@ any digital signatures will be invalidated.
|
|||||||
OCRmyPDF cannot open documents that are encrypted with a digital certificate.
|
OCRmyPDF cannot open documents that are encrypted with a digital certificate.
|
||||||
|
|
||||||
Versions of OCRmyPDF prior to 14.4.0 would invalidate existing digital signatures
|
Versions of OCRmyPDF prior to 14.4.0 would invalidate existing digital signatures
|
||||||
without warning.
|
without warning.
|
||||||
|
|||||||
+9
-3
@@ -42,7 +42,7 @@ execute the image:
|
|||||||
- Architecture
|
- Architecture
|
||||||
- Description
|
- Description
|
||||||
* - ``jbarlow83/ocrmypdf-alpine``
|
* - ``jbarlow83/ocrmypdf-alpine``
|
||||||
- x86_64 only
|
- x86_64 and arm64
|
||||||
- Recommended image, based on Alpine Linux.
|
- Recommended image, based on Alpine Linux.
|
||||||
* - ``jbarlow83/ocrmypdf-ubuntu``
|
* - ``jbarlow83/ocrmypdf-ubuntu``
|
||||||
- x86_64 and arm64
|
- x86_64 and arm64
|
||||||
@@ -65,7 +65,13 @@ The ``ocrmypdf`` image is also available, but is deprecated and will be removed
|
|||||||
in the future.
|
in the future.
|
||||||
|
|
||||||
OCRmyPDF will use all available CPU cores. See the Docker documentation for
|
OCRmyPDF will use all available CPU cores. See the Docker documentation for
|
||||||
`adjusting memory and CPU on other platforms <https://docs.docker.com/config/containers/resource_constraints/>`__.
|
`adjusting memory and CPU on other platforms <https://docs.docker.com/config/containers/resource_constraints/>`__
|
||||||
|
if you are using Docker on macOS or Windows, where you may need to manually assign
|
||||||
|
more resources. On Linux, all resources will be available automatically.
|
||||||
|
|
||||||
|
The underlying operating system and other details in Docker images are subject
|
||||||
|
to change at minor releases. If you are modifying the image, you should pin
|
||||||
|
the version you intend to use.
|
||||||
|
|
||||||
Using the Docker image on the command line
|
Using the Docker image on the command line
|
||||||
==========================================
|
==========================================
|
||||||
@@ -81,7 +87,7 @@ To start a Docker container (instance of the image):
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
docker tag jbarlow83/ocrmypdf ocrmypdf
|
docker tag jbarlow83/ocrmypdf-alpine ocrmypdf
|
||||||
docker run --rm -i ocrmypdf (... all other arguments here...) - -
|
docker run --rm -i ocrmypdf (... all other arguments here...) - -
|
||||||
|
|
||||||
For convenience, create a shell alias to hide the Docker command. It is
|
For convenience, create a shell alias to hide the Docker command. It is
|
||||||
|
|||||||
+87
-50
@@ -23,7 +23,9 @@ These platforms have one-liner installs:
|
|||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| Fedora | ``dnf install ocrmypdf tesseract-osd`` |
|
| Fedora | ``dnf install ocrmypdf tesseract-osd`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| macOS | ``brew install ocrmypdf`` |
|
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||||
|
+-------------------------------+-----------------------------------------+
|
||||||
|
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
@@ -99,12 +101,12 @@ For full details on version availability for your platform, check the
|
|||||||
Fedora
|
Fedora
|
||||||
------
|
------
|
||||||
|
|
||||||
.. |fedora-37| image:: https://repology.org/badge/version-for-repo/fedora_37/ocrmypdf.svg
|
|
||||||
:alt: Fedora 37
|
|
||||||
|
|
||||||
.. |fedora-38| image:: https://repology.org/badge/version-for-repo/fedora_38/ocrmypdf.svg
|
.. |fedora-38| image:: https://repology.org/badge/version-for-repo/fedora_38/ocrmypdf.svg
|
||||||
:alt: Fedora 38
|
:alt: Fedora 38
|
||||||
|
|
||||||
|
.. |fedora-39| image:: https://repology.org/badge/version-for-repo/fedora_39/ocrmypdf.svg
|
||||||
|
:alt: Fedora 39
|
||||||
|
|
||||||
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
||||||
:alt: Fedore Rawhide
|
:alt: Fedore Rawhide
|
||||||
|
|
||||||
@@ -113,7 +115,7 @@ Fedora
|
|||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |latest| |
|
| |latest| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |fedora-37| |fedora-38| |fedora-rawhide| |
|
| |fedora-38| |fedora-39| |fedora-rawhide| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
|
|
||||||
Users of Fedora may simply
|
Users of Fedora may simply
|
||||||
@@ -123,7 +125,7 @@ Users of Fedora may simply
|
|||||||
dnf install ocrmypdf tesseract-osd
|
dnf install ocrmypdf tesseract-osd
|
||||||
|
|
||||||
For full details on version availability, check the `Fedora Package
|
For full details on version availability, check the `Fedora Package
|
||||||
Tracker <https://apps.fedoraproject.org/packages/ocrmypdf>`__.
|
Tracker <https://packages.fedoraproject.org/pkgs/ocrmypdf/ocrmypdf/>`__.
|
||||||
|
|
||||||
If the version available for your platform is out of date, you could opt
|
If the version available for your platform is out of date, you could opt
|
||||||
to install the latest version from source. See `Installing HEAD revision
|
to install the latest version from source. See `Installing HEAD revision
|
||||||
@@ -135,10 +137,36 @@ from sources <#installing-head-revision-from-sources>`__.
|
|||||||
issues. OCRmyPDF works fine without it but will produce larger output
|
issues. OCRmyPDF works fine without it but will produce larger output
|
||||||
files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later
|
files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later
|
||||||
will automatically detect it on the ``PATH``. To add JBIG2 encoding,
|
will automatically detect it on the ``PATH``. To add JBIG2 encoding,
|
||||||
see `Installing the JBIG2 encoder <jbig2>`__.
|
see :ref:`Installing the JBIG2 encoder <jbig2>`.
|
||||||
|
|
||||||
.. _ubuntu-lts-latest:
|
.. _ubuntu-lts-latest:
|
||||||
|
|
||||||
|
RHEL 9
|
||||||
|
------
|
||||||
|
|
||||||
|
Prepare the environment by getting Python 3.11:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
dnf install python3.11 python3.11-pip
|
||||||
|
|
||||||
|
Then, follow `Requirements for pip and HEAD install <#requirements-for-pip-and-head-install>`__ to instal dependencies:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
dnf install ghostscript tesseract
|
||||||
|
|
||||||
|
and build ocrmypdf in virtual environment:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
python3.11 -m venv .venv
|
||||||
|
|
||||||
|
To add JBIG2 encoding, see :ref:`Installing the JBIG2 encoder <jbig2>`.
|
||||||
|
|
||||||
|
Note Fedora packages for language data haven't been branched for RHEL/EPEL, but you can get traineddata files directly from `tesseract
|
||||||
|
<https://github.com/tesseract-ocr/tessdata/>`__ and place them in ``/usr/share/tesseract/tessdata``.
|
||||||
|
|
||||||
Installing the latest version on Ubuntu 22.04 LTS
|
Installing the latest version on Ubuntu 22.04 LTS
|
||||||
-------------------------------------------------
|
-------------------------------------------------
|
||||||
|
|
||||||
@@ -162,37 +190,13 @@ To add JBIG2 encoding, see :ref:`jbig2`.
|
|||||||
Ubuntu 20.04 LTS
|
Ubuntu 20.04 LTS
|
||||||
----------------
|
----------------
|
||||||
|
|
||||||
Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with ``apt``. To
|
Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with ``apt``. The
|
||||||
install a more recent version, uninstall the system-provided version of
|
most convenient way to install recent OCRmyPDF on older Ubuntu is to use
|
||||||
ocrmypdf, and install the following dependencies:
|
Homebrew on Linux (Linuxbrew).
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
sudo apt-get -y remove ocrmypdf # remove system ocrmypdf, if installed
|
brew install ocrmypdf
|
||||||
sudo apt-get -y update
|
|
||||||
sudo apt-get -y install \
|
|
||||||
ghostscript \
|
|
||||||
icc-profiles-free \
|
|
||||||
libxml2 \
|
|
||||||
pngquant \
|
|
||||||
python3-pip \
|
|
||||||
tesseract-ocr \
|
|
||||||
zlib1g
|
|
||||||
|
|
||||||
To install ocrmypdf for the system:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
pip3 install ocrmypdf
|
|
||||||
|
|
||||||
To install for the current user only:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
export PATH=$HOME/.local/bin:$PATH
|
|
||||||
pip3 install --user ocrmypdf
|
|
||||||
|
|
||||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
|
||||||
|
|
||||||
Arch Linux (AUR)
|
Arch Linux (AUR)
|
||||||
----------------
|
----------------
|
||||||
@@ -215,12 +219,12 @@ you are using a VM image, such as `the official Vagrant image
|
|||||||
be completed for you.
|
be completed for you.
|
||||||
|
|
||||||
Next you should install the `base-devel package group
|
Next you should install the `base-devel package group
|
||||||
<https://www.archlinux.org/groups/x86_64/base-devel/>`__. This includes the
|
<https://archlinux.org/packages/core/any/base-devel/>`__. This includes the
|
||||||
standard tooling needed to build packages, such as a compiler and binary tools.
|
standard tooling needed to build packages, such as a compiler and binary tools.
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
sudo pacman -S base-devel
|
sudo pacman -S --needed base-devel
|
||||||
|
|
||||||
Now you are ready to install the OCRmyPDF package.
|
Now you are ready to install the OCRmyPDF package.
|
||||||
|
|
||||||
@@ -258,7 +262,7 @@ page.
|
|||||||
<https://aur.archlinux.org/packages/jbig2enc-git/>`__ and may be installed
|
<https://aur.archlinux.org/packages/jbig2enc-git/>`__ and may be installed
|
||||||
using the same series of steps as for the installation OCRmyPDF AUR
|
using the same series of steps as for the installation OCRmyPDF AUR
|
||||||
package. Alternatively, it may be built manually from source following the
|
package. Alternatively, it may be built manually from source following the
|
||||||
instructions in `Installing the JBIG2 encoder <jbig2>`__. If JBIG2 is
|
instructions in :ref:`Installing the JBIG2 encoder <jbig2>`. If JBIG2 is
|
||||||
installed, OCRmyPDF 7.0.0 and later will automatically detect it.
|
installed, OCRmyPDF 7.0.0 and later will automatically detect it.
|
||||||
|
|
||||||
Alpine Linux
|
Alpine Linux
|
||||||
@@ -323,6 +327,22 @@ languages you can optionally install them all:
|
|||||||
|
|
||||||
brew install tesseract-lang # Optional: Install all language packs
|
brew install tesseract-lang # Optional: Install all language packs
|
||||||
|
|
||||||
|
MacPorts
|
||||||
|
--------
|
||||||
|
|
||||||
|
.. image:: https://img.shields.io/badge/dynamic/json?url=https%3A%2F%2Fports.macports.org%2Fapi%2Fv1%2Fports%2Focrmypdf%2F%3Fformat%3Djson&query=version&label=MacPorts
|
||||||
|
:alt: Macports Version Information
|
||||||
|
:target: https://ports.macports.org/port/ocrmypdf
|
||||||
|
|
||||||
|
OCRmyPDF is includes in MacPorts:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
sudo port install ocrmypdf
|
||||||
|
|
||||||
|
Note that while this will install tesseract you will need to install
|
||||||
|
the appropriate tesseract `language ports <https://ports.macports.org/search/?selected_facets=categories_exact%3Atextproc&installed_file=&q=tesseract&name=on>`__.
|
||||||
|
|
||||||
Manual installation on macOS
|
Manual installation on macOS
|
||||||
----------------------------
|
----------------------------
|
||||||
|
|
||||||
@@ -391,18 +411,20 @@ package manager:
|
|||||||
|
|
||||||
* ``winget install -e --id Python.Python.3.11``
|
* ``winget install -e --id Python.Python.3.11``
|
||||||
* ``winget install -e --id UB-Mannheim.TesseractOCR``
|
* ``winget install -e --id UB-Mannheim.TesseractOCR``
|
||||||
* ``winget install -e --id ArtifexSoftware.GhostScript``
|
|
||||||
|
|
||||||
|
You will need to install Ghostscript manually, `since it does not support automated
|
||||||
|
installs anymore <https://artifex.com/news/ghostscript-10.01.0-disabling-silent-install-option>`_.
|
||||||
|
|
||||||
|
* `Ghostscript download page <https://ghostscript.com/releases/gsdnld.html>`_.`
|
||||||
|
|
||||||
(Or alternately, using the `Chocolatey <https://chocolatey.org/>`_ package manager, install
|
(Or alternately, using the `Chocolatey <https://chocolatey.org/>`_ package manager, install
|
||||||
the following when running in an Administrator command prompt):
|
the following when running in an Administrator command prompt):
|
||||||
|
|
||||||
* ``choco install python3``
|
* ``choco install python3``
|
||||||
* ``choco install --pre tesseract``
|
* ``choco install --pre tesseract``
|
||||||
* ``choco install ghostscript``
|
|
||||||
* ``choco install pngquant`` (optional)
|
* ``choco install pngquant`` (optional)
|
||||||
|
|
||||||
Either set of commands will install the required software. At the mmoment there is no
|
Either set of commands will install the required software. At the moment there is no
|
||||||
single command to install Windows.
|
single command to install Windows.
|
||||||
|
|
||||||
You may then use ``pip`` to install ocrmypdf. (This can performed by a user or
|
You may then use ``pip`` to install ocrmypdf. (This can performed by a user or
|
||||||
@@ -452,7 +474,7 @@ Cygwin64
|
|||||||
|
|
||||||
First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``::
|
First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``::
|
||||||
|
|
||||||
python39 (or later)
|
python310 (or later)
|
||||||
python3?-devel
|
python3?-devel
|
||||||
python3?-pip
|
python3?-pip
|
||||||
python3?-lxml
|
python3?-lxml
|
||||||
@@ -538,9 +560,25 @@ try:
|
|||||||
|
|
||||||
pip install --user ocrmypdf
|
pip install --user ocrmypdf
|
||||||
|
|
||||||
|
(If the message appears ``Requirement already satisfied: ocrmypdf in...``,
|
||||||
|
you will need to use ``pip install --user --upgrade ocrmypdf``.)
|
||||||
|
|
||||||
You should then be able to run ``ocrmypdf --version`` and see that the
|
You should then be able to run ``ocrmypdf --version`` and see that the
|
||||||
latest version was located.
|
latest version was located.
|
||||||
|
|
||||||
|
Installing with pipx
|
||||||
|
====================
|
||||||
|
|
||||||
|
Some users may prefer pipx. As with the method above, you will need to
|
||||||
|
satisfy all non-Python dependencies. Then if pipx is installed, you
|
||||||
|
can use
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
pipx run ocrmypdf
|
||||||
|
|
||||||
|
(If not installed, pipx will install first.)
|
||||||
|
|
||||||
Requirements for pip and HEAD install
|
Requirements for pip and HEAD install
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
@@ -550,8 +588,8 @@ manager. ``pip`` cannot provide them.
|
|||||||
|
|
||||||
The following versions are required:
|
The following versions are required:
|
||||||
|
|
||||||
- Python 3.9 or newer
|
- Python 3.10 or newer
|
||||||
- Ghostscript 9.55 or newer
|
- Ghostscript 9.54 or newer
|
||||||
- Tesseract 4.1.1 or newer
|
- Tesseract 4.1.1 or newer
|
||||||
- jbig2enc 0.29 or newer
|
- jbig2enc 0.29 or newer
|
||||||
- pngquant 2.5 or newer
|
- pngquant 2.5 or newer
|
||||||
@@ -586,7 +624,7 @@ unfortunately, the ``pip install`` command cannot satisfy all of them.
|
|||||||
Installing HEAD revision from sources
|
Installing HEAD revision from sources
|
||||||
=====================================
|
=====================================
|
||||||
|
|
||||||
If you have ``git`` and Python 3.9 or newer installed, you can install
|
If you have ``git`` and Python 3.10 or newer installed, you can install
|
||||||
from source. When the ``pip`` installer runs, it will alert you if
|
from source. When the ``pip`` installer runs, it will alert you if
|
||||||
dependencies are missing.
|
dependencies are missing.
|
||||||
|
|
||||||
@@ -602,8 +640,7 @@ environment:
|
|||||||
|
|
||||||
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
|
|
||||||
Or, to install in `development
|
Or, to install in editable mode
|
||||||
mode <https://pythonhosted.org/setuptools/setuptools.html#development-mode>`__,
|
|
||||||
allowing customization of OCRmyPDF, use the ``-e`` flag:
|
allowing customization of OCRmyPDF, use the ``-e`` flag:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
@@ -643,7 +680,7 @@ To install all of the development and test requirements:
|
|||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
python -m .venv
|
python -m venv .venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
cd OCRmyPDF
|
cd OCRmyPDF
|
||||||
pip install -e .[test]
|
pip install -e .[test]
|
||||||
@@ -669,7 +706,7 @@ To manually install the ``fish`` completion, copy
|
|||||||
Note on 32-bit support
|
Note on 32-bit support
|
||||||
======================
|
======================
|
||||||
|
|
||||||
Many Python libraries no longer 32-bit binary wheels for Linux. This
|
Many Python libraries no longer provide 32-bit binary wheels for Linux. This
|
||||||
includes many of the libraries that OCRmyPDF depends on, such as
|
includes many of the libraries that OCRmyPDF depends on, such as
|
||||||
Pillow. The easiest way to express this to end users is to say we don't
|
Pillow. The easiest way to express this to end users is to say we don't
|
||||||
support 32-bit Linux.
|
support 32-bit Linux.
|
||||||
@@ -679,4 +716,4 @@ can still install and use OCRmyPDF. A warning message will appear.
|
|||||||
In practice, OCRmyPDF may need more than 32-bit memory space to run when
|
In practice, OCRmyPDF may need more than 32-bit memory space to run when
|
||||||
large documents are processed, so there are practical limitations to what
|
large documents are processed, so there are practical limitations to what
|
||||||
users can accomplish with it. Still, for the common use case of an 32-bit
|
users can accomplish with it. Still, for the common use case of an 32-bit
|
||||||
ARM NAS or Raspberry Pi processing small documents, it should work.
|
ARM NAS or Raspberry Pi processing small documents, it should work.
|
||||||
|
|||||||
+9
-1
@@ -41,7 +41,15 @@ For all other platforms, you would need to build the JBIG2 encoder from source:
|
|||||||
|
|
||||||
Dependencies include libtoolize and libleptonica, which on Ubuntu systems
|
Dependencies include libtoolize and libleptonica, which on Ubuntu systems
|
||||||
are packaged as libtool and libleptonica-dev. On Fedora (35) they are packaged
|
are packaged as libtool and libleptonica-dev. On Fedora (35) they are packaged
|
||||||
as libtool and leptonica-devel.
|
as libtool and leptonica-devel. For this to work, please make sure to install
|
||||||
|
``autotools``, ``automake``, ``libtool`` and ``leptonica`` first if not already
|
||||||
|
installed.
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
[sudo] apt install autotools-dev automake libtool libleptonica-dev
|
||||||
|
..
|
||||||
|
|
||||||
|
|
||||||
Lossy mode JBIG2
|
Lossy mode JBIG2
|
||||||
================
|
================
|
||||||
|
|||||||
+47
-13
@@ -18,16 +18,26 @@ Tesseract's documentation also lists the three-letter code for your language.
|
|||||||
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
||||||
are not, e.g. German is ``deu`` and French is ``fra``.
|
are not, e.g. German is ``deu`` and French is ``fra``.
|
||||||
|
|
||||||
|
Language packs (strictly speaking, Tesseract "traineddata" files) generally correspond
|
||||||
|
to the language in question, but different language packs are used in certain
|
||||||
|
situations. For German, the "Fraktur" language pack can assist with reading older
|
||||||
|
materials in the Fraktur typeface family (``deu_frak``). Some communities have changed
|
||||||
|
their script from Cyrillic to Latin; the Cyrillic version of Uzbek is available
|
||||||
|
as ``uzb_cyrl`` and the Latin version is ``uzb``.
|
||||||
|
|
||||||
After you have installed a language pack, you can use it with ``ocrmypdf -l <language>``,
|
After you have installed a language pack, you can use it with ``ocrmypdf -l <language>``,
|
||||||
for example ``ocrmypdf -l spa``. For multilingual documents, you can specify
|
for example ``ocrmypdf -l spa``. For multilingual documents, you can specify
|
||||||
all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French.
|
all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French.
|
||||||
English is assumed by default unless other language(s) are specified.
|
English is assumed by default unless other language(s) are specified.
|
||||||
|
|
||||||
For Linux users, you can often find packages that provide language
|
For Linux users, you can often find packages that provide language
|
||||||
packs:
|
packs.
|
||||||
|
|
||||||
Debian and Ubuntu users
|
Platform install steps
|
||||||
=======================
|
======================
|
||||||
|
|
||||||
|
Debian and Ubuntu (apt)
|
||||||
|
-----------------------
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -42,8 +52,8 @@ to what languages it should search for. Multiple languages can be
|
|||||||
requested using either ``-l eng+fra`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fra``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
Fedora users
|
Fedora
|
||||||
============
|
------
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -58,8 +68,24 @@ to what languages it should search for. Multiple languages can be
|
|||||||
requested using either ``-l eng+fra`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fra``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
Gentoo users
|
Archlinux
|
||||||
============
|
------
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
# Display a list of all Tesseract language packs
|
||||||
|
pacman -Ss tesseract-data
|
||||||
|
|
||||||
|
# Install German language pack
|
||||||
|
pacman -S tesseract-data-deu
|
||||||
|
|
||||||
|
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||||
|
to what languages it should search for. Multiple languages can be
|
||||||
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
|
``-l eng -l fra``.
|
||||||
|
|
||||||
|
Gentoo
|
||||||
|
------
|
||||||
|
|
||||||
On Gentoo the package ``app-text/tessdata_fast``, which ``app-text/tesseract`` depends on, handles Tesseract languages.
|
On Gentoo the package ``app-text/tessdata_fast``, which ``app-text/tesseract`` depends on, handles Tesseract languages.
|
||||||
It accepts USE flags to select what languages should be installed, these can be set in ``/etc/portage/package.use``.
|
It accepts USE flags to select what languages should be installed, these can be set in ``/etc/portage/package.use``.
|
||||||
@@ -85,23 +111,31 @@ to what languages it should search for. Multiple languages can be
|
|||||||
requested using either ``-l eng+fra`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fra``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
macOS users
|
macOS
|
||||||
===========
|
-----
|
||||||
|
|
||||||
You can install additional language packs by
|
You can install additional language packs by
|
||||||
:ref:`installing Tesseract using Homebrew with all language packs <macos-all-languages>`.
|
:ref:`installing Tesseract using Homebrew with all language packs <macos-all-languages>`.
|
||||||
|
|
||||||
Docker users
|
Docker
|
||||||
============
|
------
|
||||||
|
|
||||||
Users of the OCRmyPDF Docker image should install language packs into a
|
Users of the OCRmyPDF Docker image should install language packs into a
|
||||||
derived Docker image as
|
derived Docker image as
|
||||||
:ref:`described in that section <docker-lang-packs>`.
|
:ref:`described in that section <docker-lang-packs>`.
|
||||||
|
|
||||||
Windows users
|
Windows
|
||||||
=============
|
-------
|
||||||
|
|
||||||
The Tesseract installer provided by Chocolatey currently includes only English language.
|
The Tesseract installer provided by Chocolatey currently includes only English language.
|
||||||
To install other languages, download the respective language pack (``.traineddata`` file)
|
To install other languages, download the respective language pack (``.traineddata`` file)
|
||||||
from https://github.com/tesseract-ocr/tessdata/ and place it in
|
from https://github.com/tesseract-ocr/tessdata/ and place it in
|
||||||
``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed).
|
``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed).
|
||||||
|
|
||||||
|
Custom language packs
|
||||||
|
=====================
|
||||||
|
|
||||||
|
If you have fine-tuned or trained Tesseract and generated custom trained data, you can
|
||||||
|
copy your ``customlang.traineddata`` file into your Tesseract "tessdata" folder, and
|
||||||
|
then use the ``-l customlang`` argument to tell OCRmyPDF to pass that language on to
|
||||||
|
Tesseract.
|
||||||
|
|||||||
+13
-3
@@ -38,12 +38,13 @@ on ARM and x86_64. Performance may be poor on other processor architectures.
|
|||||||
Versioning scheme
|
Versioning scheme
|
||||||
-----------------
|
-----------------
|
||||||
|
|
||||||
OCRmyPDF uses setuptools-scm for versioning, which derives the version from
|
OCRmyPDF uses hatch-vcs for versioning, which derives the version from
|
||||||
Git as a single source of truth. This may be unsuitable for some distributions, e.g.
|
Git as a single source of truth. This may be unsuitable for some distributions, e.g.
|
||||||
to indicate that your distribution modifies OCRmyPDF in some way.
|
to indicate that your distribution modifies OCRmyPDF in some way.
|
||||||
|
|
||||||
You can patch the ``__version__`` variable in ``src/ocrmypdf/_version.py`` if
|
You can patch the ``__version__`` variable in ``src/ocrmypdf/_version.py`` if
|
||||||
necessary.
|
necessary, or set the environment variable ``SETUPTOOLS_SCM_PRETEND_VERSION``
|
||||||
|
to the required version, if you need to override versioning for some reason.
|
||||||
|
|
||||||
jbig2enc
|
jbig2enc
|
||||||
--------
|
--------
|
||||||
@@ -64,4 +65,13 @@ installation documentation.
|
|||||||
|
|
||||||
If you maintain a Linux distribution that supports 32-bit x86 or ARM, OCRmyPDF
|
If you maintain a Linux distribution that supports 32-bit x86 or ARM, OCRmyPDF
|
||||||
should continue to work as long as all of its dependencies continue to be
|
should continue to work as long as all of its dependencies continue to be
|
||||||
available in 32-bit form. Please note we do not test on 32-bit platforms.
|
available in 32-bit form. Please note we do not test on 32-bit platforms.
|
||||||
|
|
||||||
|
HEIF/HEIC
|
||||||
|
---------
|
||||||
|
|
||||||
|
OCRmyPDF defaults to installing the pi-heif PyPI package, which supports converting
|
||||||
|
HEIF (High Efficiency Image File Format) images to PDF from the command line.
|
||||||
|
If your distribution does not have this library available, you can exclude it and
|
||||||
|
OCRmyPDF will gracefully degrade automatically, losing only support for this
|
||||||
|
feature.
|
||||||
@@ -17,7 +17,7 @@ If running OCRmyPDF quickly is your main goal, you can use settings such as:
|
|||||||
|
|
||||||
* ``--optimize 0`` to disable file size optimization
|
* ``--optimize 0`` to disable file size optimization
|
||||||
* ``--output-type pdf`` to disable PDF/A generation
|
* ``--output-type pdf`` to disable PDF/A generation
|
||||||
* ``--fast-web-view 0`` to disable fast web view optimization
|
* ``--fast-web-view 999999`` to disable fast web view optimization
|
||||||
* ``--skip-big`` to skip large images, if some pages have large images
|
* ``--skip-big`` to skip large images, if some pages have large images
|
||||||
|
|
||||||
You can also avoid:
|
You can also avoid:
|
||||||
|
|||||||
+11
-12
@@ -29,6 +29,9 @@ conventions. Note that: plugins installed with as setuptools entrypoints are
|
|||||||
not checked currently, because OCRmyPDF assumes you may not want to enable
|
not checked currently, because OCRmyPDF assumes you may not want to enable
|
||||||
plugins for all files.
|
plugins for all files.
|
||||||
|
|
||||||
|
See [OCRmyPDF-EasyOCR](https://github.com/ocrmypdf/OCRmyPDF-EasyOCR) for an
|
||||||
|
example of a straightforward, fully working plugin.
|
||||||
|
|
||||||
Script plugins
|
Script plugins
|
||||||
==============
|
==============
|
||||||
|
|
||||||
@@ -70,14 +73,15 @@ similar to ``pytest`` packages such as ``pytest-cov`` (the package) and
|
|||||||
module), just like pytest plugins. At the same time, please make it clear
|
module), just like pytest plugins. At the same time, please make it clear
|
||||||
that your package is not official.
|
that your package is not official.
|
||||||
|
|
||||||
Setuptools plugins
|
Plugins
|
||||||
==================
|
=======
|
||||||
|
|
||||||
You can also create a plugin that OCRmyPDF will always automatically load if both are
|
You can also create a plugin that OCRmyPDF will always automatically load if both are
|
||||||
installed in the same virtual environment, using a setuptools entrypoint.
|
installed in the same virtual environment, using a project entrypoint.
|
||||||
|
OCRmyPDF uses the entrypoint namespace "ocrmypdf".
|
||||||
|
|
||||||
Your package's ``pyproject.toml`` would need to contain the following, for a plugin
|
For example, ``pyproject.toml`` would need to contain the following, for a plugin named
|
||||||
named ``ocrmypdf-exampleplugin``:
|
``ocrmypdf-exampleplugin``:
|
||||||
|
|
||||||
.. code-block:: toml
|
.. code-block:: toml
|
||||||
|
|
||||||
@@ -87,13 +91,6 @@ named ``ocrmypdf-exampleplugin``:
|
|||||||
[project.entry-points."ocrmypdf"]
|
[project.entry-points."ocrmypdf"]
|
||||||
exampleplugin = "exampleplugin.pluginmodule"
|
exampleplugin = "exampleplugin.pluginmodule"
|
||||||
|
|
||||||
.. code-block:: ini
|
|
||||||
|
|
||||||
# equivalent setup.cfg
|
|
||||||
[options.entry_points]
|
|
||||||
ocrmypdf =
|
|
||||||
exampleplugin = exampleplugin.pluginmodule
|
|
||||||
|
|
||||||
Plugin requirements
|
Plugin requirements
|
||||||
===================
|
===================
|
||||||
|
|
||||||
@@ -179,9 +176,11 @@ Execution and progress reporting
|
|||||||
|
|
||||||
.. autoclass:: ocrmypdf.pluginspec.ProgressBar
|
.. autoclass:: ocrmypdf.pluginspec.ProgressBar
|
||||||
:members:
|
:members:
|
||||||
|
:special-members: __init__, __enter__, __exit__
|
||||||
|
|
||||||
.. autoclass:: ocrmypdf.pluginspec.Executor
|
.. autoclass:: ocrmypdf.pluginspec.Executor
|
||||||
:members:
|
:members:
|
||||||
|
:special-members: __call__
|
||||||
|
|
||||||
.. autofunction:: ocrmypdf.pluginspec.get_logging_console
|
.. autofunction:: ocrmypdf.pluginspec.get_logging_console
|
||||||
|
|
||||||
|
|||||||
@@ -20,6 +20,8 @@ The most recent release of OCRmyPDF is |OCRmyPDF PyPI|. Any newer versions
|
|||||||
referred to in these notes may exist the main branch but have not been
|
referred to in these notes may exist the main branch but have not been
|
||||||
tagged yet.
|
tagged yet.
|
||||||
|
|
||||||
|
OCRmyPDF typically supports the three most recent Python versions.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
|
|
||||||
Attention maintainers: these release notes may be updated with information
|
Attention maintainers: these release notes may be updated with information
|
||||||
@@ -28,6 +30,233 @@ tagged yet.
|
|||||||
|
|
||||||
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||||
|
|
||||||
|
v16.6.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Remove invalid hyperlink annotations to satisfy Ghostscript 10.x during PDF/A
|
||||||
|
conversion. :issue:`1425`
|
||||||
|
|
||||||
|
v16.6.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed some issues with Docker build, such as removing unnecessary content and using
|
||||||
|
a stable Tesseract version.
|
||||||
|
- Reverted Docker image to Ubuntu 22.04 to access older/more stable Ghostscript
|
||||||
|
for now.
|
||||||
|
- Clarified batch commands in documentation.
|
||||||
|
- Fixed an issue with JSON serialization and pickling of HOCRResult. :issue:`1427`
|
||||||
|
|
||||||
|
v16.6.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an issue where damaged PDFs would fail with ``--redo-ocr``. :issue:`1403`
|
||||||
|
- Fixed an error that prevented JBIG2 optimization on Windows if the image
|
||||||
|
was optimized in an earlier step. :issue:`1396`
|
||||||
|
- Fixed an error detecting the version of unpaper 7.0.0. :issue:`1409`
|
||||||
|
- Fixed a performance regression when scanning pages. :issue:`1378`. Thanks @aliemjay.
|
||||||
|
- Fixed Alpine Docker image by enforcing Alpine 3.19. Alpine 3.20 includes a
|
||||||
|
defective version of Tesseract OCR and so is not usable.
|
||||||
|
- Upgraded Ubuntu Docker image to use Ubuntu 24.04.
|
||||||
|
- Build and test scripts/actions switched to uv.
|
||||||
|
- When running in a container, we now remind the user that temporary folders
|
||||||
|
are inside the container and may not be accessible.
|
||||||
|
- Fixed Linux test coverage matrix, which was missing some key versions.
|
||||||
|
|
||||||
|
v16.5.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed issue with interpreting PDFs that have images with array masks.
|
||||||
|
:issue:`1377`
|
||||||
|
- Enabled testing on Python 3.13.
|
||||||
|
- Fixed a test that did not work correctly but still passed. :issue:`1382`
|
||||||
|
- Improved "PDF/A conversion failed" warning message to better describe implications.
|
||||||
|
- Updated documentation to better explain OCR_JSON_SETTINGS in batch processing.
|
||||||
|
- Build backend changed from setuptools to hatchling.
|
||||||
|
|
||||||
|
v16.4.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Work around pdfminer.six issue where a token on the buffer boundary is incorrectly
|
||||||
|
parsed as two tokens. :issue:`1361`
|
||||||
|
- New rules are applied to stencil masks and explicit masks when calculating the
|
||||||
|
optimal page DPI for rendering. :issue:`1362`
|
||||||
|
- Fixed attempts to use an incompatible jbig2.EXE provided by TeX Live. :issue:`1363`
|
||||||
|
|
||||||
|
v16.4.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed order of filenames passed to Ghostscript for PDF/A generation. :issue:`1359`
|
||||||
|
- Suppressed missing jbig2dec warning message. :issue:`1358`
|
||||||
|
- Fixed calculation of image size when soft mask dimensions don't match image
|
||||||
|
dimension. :issue:`1351`
|
||||||
|
- Several fixes to documentation. Thanks to users Iris and JoKalliauer
|
||||||
|
who contributed these changes.
|
||||||
|
- Fixed error on processing PDFs that are missing certain image metadata. :issue:`1315`
|
||||||
|
|
||||||
|
v16.4.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed calculation of image printed area (used in finding weighted DPI for OCR).
|
||||||
|
:issue:`1334`
|
||||||
|
- Fixed "NotImplementedError: not sure how to get colorspace" error
|
||||||
|
messages in logs which simply records a failure to optimize images with
|
||||||
|
print production colorspaces. :issue:`1315`
|
||||||
|
|
||||||
|
v16.4.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Selecting the ``osd`` and ``equ`` pseudo-languages with ``-l/--language`` now
|
||||||
|
exits with an error when using Tesseract OCR, because these are not
|
||||||
|
regular Tesseract languages but implementation details implemented.
|
||||||
|
Using them can cause Tesseract to crash.
|
||||||
|
- The hOCR renderer is more tolerant of extra whitespace in input files.
|
||||||
|
- watcher.py now changes the output file extension to .pdf when the input is not
|
||||||
|
.pdf.
|
||||||
|
- Improved handling of PDFs that contain circularly referenced Form XObjects.
|
||||||
|
:issue:`1321`
|
||||||
|
- Fixed Alpine Docker image for ARM64, which was not building correctly.
|
||||||
|
- Docker images now use pikepdf 9.0.0.
|
||||||
|
- Prevent use of Tesseract OCR 5.4.0, a version with known regressions.
|
||||||
|
- Disabled progressbar for "Linearizing" when ``--no-progress-bar`` set.
|
||||||
|
- Fixed some tests that warn about missing JBIG2 decoding via pikepdf, by
|
||||||
|
installing the necessary libraries during tests.
|
||||||
|
|
||||||
|
v16.3.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed a test suite failure with Ghostscript 10.03.0+. :issue:`1316`
|
||||||
|
- Fixed an issue with the presentation of the "OCR" progress bar. :issue:`1313`
|
||||||
|
|
||||||
|
v16.3.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed progress bar not displaying for Ghostscript PDF/A conversion. :issue:`1313`
|
||||||
|
- Added progress bar for linearization. :issue:`1313`
|
||||||
|
- If `--rotate-pages-threshold` issued without `--rotate-pages` we now exit with
|
||||||
|
an error since the user likely intended to use `--rotate-pages`. :issue:`1309`
|
||||||
|
- If Tesseract hOCR gives an invalid line box, print an error message instead of
|
||||||
|
exiting with an error. :issue:`1312`
|
||||||
|
|
||||||
|
v16.2.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed issue 'NoneType' object has no attribute 'get' when optimizing certain PDFs.
|
||||||
|
:issue:`1293,1271`
|
||||||
|
- Switched formatting from black to ruff.
|
||||||
|
- Added support for sending sidecar output to io.BytesIO.
|
||||||
|
- Added support for converting HEIF/HEIC images (the native image of iPhones and
|
||||||
|
some other devices) to PDFs, when the appropriate pi-hief library is installed.
|
||||||
|
This library is marked as a dependency, but maintainers may opt out if needed.
|
||||||
|
- We now default to downsampling large images that would exceed Tesseract's internal
|
||||||
|
limits, but only if it cause processing to fail. Previously, this behavior only
|
||||||
|
occurred if specifically requested on command line. It can still be configured
|
||||||
|
and disabled. See the --tesseract command line options.
|
||||||
|
- Added Macports install instructions. Thanks @akierig.
|
||||||
|
- Improved logging output when an unexpected error occurs while trying to obtain
|
||||||
|
the version of a third party program.
|
||||||
|
|
||||||
|
v16.1.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed test suite failure when using Ghostscript 10.3.
|
||||||
|
- Other minor corrections.
|
||||||
|
|
||||||
|
v16.1.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed PyPy 3.10 support.
|
||||||
|
|
||||||
|
v16.1.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Improved hOCR renderer is now default for left to right languages.
|
||||||
|
- Improved handling of rotated pages. Previously, OCR text might be missing for
|
||||||
|
pages that were rotated with a /Rotate tag on the page entry.
|
||||||
|
- Improved handling of cropped pages. Previously, in some cases a page with a
|
||||||
|
crop box would not have its OCR applied correctly and misalignment between
|
||||||
|
OCR text and visible text coudl occur.
|
||||||
|
- Documentation improvements, especially installation instructions for less
|
||||||
|
common platforms.
|
||||||
|
|
||||||
|
v16.0.4
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed some issues for left-to-right text with the new hOCR renderer. It is still
|
||||||
|
not default yet but will be made so soon. Right-to-left text is still in progress.
|
||||||
|
- Added an error to prevent use of several versions of Ghostscript that seem
|
||||||
|
corrupt existing text in input PDFs. Newly generated OCR is not affected.
|
||||||
|
For best results, use Ghostscript 10.02.1 or newer, which contains the fix
|
||||||
|
for the issue.
|
||||||
|
|
||||||
|
v16.0.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Changed minimum required Ghostscript to 9.54, to support users of RHEL 9 and its
|
||||||
|
derivatives, since that is the latest version available there.
|
||||||
|
- Removed warning message about CVE-2023-43115, on the assumption that most
|
||||||
|
distributions have backported the patch by now.
|
||||||
|
|
||||||
|
v16.0.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Temporarily changed PDF text renderer back to sandwich by default to address
|
||||||
|
regressions in macOS Preview.
|
||||||
|
|
||||||
|
v16.0.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed text rendering issue with new hOCR text renderer - extraneous byte order
|
||||||
|
marks.
|
||||||
|
- Tightened dependencies.
|
||||||
|
|
||||||
|
v16.0.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Added OCR text renderer, combined the best ideas of Tesseract's PDF
|
||||||
|
generator and the older hOCR transformer renderer. The result is a hopefully
|
||||||
|
permanent fix for wordssmushedtogetherwithoutspaces issues in extracted text,
|
||||||
|
better registration/position of text on skewed baselines :issue:`1009`,
|
||||||
|
fixes to character output when the German Fraktur script is used :issue:`1191`,
|
||||||
|
proper rendering of right to left languages (Arabic, Hebrew, Persian) :issue:`1157`.
|
||||||
|
Asian languages may still have excessive word breaks compared to expectations.
|
||||||
|
The new renderer is the default; the old sandwich renderer is still available
|
||||||
|
using ``--pdf-renderer sandwich``; the old hOCR renderer is no more.
|
||||||
|
- The ``ocrmypdf.hocrtransform`` API has changed substantially.
|
||||||
|
- Support for Python 3.9 has been dropped. Python 3.10+ is now required.
|
||||||
|
- pikepdf >= 8.8.0 is now required.
|
||||||
|
|
||||||
|
|
||||||
|
v15.4.4
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed documentation for installing Ghostscript on Windows. :issue:`1198`
|
||||||
|
- Added warning message about security issue in older versions of Ghostscript.
|
||||||
|
|
||||||
|
v15.4.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed deprecation warning in pikepdf older than 8.7.1; pikepdf >= 8.7.1 is
|
||||||
|
now required.
|
||||||
|
|
||||||
|
v15.4.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- We now raise an exception on a certain class of PDFs that likely need an
|
||||||
|
explicit color conversion strategy selected to display correctly
|
||||||
|
for PDF/A conversion.
|
||||||
|
- Fixed an error that occurred while trying to write a log message after the
|
||||||
|
debug log handler was removed.
|
||||||
|
|
||||||
|
v15.4.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed misc/watcher.py regressions: accept ``--ocr-json-settings`` as either
|
||||||
|
filename or JSON string, as previously; and argument count mismatch.
|
||||||
|
:issue:`1183,1185`
|
||||||
|
- We no longer attempt to set /ProcSet in the PDF output, since this is an
|
||||||
|
obsolete PDF feature.
|
||||||
|
- Documentation improvements.
|
||||||
|
|
||||||
v15.4.0
|
v15.4.0
|
||||||
=======
|
=======
|
||||||
|
|
||||||
@@ -50,6 +279,8 @@ v15.4.0
|
|||||||
rather than fork, since this is method is more robust and avoids some
|
rather than fork, since this is method is more robust and avoids some
|
||||||
issues when threads are present.
|
issues when threads are present.
|
||||||
- Fixed an instance where the user's request to ``--no-use-threads`` was ignored.
|
- Fixed an instance where the user's request to ``--no-use-threads`` was ignored.
|
||||||
|
- If a PDF does not have language metadata on its top level object, we add
|
||||||
|
the OCR language.
|
||||||
- Replace some cryptic test error messages with more helpful ones.
|
- Replace some cryptic test error messages with more helpful ones.
|
||||||
- Debug messages for how OCRmyPDF picks the colorspace for a page are now
|
- Debug messages for how OCRmyPDF picks the colorspace for a page are now
|
||||||
more descriptive.
|
more descriptive.
|
||||||
|
|||||||
+48
-10
@@ -1,5 +1,6 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# SPDX-FileCopyrightText: 2016 findingorder <https://github.com/findingorder>
|
# SPDX-FileCopyrightText: 2016 findingorder <https://github.com/findingorder>
|
||||||
|
# SPDX-FileCopyrightText: 2024 nilsro <https://github.com/nilsro>
|
||||||
# SPDX-License-Identifier: MIT
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
"""Example of using ocrmypdf as a library in a script.
|
"""Example of using ocrmypdf as a library in a script.
|
||||||
@@ -13,7 +14,11 @@ You should edit this script to meet your needs.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import filecmp
|
||||||
import logging
|
import logging
|
||||||
|
import os
|
||||||
|
import posixpath
|
||||||
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -22,32 +27,65 @@ import ocrmypdf
|
|||||||
# pylint: disable=logging-format-interpolation
|
# pylint: disable=logging-format-interpolation
|
||||||
# pylint: disable=logging-not-lazy
|
# pylint: disable=logging-not-lazy
|
||||||
|
|
||||||
|
|
||||||
|
def filecompare(a, b):
|
||||||
|
try:
|
||||||
|
return filecmp.cmp(a, b, shallow=True)
|
||||||
|
except FileNotFoundError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
script_dir = Path(__file__).parent
|
script_dir = Path(__file__).parent
|
||||||
|
# set archive_dir to a path for backup original documents. Leave empty if not required.
|
||||||
|
archive_dir = "/pdfbak"
|
||||||
|
|
||||||
if len(sys.argv) > 1:
|
if len(sys.argv) > 1:
|
||||||
start_dir = Path(sys.argv[1])
|
start_dir = Path(sys.argv[1])
|
||||||
else:
|
else:
|
||||||
start_dir = Path('.')
|
start_dir = Path(".")
|
||||||
|
|
||||||
if len(sys.argv) > 2:
|
if len(sys.argv) > 2:
|
||||||
log_file = Path(sys.argv[2])
|
log_file = Path(sys.argv[2])
|
||||||
else:
|
else:
|
||||||
log_file = script_dir.with_name('ocr-tree.log')
|
log_file = script_dir.with_name("ocr-tree.log")
|
||||||
|
|
||||||
logging.basicConfig(
|
logging.basicConfig(
|
||||||
level=logging.INFO,
|
level=logging.INFO,
|
||||||
format='%(asctime)s %(message)s',
|
format="%(asctime)s %(message)s",
|
||||||
filename=log_file,
|
filename=log_file,
|
||||||
filemode='a',
|
filemode="a",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
logging.info(f"Start directory {start_dir}")
|
||||||
|
|
||||||
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
||||||
|
|
||||||
for filename in start_dir.glob("**/*.py"):
|
for filename in start_dir.glob("**/*.pdf"):
|
||||||
logging.info(f"Processing {filename}")
|
logging.info(f"Processing {filename}")
|
||||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
if ocrmypdf.pdfa.file_claims_pdfa(filename)["pass"]:
|
||||||
if result == ocrmypdf.ExitCode.already_done_ocr:
|
logging.info("Skipped document because it already contained text")
|
||||||
logging.error("Skipped document because it already contained text")
|
else:
|
||||||
elif result == ocrmypdf.ExitCode.ok:
|
archive_filename = archive_dir + str(filename)
|
||||||
|
if len(archive_dir) > 0 and not filecompare(filename, archive_filename):
|
||||||
|
logging.info(f"Archiving document to {archive_filename}")
|
||||||
|
try:
|
||||||
|
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||||
|
except OSError:
|
||||||
|
os.makedirs(posixpath.dirname(archive_filename))
|
||||||
|
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||||
|
try:
|
||||||
|
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||||
|
logging.info(result)
|
||||||
|
except ocrmypdf.exceptions.EncryptedPdfError:
|
||||||
|
logging.info("Skipped document because it is encrypted")
|
||||||
|
except ocrmypdf.exceptions.PriorOcrFoundError:
|
||||||
|
logging.info("Skipped document because it already contained text")
|
||||||
|
except ocrmypdf.exceptions.DigitalSignatureError:
|
||||||
|
logging.info("Skipped document because it has a digital signature")
|
||||||
|
except ocrmypdf.exceptions.TaggedPDFError:
|
||||||
|
logging.info(
|
||||||
|
"Skipped document because it does not need ocr as it is tagged"
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
logging.error("Unhandled error occured")
|
||||||
logging.info("OCR complete")
|
logging.info("OCR complete")
|
||||||
logging.info(result)
|
|
||||||
|
|||||||
@@ -0,0 +1,42 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
"""Helper script for bisecting PDFs to find a page with an issue."""
|
||||||
|
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import pikepdf
|
||||||
|
|
||||||
|
if len(sys.argv) != 2:
|
||||||
|
print(f"Usage: {sys.argv[0]} <input.pdf>")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
with pikepdf.open(sys.argv[1]) as pdf:
|
||||||
|
num_pages = len(pdf.pages)
|
||||||
|
low = 0
|
||||||
|
high = num_pages - 1
|
||||||
|
while low <= high:
|
||||||
|
mid = (low + high) // 2
|
||||||
|
with pikepdf.new() as new_pdf:
|
||||||
|
new_pdf.pages.extend(pdf.pages[low : mid + 1])
|
||||||
|
new_pdf.save(f"bisect-issue-{low + 1}-{mid + 1}.pdf")
|
||||||
|
print(f"Is bisect-issue-{low + 1}-{mid + 1}.pdf good or bad?", end=" ")
|
||||||
|
while True:
|
||||||
|
response = input().lower()
|
||||||
|
if response == "good":
|
||||||
|
low = mid + 1
|
||||||
|
break
|
||||||
|
elif response == "bad":
|
||||||
|
high = mid - 1
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
print("Please respond with 'good' or 'bad'.")
|
||||||
|
print(f"The issue is on page {low + 1} of the original PDF.")
|
||||||
|
with pikepdf.new() as new_pdf:
|
||||||
|
new_pdf.pages.extend(pdf.pages[low])
|
||||||
|
new_pdf.save(f"bisect-issue-bad-{low + 1}.pdf")
|
||||||
|
with pikepdf.new() as new_pdf:
|
||||||
|
new_pdf.pages.extend(pdf.pages[:low])
|
||||||
|
new_pdf.pages.extend(pdf.pages[low + 1 :])
|
||||||
|
new_pdf.save(f"bisect-issue-good-{low + 1}.pdf")
|
||||||
+4
-3
@@ -53,9 +53,10 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
|||||||
]
|
]
|
||||||
logging.info(cmd)
|
logging.info(cmd)
|
||||||
full_path_ocr = os.path.join(dir_name, filename_ocr)
|
full_path_ocr = os.path.join(dir_name, filename_ocr)
|
||||||
with open(filename, 'rb') as input_file, open(
|
with (
|
||||||
full_path_ocr, 'wb'
|
open(filename, 'rb') as input_file,
|
||||||
) as output_file:
|
open(full_path_ocr, 'wb') as output_file,
|
||||||
|
):
|
||||||
proc = subprocess.run(
|
proc = subprocess.run(
|
||||||
cmd,
|
cmd,
|
||||||
stdin=input_file,
|
stdin=input_file,
|
||||||
|
|||||||
+22
-15
@@ -7,7 +7,6 @@
|
|||||||
|
|
||||||
# Do not enable annotations!
|
# Do not enable annotations!
|
||||||
# https://github.com/tiangolo/typer/discussions/598
|
# https://github.com/tiangolo/typer/discussions/598
|
||||||
# from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
@@ -47,15 +46,18 @@ class LoggingLevelEnum(str, Enum):
|
|||||||
CRITICAL = "CRITICAL"
|
CRITICAL = "CRITICAL"
|
||||||
|
|
||||||
|
|
||||||
def get_output_dir(root: Path, basename: str, output_dir_year_month: bool) -> Path:
|
def get_output_path(root: Path, basename: str, output_dir_year_month: bool) -> Path:
|
||||||
|
assert '/' not in basename, "basename must not contain '/'"
|
||||||
if output_dir_year_month:
|
if output_dir_year_month:
|
||||||
today = datetime.today()
|
today = datetime.today()
|
||||||
output_directory_year_month = root / str(today.year) / f'{today.month:02d}'
|
output_directory_year_month = root / str(today.year) / f'{today.month:02d}'
|
||||||
if not output_directory_year_month.exists():
|
if not output_directory_year_month.exists():
|
||||||
output_directory_year_month.mkdir(parents=True, exist_ok=True)
|
output_directory_year_month.mkdir(parents=True, exist_ok=True)
|
||||||
output_path = Path(output_directory_year_month) / basename
|
output_path = Path(output_directory_year_month) / Path(basename).with_suffix(
|
||||||
|
'.pdf'
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
output_path = root / basename
|
output_path = root / Path(basename).with_suffix('.pdf')
|
||||||
return output_path
|
return output_path
|
||||||
|
|
||||||
|
|
||||||
@@ -92,7 +94,6 @@ def execute_ocrmypdf(
|
|||||||
file_path: Path,
|
file_path: Path,
|
||||||
archive_dir: Path,
|
archive_dir: Path,
|
||||||
output_dir: Path,
|
output_dir: Path,
|
||||||
deskew: bool,
|
|
||||||
ocrmypdf_kwargs: dict[str, Any],
|
ocrmypdf_kwargs: dict[str, Any],
|
||||||
on_success_delete: bool,
|
on_success_delete: bool,
|
||||||
on_success_archive: bool,
|
on_success_archive: bool,
|
||||||
@@ -100,7 +101,7 @@ def execute_ocrmypdf(
|
|||||||
retries_loading_file: int,
|
retries_loading_file: int,
|
||||||
output_dir_year_month: bool,
|
output_dir_year_month: bool,
|
||||||
):
|
):
|
||||||
output_path = get_output_dir(output_dir, file_path.name, output_dir_year_month)
|
output_path = get_output_path(output_dir, file_path.name, output_dir_year_month)
|
||||||
|
|
||||||
log.info("-" * 20)
|
log.info("-" * 20)
|
||||||
log.info(f'New file: {file_path}. Waiting until fully written...')
|
log.info(f'New file: {file_path}. Waiting until fully written...')
|
||||||
@@ -108,10 +109,14 @@ def execute_ocrmypdf(
|
|||||||
log.info(f"Gave up waiting for {file_path} to become ready")
|
log.info(f"Gave up waiting for {file_path} to become ready")
|
||||||
return
|
return
|
||||||
log.info(f'Attempting to OCRmyPDF to: {output_path}')
|
log.info(f'Attempting to OCRmyPDF to: {output_path}')
|
||||||
|
|
||||||
|
log.debug(
|
||||||
|
f'OCRmyPDF input_file={file_path} output_file={output_path} '
|
||||||
|
f'kwargs: {ocrmypdf_kwargs}'
|
||||||
|
)
|
||||||
exit_code = ocrmypdf.ocr(
|
exit_code = ocrmypdf.ocr(
|
||||||
input_file=file_path,
|
input_file=file_path,
|
||||||
output_file=output_path,
|
output_file=output_path,
|
||||||
deskew=deskew,
|
|
||||||
**ocrmypdf_kwargs,
|
**ocrmypdf_kwargs,
|
||||||
)
|
)
|
||||||
if exit_code == 0:
|
if exit_code == 0:
|
||||||
@@ -128,7 +133,7 @@ def execute_ocrmypdf(
|
|||||||
|
|
||||||
|
|
||||||
class HandleObserverEvent(PatternMatchingEventHandler):
|
class HandleObserverEvent(PatternMatchingEventHandler):
|
||||||
def __init__(
|
def __init__( # noqa: D107
|
||||||
self,
|
self,
|
||||||
patterns=None,
|
patterns=None,
|
||||||
ignore_patterns=None,
|
ignore_patterns=None,
|
||||||
@@ -146,7 +151,7 @@ class HandleObserverEvent(PatternMatchingEventHandler):
|
|||||||
|
|
||||||
def on_any_event(self, event):
|
def on_any_event(self, event):
|
||||||
if event.event_type in ['created']:
|
if event.event_type in ['created']:
|
||||||
execute_ocrmypdf(event.src_path, **self._settings)
|
execute_ocrmypdf(file_path=Path(event.src_path), **self._settings)
|
||||||
|
|
||||||
|
|
||||||
@app.command()
|
@app.command()
|
||||||
@@ -188,7 +193,7 @@ def main(
|
|||||||
bool,
|
bool,
|
||||||
typer.Option(
|
typer.Option(
|
||||||
envvar='OCR_OUTPUT_DIRECTORY_YEAR_MONTH',
|
envvar='OCR_OUTPUT_DIRECTORY_YEAR_MONTH',
|
||||||
help='Create a subdirectory in the output directory for each year and month',
|
help='Create a subdirectory in the output directory for each year/month',
|
||||||
),
|
),
|
||||||
] = False,
|
] = False,
|
||||||
on_success_delete: Annotated[
|
on_success_delete: Annotated[
|
||||||
@@ -213,10 +218,10 @@ def main(
|
|||||||
),
|
),
|
||||||
] = False,
|
] = False,
|
||||||
ocr_json_settings: Annotated[
|
ocr_json_settings: Annotated[
|
||||||
typer.FileText,
|
str,
|
||||||
typer.Option(
|
typer.Option(
|
||||||
envvar='OCR_JSON_SETTINGS',
|
envvar='OCR_JSON_SETTINGS',
|
||||||
help='JSON settings to pass to OCRmyPDF',
|
help='JSON settings to pass to OCRmyPDF (JSON string or file path)',
|
||||||
),
|
),
|
||||||
] = None,
|
] = None,
|
||||||
poll_new_file_seconds: Annotated[
|
poll_new_file_seconds: Annotated[
|
||||||
@@ -288,7 +293,10 @@ def main(
|
|||||||
f"LOGLEVEL: {loglevel.value}"
|
f"LOGLEVEL: {loglevel.value}"
|
||||||
)
|
)
|
||||||
|
|
||||||
json_settings = json.loads(ocr_json_settings.read() if ocr_json_settings else '{}')
|
if ocr_json_settings and Path(ocr_json_settings).exists():
|
||||||
|
json_settings = json.loads(Path(ocr_json_settings).read_text())
|
||||||
|
else:
|
||||||
|
json_settings = json.loads(ocr_json_settings or '{}')
|
||||||
|
|
||||||
if 'input_file' in json_settings or 'output_file' in json_settings:
|
if 'input_file' in json_settings or 'output_file' in json_settings:
|
||||||
log.error(
|
log.error(
|
||||||
@@ -301,8 +309,7 @@ def main(
|
|||||||
settings={
|
settings={
|
||||||
'archive_dir': archive_dir,
|
'archive_dir': archive_dir,
|
||||||
'output_dir': output_dir,
|
'output_dir': output_dir,
|
||||||
'deskew': deskew,
|
'ocrmypdf_kwargs': json_settings | {'deskew': deskew},
|
||||||
'ocrmypdf_kwargs': json_settings,
|
|
||||||
'on_success_delete': on_success_delete,
|
'on_success_delete': on_success_delete,
|
||||||
'on_success_archive': on_success_archive,
|
'on_success_archive': on_success_archive,
|
||||||
'poll_new_file_seconds': poll_new_file_seconds,
|
'poll_new_file_seconds': poll_new_file_seconds,
|
||||||
|
|||||||
+34
-48
@@ -1,8 +1,8 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
[build-system]
|
[build-system]
|
||||||
requires = ["setuptools >= 61", "setuptools_scm[toml] >= 7.0.5", "wheel"]
|
requires = ["hatchling", "hatch-vcs"]
|
||||||
build-backend = "setuptools.build_meta"
|
build-backend = "hatchling.build"
|
||||||
|
|
||||||
[project]
|
[project]
|
||||||
name = "ocrmypdf"
|
name = "ocrmypdf"
|
||||||
@@ -10,18 +10,17 @@ dynamic = ["version"]
|
|||||||
description = "OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched"
|
description = "OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched"
|
||||||
readme = "README.md"
|
readme = "README.md"
|
||||||
license = { text = "MPL-2.0" }
|
license = { text = "MPL-2.0" }
|
||||||
requires-python = ">=3.9"
|
requires-python = ">=3.10"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"Pillow>=10.0.1",
|
|
||||||
"deprecation>=2.1.0",
|
"deprecation>=2.1.0",
|
||||||
"img2pdf>=0.4.4",
|
"img2pdf>=0.5",
|
||||||
"packaging>=20",
|
"packaging>=20",
|
||||||
"pdfminer.six>=20220319",
|
"pdfminer.six>=20220319",
|
||||||
"pikepdf>=8",
|
"pi-heif", # Heif image format - maintainers: if this is removed, it will NOT break
|
||||||
"pluggy>=0.13.0",
|
"pikepdf>=8.10.1",
|
||||||
"reportlab>=3.6.8",
|
"Pillow>=10.0.1",
|
||||||
|
"pluggy>=1",
|
||||||
"rich>=13",
|
"rich>=13",
|
||||||
"typing-extensions>=4;python_version<'3.10'",
|
|
||||||
]
|
]
|
||||||
authors = [{ name = "James R. Barlow", email = "james@purplerock.ca" }]
|
authors = [{ name = "James R. Barlow", email = "james@purplerock.ca" }]
|
||||||
classifiers = [
|
classifiers = [
|
||||||
@@ -47,6 +46,7 @@ keywords = ["PDF", "OCR", "optical character recognition", "PDF/A", "scanning"]
|
|||||||
Documentation = "https://ocrmypdf.readthedocs.io/"
|
Documentation = "https://ocrmypdf.readthedocs.io/"
|
||||||
Source = "https://github.com/ocrmypdf/OCRmyPDF"
|
Source = "https://github.com/ocrmypdf/OCRmyPDF"
|
||||||
Tracker = "https://github.com/ocrmypdf/OCRmyPDF/issues"
|
Tracker = "https://github.com/ocrmypdf/OCRmyPDF/issues"
|
||||||
|
Changelog = "https://github.com/ocrmypdf/OCRmyPDF/docs/release_notes.rst"
|
||||||
|
|
||||||
[project.optional-dependencies]
|
[project.optional-dependencies]
|
||||||
docs = ["sphinx", "sphinx-issues", "sphinx-rtd-theme"]
|
docs = ["sphinx", "sphinx-issues", "sphinx-rtd-theme"]
|
||||||
@@ -58,49 +58,24 @@ test = [
|
|||||||
"pytest-cov>=3.0.0",
|
"pytest-cov>=3.0.0",
|
||||||
"pytest-xdist>=2.5.0",
|
"pytest-xdist>=2.5.0",
|
||||||
"python-xmp-toolkit==2.0.1", # also requires apt-get install libexempi3
|
"python-xmp-toolkit==2.0.1", # also requires apt-get install libexempi3
|
||||||
|
"reportlab>=3.6.8",
|
||||||
"types-Pillow",
|
"types-Pillow",
|
||||||
"types-humanfriendly",
|
"types-humanfriendly",
|
||||||
]
|
]
|
||||||
watcher = ["watchdog>=1.0.2", "typer[all]", "python-dotenv"]
|
watcher = ["watchdog>=1.0.2", "typer-slim[standard]", "python-dotenv"]
|
||||||
webservice = ["Flask>=2.0.1"]
|
webservice = ["Flask>=2.0.1"]
|
||||||
|
|
||||||
[project.scripts]
|
[project.scripts]
|
||||||
ocrmypdf = "ocrmypdf.__main__:run"
|
ocrmypdf = "ocrmypdf.__main__:run"
|
||||||
|
|
||||||
[tool.setuptools.package-data]
|
[tool.hatch.version]
|
||||||
ocrmypdf = ["data/sRGB.icc", "py.typed"]
|
source = "vcs"
|
||||||
|
|
||||||
[tool.setuptools.packages.find]
|
[tool.hatch.build.hooks.vcs]
|
||||||
where = ["src"]
|
version-file = "src/ocrmypdf/_version.py"
|
||||||
namespaces = false
|
|
||||||
|
|
||||||
[tool.setuptools_scm]
|
|
||||||
|
|
||||||
[tool.distutils.bdist_wheel]
|
[tool.distutils.bdist_wheel]
|
||||||
python-tag = "py39"
|
python-tag = "py310"
|
||||||
|
|
||||||
[tool.black]
|
|
||||||
line-length = 88
|
|
||||||
target-version = ["py39", "py310", "py311"]
|
|
||||||
skip-string-normalization = true
|
|
||||||
include = '\.pyi?$'
|
|
||||||
exclude = '''
|
|
||||||
/(
|
|
||||||
\.eggs
|
|
||||||
| \.git
|
|
||||||
| \.hg
|
|
||||||
| \.mypy_cache
|
|
||||||
| \.tox
|
|
||||||
| \.venv
|
|
||||||
| _build
|
|
||||||
| buck-out
|
|
||||||
| build
|
|
||||||
| dist
|
|
||||||
| docs
|
|
||||||
| misc
|
|
||||||
| \.egg-info
|
|
||||||
)/
|
|
||||||
'''
|
|
||||||
|
|
||||||
[tool.coverage.run]
|
[tool.coverage.run]
|
||||||
branch = true
|
branch = true
|
||||||
@@ -132,7 +107,11 @@ norecursedirs = ["lib", ".pc", ".git", "venv", "output", "cache", "resources"]
|
|||||||
testpaths = ["tests"]
|
testpaths = ["tests"]
|
||||||
addopts = "-n auto"
|
addopts = "-n auto"
|
||||||
markers = ["slow"]
|
markers = ["slow"]
|
||||||
filterwarnings = ["ignore:.*XMLParser.*:DeprecationWarning"]
|
filterwarnings = [
|
||||||
|
"ignore:.*XMLParser.*:DeprecationWarning",
|
||||||
|
"ignore:.*ast.NameConstant.*:DeprecationWarning:reportlab",
|
||||||
|
"ignore:.*distutils.*:DeprecationWarning:libxmp",
|
||||||
|
]
|
||||||
|
|
||||||
[tool.mypy]
|
[tool.mypy]
|
||||||
|
|
||||||
@@ -150,7 +129,10 @@ module = [
|
|||||||
ignore_missing_imports = true
|
ignore_missing_imports = true
|
||||||
|
|
||||||
[tool.ruff]
|
[tool.ruff]
|
||||||
select = [
|
target-version = "py310"
|
||||||
|
|
||||||
|
[tool.ruff.lint]
|
||||||
|
"select" = [
|
||||||
"D", # pydocstyle
|
"D", # pydocstyle
|
||||||
"E", # pycodestyle
|
"E", # pycodestyle
|
||||||
"W", # pycodestyle
|
"W", # pycodestyle
|
||||||
@@ -158,17 +140,21 @@ select = [
|
|||||||
"I001", # isort
|
"I001", # isort
|
||||||
"UP", # pyupgrade
|
"UP", # pyupgrade
|
||||||
]
|
]
|
||||||
target-version = "py39"
|
|
||||||
|
|
||||||
[tool.ruff.isort]
|
[tool.ruff.lint.isort]
|
||||||
known-first-party = ["ocrmypdf"]
|
known-first-party = ["ocrmypdf"]
|
||||||
required-imports = ["from __future__ import annotations"]
|
|
||||||
|
|
||||||
[tool.ruff.pydocstyle]
|
[tool.ruff.lint.pydocstyle]
|
||||||
convention = "google"
|
convention = "google"
|
||||||
|
|
||||||
[tool.ruff.per-file-ignores]
|
[tool.ruff.lint.per-file-ignores]
|
||||||
"docs/conf.py" = ["D100", "D101", "D105"]
|
"docs/conf.py" = ["D100", "D101", "D105"]
|
||||||
"tests/*.py" = ["D100", "D101", "D102", "D103", "D105"]
|
"tests/*.py" = ["D100", "D101", "D102", "D103", "D105"]
|
||||||
"misc/*.py" = ["D103", "D101", "D102"]
|
"misc/*.py" = ["D103", "D101", "D102"]
|
||||||
"src/ocrmypdf/builtin_plugins/*.py" = ["D103", "D102", "D105"]
|
"src/ocrmypdf/builtin_plugins/*.py" = ["D103", "D102", "D105"]
|
||||||
|
|
||||||
|
[tool.ruff.format]
|
||||||
|
quote-style = "preserve"
|
||||||
|
|
||||||
|
[dependency-groups]
|
||||||
|
dev = ["mypy>=1.13.0"]
|
||||||
|
|||||||
+2
-2
@@ -18,8 +18,8 @@ architectures: [amd64]
|
|||||||
|
|
||||||
environment:
|
environment:
|
||||||
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
||||||
GS_LIB: $SNAP/usr/share/ghostscript/9.55/Resource/Init
|
GS_LIB: $SNAP/usr/share/ghostscript/9.55.0/Resource/Init
|
||||||
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.55/Resource/Font
|
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.55.0/Resource/Font
|
||||||
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
||||||
|
|
||||||
apps:
|
apps:
|
||||||
|
|||||||
@@ -9,11 +9,12 @@ from pluggy import HookimplMarker as _HookimplMarker
|
|||||||
|
|
||||||
from ocrmypdf import helpers, hocrtransform, pdfa, pdfinfo
|
from ocrmypdf import helpers, hocrtransform, pdfa, pdfinfo
|
||||||
from ocrmypdf._concurrent import Executor
|
from ocrmypdf._concurrent import Executor
|
||||||
|
from ocrmypdf._defaults import PROGRAM_NAME
|
||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._pipelines._common import (
|
from ocrmypdf._pipelines._common import (
|
||||||
configure_debug_logging,
|
configure_debug_logging,
|
||||||
)
|
)
|
||||||
from ocrmypdf._version import PROGRAM_NAME, __version__
|
from ocrmypdf._version import __version__
|
||||||
from ocrmypdf.api import (
|
from ocrmypdf.api import (
|
||||||
Verbosity,
|
Verbosity,
|
||||||
configure_logging,
|
configure_logging,
|
||||||
@@ -37,7 +38,6 @@ from ocrmypdf.pluginspec import OcrEngine, OrientationConfidence
|
|||||||
|
|
||||||
hookimpl = _HookimplMarker('ocrmypdf')
|
hookimpl = _HookimplMarker('ocrmypdf')
|
||||||
|
|
||||||
|
|
||||||
__all__ = [
|
__all__ = [
|
||||||
'__version__',
|
'__version__',
|
||||||
'BadArgsError',
|
'BadArgsError',
|
||||||
|
|||||||
@@ -0,0 +1,66 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
"""OCRmyPDF PDF annotation cleanup."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from pikepdf import Dictionary, Name, NameTree, Pdf
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def remove_broken_goto_annotations(pdf: Pdf) -> bool:
|
||||||
|
"""Remove broken goto annotations from a PDF.
|
||||||
|
|
||||||
|
If a PDF contains a GoTo Action that points to a named destination that does not
|
||||||
|
exist, Ghostscript PDF/A conversion will fail. In any event, a named destination
|
||||||
|
that is not defined is not useful.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
pdf: Opened PDF file.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True if the file was modified, False if not.
|
||||||
|
"""
|
||||||
|
modified = False
|
||||||
|
|
||||||
|
# Check if there are any named destinations
|
||||||
|
if Name.Names not in pdf.Root:
|
||||||
|
return modified
|
||||||
|
if Name.Dests not in pdf.Root[Name.Names]:
|
||||||
|
return modified
|
||||||
|
|
||||||
|
dests = pdf.Root[Name.Names][Name.Dests]
|
||||||
|
if not isinstance(dests, Dictionary):
|
||||||
|
return modified
|
||||||
|
nametree = NameTree(dests)
|
||||||
|
|
||||||
|
# Create a set of all named destinations
|
||||||
|
names = set(k for k in nametree.keys())
|
||||||
|
|
||||||
|
for n, page in enumerate(pdf.pages):
|
||||||
|
if Name.Annots not in page:
|
||||||
|
continue
|
||||||
|
for annot in page[Name.Annots]:
|
||||||
|
if not isinstance(annot, Dictionary):
|
||||||
|
continue
|
||||||
|
if Name.A not in annot or Name.D not in annot[Name.A]:
|
||||||
|
continue
|
||||||
|
# We found an annotation that points to a named destination
|
||||||
|
named_destination = str(annot[Name.A][Name.D])
|
||||||
|
if named_destination not in names:
|
||||||
|
# If there is no corresponding named destination, remove the
|
||||||
|
# annotation. Having no destination set is still valid and just
|
||||||
|
# makes the link non-functional.
|
||||||
|
log.warning(
|
||||||
|
f"Disabling a hyperlink annotation on page {n + 1} to a "
|
||||||
|
"non-existent named destination "
|
||||||
|
f"{named_destination}."
|
||||||
|
)
|
||||||
|
del annot[Name.A][Name.D]
|
||||||
|
modified = True
|
||||||
|
|
||||||
|
return modified
|
||||||
@@ -7,8 +7,8 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import threading
|
import threading
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
from collections.abc import Iterable
|
from collections.abc import Callable, Iterable
|
||||||
from typing import Callable, TypeVar
|
from typing import Any, TypeVar
|
||||||
|
|
||||||
from ocrmypdf._progressbar import NullProgressBar, ProgressBar
|
from ocrmypdf._progressbar import NullProgressBar, ProgressBar
|
||||||
|
|
||||||
@@ -19,6 +19,10 @@ def _task_noop(*_args, **_kwargs):
|
|||||||
return
|
return
|
||||||
|
|
||||||
|
|
||||||
|
def _task_finished_noop(_result: Any, pbar: ProgressBar):
|
||||||
|
pbar.update()
|
||||||
|
|
||||||
|
|
||||||
class Executor(ABC):
|
class Executor(ABC):
|
||||||
"""Abstract concurrent executor."""
|
"""Abstract concurrent executor."""
|
||||||
|
|
||||||
@@ -66,7 +70,7 @@ class Executor(ABC):
|
|||||||
if not worker_initializer:
|
if not worker_initializer:
|
||||||
worker_initializer = _task_noop
|
worker_initializer = _task_noop
|
||||||
if not task_finished:
|
if not task_finished:
|
||||||
task_finished = _task_noop
|
task_finished = _task_finished_noop
|
||||||
if not task:
|
if not task:
|
||||||
task = _task_noop
|
task = _task_noop
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,10 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
# Enforce English hegemony
|
||||||
|
DEFAULT_LANGUAGE = 'eng'
|
||||||
|
|
||||||
|
# Default rotation threshold
|
||||||
|
DEFAULT_ROTATE_PAGES_THRESHOLD = 14.0
|
||||||
|
|
||||||
|
PROGRAM_NAME = 'OCRmyPDF'
|
||||||
@@ -8,6 +8,7 @@ from __future__ import annotations
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
|
from collections import deque
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -16,7 +17,7 @@ from subprocess import PIPE, CalledProcessError
|
|||||||
from packaging.version import Version
|
from packaging.version import Version
|
||||||
from PIL import Image, UnidentifiedImageError
|
from PIL import Image, UnidentifiedImageError
|
||||||
|
|
||||||
from ocrmypdf.exceptions import SubprocessOutputError
|
from ocrmypdf.exceptions import ColorConversionNeededError, SubprocessOutputError
|
||||||
from ocrmypdf.helpers import Resolution
|
from ocrmypdf.helpers import Resolution
|
||||||
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
||||||
|
|
||||||
@@ -29,38 +30,44 @@ COLOR_CONVERSION_STRATEGIES = frozenset(
|
|||||||
'UseDeviceIndependentColor',
|
'UseDeviceIndependentColor',
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
# Ghostscript executable - gswin32c is not supported
|
||||||
|
GS = 'gswin64c' if os.name == 'nt' else 'gs'
|
||||||
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
class DuplicateFilter(logging.Filter):
|
class DuplicateFilter(logging.Filter):
|
||||||
"""Filter out duplicate log messages."""
|
"""Filter out duplicate log messages.
|
||||||
|
|
||||||
def __init__(self, logger: logging.Logger):
|
A context window of default 5 messages is used to determine if a message is a
|
||||||
self.last: logging.LogRecord | None = None
|
duplicate. This is because some Ghostscript messages are word wrapped.
|
||||||
self.count = 0
|
"""
|
||||||
|
|
||||||
|
def __init__(self, logger: logging.Logger, context_window=5):
|
||||||
|
self.window: deque[str] = deque([], maxlen=context_window)
|
||||||
self.logger = logger
|
self.logger = logger
|
||||||
|
self.levelno = logging.DEBUG
|
||||||
|
self.count = 0
|
||||||
|
|
||||||
def filter(self, record):
|
def filter(self, record):
|
||||||
if self.last and record.msg == self.last.msg:
|
if record.msg in self.window:
|
||||||
self.count += 1
|
self.count += 1
|
||||||
|
self.levelno = record.levelno
|
||||||
return False
|
return False
|
||||||
else:
|
else:
|
||||||
if self.count >= 1:
|
if self.count >= 1:
|
||||||
rep_msg = f"(previous message repeated {self.count} times)"
|
rep_msg = f"(suppressed {self.count} repeated lines)"
|
||||||
self.count = 0 # Avoid infinite recursion
|
self.count = 0 # Avoid infinite recursion
|
||||||
self.logger.log(self.last.levelno, rep_msg)
|
self.logger.log(self.levelno, rep_msg)
|
||||||
self.last = record
|
self.window.clear()
|
||||||
|
self.window.append(record.msg)
|
||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
log.addFilter(DuplicateFilter(log))
|
log.addFilter(DuplicateFilter(log))
|
||||||
|
|
||||||
|
|
||||||
# Ghostscript executable - gswin32c is not supported
|
|
||||||
GS = 'gswin64c' if os.name == 'nt' else 'gs'
|
|
||||||
|
|
||||||
|
|
||||||
def version() -> Version:
|
def version() -> Version:
|
||||||
return Version(get_version(GS))
|
return Version(get_version(GS))
|
||||||
|
|
||||||
@@ -70,6 +77,20 @@ def _gs_error_reported(stream) -> bool:
|
|||||||
return bool(match)
|
return bool(match)
|
||||||
|
|
||||||
|
|
||||||
|
def _gs_devicen_reported(stream) -> bool:
|
||||||
|
"""Did Ghostscript warn about a DeviceN with inappropriate alternate?
|
||||||
|
|
||||||
|
If so, we need the user to select a color conversion, or the resulting PDF will
|
||||||
|
not present correctly in some PDF viewers.
|
||||||
|
"""
|
||||||
|
match = re.search(
|
||||||
|
r'DeviceN.*inappropriate alternate',
|
||||||
|
stream,
|
||||||
|
flags=re.IGNORECASE | re.MULTILINE,
|
||||||
|
)
|
||||||
|
return bool(match)
|
||||||
|
|
||||||
|
|
||||||
def rasterize_pdf(
|
def rasterize_pdf(
|
||||||
input_file: os.PathLike,
|
input_file: os.PathLike,
|
||||||
output_file: os.PathLike,
|
output_file: os.PathLike,
|
||||||
@@ -156,6 +177,17 @@ class GhostscriptFollower:
|
|||||||
self.progressbar_class = progressbar_class
|
self.progressbar_class = progressbar_class
|
||||||
self.progressbar = None
|
self.progressbar = None
|
||||||
|
|
||||||
|
def __enter__(self):
|
||||||
|
# We can't actually set up the progressbar here, because we don't know
|
||||||
|
# how many pages there are until the first __call__() happens. So we
|
||||||
|
# do it in __call__().
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
|
if self.progressbar:
|
||||||
|
return self.progressbar.__exit__(exc_type, exc_value, traceback)
|
||||||
|
return False
|
||||||
|
|
||||||
def __call__(self, line):
|
def __call__(self, line):
|
||||||
if not self.progressbar_class:
|
if not self.progressbar_class:
|
||||||
return
|
return
|
||||||
@@ -166,7 +198,8 @@ class GhostscriptFollower:
|
|||||||
self.progressbar = self.progressbar_class(
|
self.progressbar = self.progressbar_class(
|
||||||
total=self.count, desc="PDF/A conversion", unit='page'
|
total=self.count, desc="PDF/A conversion", unit='page'
|
||||||
)
|
)
|
||||||
return
|
# Now that we know the count, we can set up the progressbar.
|
||||||
|
self.progressbar.__enter__()
|
||||||
else:
|
else:
|
||||||
if self.re_page.match(line.strip()):
|
if self.re_page.match(line.strip()):
|
||||||
self.progressbar.update()
|
self.progressbar.update()
|
||||||
@@ -243,9 +276,11 @@ def generate_pdfa(
|
|||||||
]
|
]
|
||||||
)
|
)
|
||||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||||
|
|
||||||
try:
|
try:
|
||||||
with Path(output_file).open('wb') as output:
|
with (
|
||||||
|
Path(output_file).open('wb') as output,
|
||||||
|
GhostscriptFollower(progressbar_class) as pbar,
|
||||||
|
):
|
||||||
p = run_polling_stderr(
|
p = run_polling_stderr(
|
||||||
args_gs,
|
args_gs,
|
||||||
stdout=output,
|
stdout=output,
|
||||||
@@ -254,7 +289,7 @@ def generate_pdfa(
|
|||||||
text=True,
|
text=True,
|
||||||
encoding='utf-8',
|
encoding='utf-8',
|
||||||
errors='replace',
|
errors='replace',
|
||||||
callback=GhostscriptFollower(progressbar_class),
|
callback=pbar,
|
||||||
)
|
)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
# Ghostscript does not change return code when it fails to create
|
# Ghostscript does not change return code when it fails to create
|
||||||
@@ -272,3 +307,5 @@ def generate_pdfa(
|
|||||||
# the **** pattern to split the stderr into parts.
|
# the **** pattern to split the stderr into parts.
|
||||||
for part in stderr.split('****'):
|
for part in stderr.split('****'):
|
||||||
log.error(part)
|
log.error(part)
|
||||||
|
if _gs_devicen_reported(stderr):
|
||||||
|
raise ColorConversionNeededError()
|
||||||
|
|||||||
@@ -5,7 +5,7 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from subprocess import PIPE
|
from subprocess import PIPE, CalledProcessError
|
||||||
|
|
||||||
from packaging.version import Version
|
from packaging.version import Version
|
||||||
|
|
||||||
@@ -14,7 +14,13 @@ from ocrmypdf.subprocess import get_version, run
|
|||||||
|
|
||||||
|
|
||||||
def version() -> Version:
|
def version() -> Version:
|
||||||
return Version(get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*'))
|
try:
|
||||||
|
version = get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||||
|
except CalledProcessError as e:
|
||||||
|
# TeX Live for Windows provides an incompatible jbig2.EXE which may
|
||||||
|
# be on the PATH.
|
||||||
|
raise MissingDependencyError('jbig2enc') from e
|
||||||
|
return Version(version)
|
||||||
|
|
||||||
|
|
||||||
def available():
|
def available():
|
||||||
|
|||||||
@@ -220,7 +220,8 @@ def get_deskew(
|
|||||||
|
|
||||||
def tesseract_log_output(stream: bytes) -> None:
|
def tesseract_log_output(stream: bytes) -> None:
|
||||||
tlog = TesseractLoggerAdapter(
|
tlog = TesseractLoggerAdapter(
|
||||||
log, extra=log.extra if hasattr(log, 'extra') else None # type: ignore
|
log,
|
||||||
|
extra=log.extra if hasattr(log, 'extra') else None, # type: ignore
|
||||||
)
|
)
|
||||||
|
|
||||||
if not stream:
|
if not stream:
|
||||||
|
|||||||
@@ -8,48 +8,26 @@ from __future__ import annotations
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import shlex
|
import shlex
|
||||||
import sys
|
|
||||||
from collections.abc import Iterator
|
from collections.abc import Iterator
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT
|
from subprocess import PIPE, STDOUT
|
||||||
from typing import Union
|
from tempfile import TemporaryDirectory
|
||||||
|
|
||||||
from packaging.version import Version
|
from packaging.version import Version
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
from ocrmypdf.exceptions import SubprocessOutputError
|
||||||
from ocrmypdf.subprocess import get_version, run
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
# unpaper documentation:
|
# unpaper documentation:
|
||||||
# https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md
|
# https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md
|
||||||
|
|
||||||
|
|
||||||
if sys.version_info >= (3, 10):
|
|
||||||
from tempfile import TemporaryDirectory
|
|
||||||
else:
|
|
||||||
from tempfile import TemporaryDirectory as _TemporaryDirectory
|
|
||||||
|
|
||||||
class TemporaryDirectory(_TemporaryDirectory):
|
|
||||||
"""Shim to consume ignore_cleanup_errors kwarg on Python 3.9 and older.
|
|
||||||
|
|
||||||
The argument is consumed without action. If users are getting errors related
|
|
||||||
to temporary file cleanup, they should upgrade to Python 3.10 which properly
|
|
||||||
cleans up temporary directories on Windows.
|
|
||||||
|
|
||||||
See: https://github.com/python/cpython/pull/24793
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, ignore_cleanup_errors=False, **kwargs):
|
|
||||||
super().__init__(**kwargs)
|
|
||||||
|
|
||||||
del _TemporaryDirectory
|
|
||||||
|
|
||||||
|
|
||||||
UNPAPER_IMAGE_PIXEL_LIMIT = 256 * 1024 * 1024
|
UNPAPER_IMAGE_PIXEL_LIMIT = 256 * 1024 * 1024
|
||||||
|
|
||||||
DecFloat = Union[Decimal, float]
|
DecFloat = Decimal | float
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -70,7 +48,7 @@ class UnpaperImageTooLargeError(Exception):
|
|||||||
|
|
||||||
|
|
||||||
def version() -> Version:
|
def version() -> Version:
|
||||||
return Version(get_version('unpaper'))
|
return Version(get_version('unpaper', regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)'))
|
||||||
|
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
|
|||||||
+64
-44
@@ -7,32 +7,46 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import logging
|
import logging
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
|
from enum import Enum
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
from pikepdf import (
|
from pikepdf import (
|
||||||
Dictionary,
|
Dictionary,
|
||||||
|
Matrix,
|
||||||
Name,
|
Name,
|
||||||
Object,
|
|
||||||
Operator,
|
Operator,
|
||||||
|
Page,
|
||||||
Pdf,
|
Pdf,
|
||||||
PdfError,
|
PdfError,
|
||||||
PdfMatrix,
|
|
||||||
Stream,
|
Stream,
|
||||||
parse_content_stream,
|
parse_content_stream,
|
||||||
unparse_content_stream,
|
unparse_content_stream,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
from ocrmypdf._jobcontext import PdfContext
|
||||||
|
|
||||||
|
|
||||||
|
class RenderMode(Enum):
|
||||||
|
ON_TOP = 0
|
||||||
|
UNDERNEATH = 1
|
||||||
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
MAX_REPLACE_PAGES = 100
|
MAX_REPLACE_PAGES = 100
|
||||||
|
|
||||||
|
|
||||||
def _ensure_dictionary(obj, name):
|
def _ensure_dictionary(obj: Dictionary | Stream, name: Name):
|
||||||
if name not in obj:
|
if name not in obj:
|
||||||
obj[name] = Dictionary({})
|
obj[name] = Dictionary({})
|
||||||
return obj[name]
|
return obj[name]
|
||||||
|
|
||||||
|
|
||||||
def _update_resources(*, obj, font, font_key, procset):
|
def _update_resources(
|
||||||
|
*,
|
||||||
|
obj: Dictionary | Stream,
|
||||||
|
font: Dictionary | None,
|
||||||
|
font_key: Name | None,
|
||||||
|
):
|
||||||
"""Update this obj's fonts with a reference to the Glyphless font.
|
"""Update this obj's fonts with a reference to the Glyphless font.
|
||||||
|
|
||||||
obj can be a page or Form XObject.
|
obj can be a page or Form XObject.
|
||||||
@@ -42,13 +56,8 @@ def _update_resources(*, obj, font, font_key, procset):
|
|||||||
if font_key is not None and font_key not in fonts:
|
if font_key is not None and font_key not in fonts:
|
||||||
fonts[font_key] = font
|
fonts[font_key] = font
|
||||||
|
|
||||||
# Reassign /ProcSet to one that just lists everything - ProcSet is
|
|
||||||
# obsolete and doesn't matter but recommended for old viewer support
|
|
||||||
if procset:
|
|
||||||
resources['/ProcSet'] = procset
|
|
||||||
|
|
||||||
|
def strip_invisible_text(pdf: Pdf, page: Page):
|
||||||
def strip_invisible_text(pdf, page):
|
|
||||||
stream = []
|
stream = []
|
||||||
in_text_obj = False
|
in_text_obj = False
|
||||||
render_mode = 0
|
render_mode = 0
|
||||||
@@ -79,22 +88,20 @@ def strip_invisible_text(pdf, page):
|
|||||||
class OcrGrafter:
|
class OcrGrafter:
|
||||||
"""Manages grafting text-only PDFs onto regular PDFs."""
|
"""Manages grafting text-only PDFs onto regular PDFs."""
|
||||||
|
|
||||||
def __init__(self, context):
|
def __init__(self, context: PdfContext):
|
||||||
self.context = context
|
self.context = context
|
||||||
self.path_base = context.origin
|
self.path_base = context.origin
|
||||||
|
|
||||||
self.pdf_base = Pdf.open(self.path_base)
|
self.pdf_base = Pdf.open(self.path_base)
|
||||||
self.font, self.font_key = None, None
|
self.font: Dictionary | None = None
|
||||||
|
self.font_key: Name | None = None
|
||||||
|
|
||||||
self.pdfinfo = context.pdfinfo
|
self.pdfinfo = context.pdfinfo
|
||||||
self.output_file = context.get_path('graft_layers.pdf')
|
self.output_file = context.get_path('graft_layers.pdf')
|
||||||
|
|
||||||
self.procset = self.pdf_base.make_indirect(
|
|
||||||
Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]')
|
|
||||||
)
|
|
||||||
|
|
||||||
self.emplacements = 1
|
self.emplacements = 1
|
||||||
self.interim_count = 0
|
self.interim_count = 0
|
||||||
|
self.render_mode = RenderMode.UNDERNEATH
|
||||||
|
|
||||||
def graft_page(
|
def graft_page(
|
||||||
self,
|
self,
|
||||||
@@ -119,7 +126,9 @@ class OcrGrafter:
|
|||||||
foreign_image_page = pdf_image.pages[0]
|
foreign_image_page = pdf_image.pages[0]
|
||||||
self.pdf_base.pages.append(foreign_image_page)
|
self.pdf_base.pages.append(foreign_image_page)
|
||||||
local_image_page = self.pdf_base.pages[-1]
|
local_image_page = self.pdf_base.pages[-1]
|
||||||
self.pdf_base.pages[pageno].emplace(local_image_page)
|
self.pdf_base.pages[pageno].emplace(
|
||||||
|
local_image_page, retain=(Name.Parent,)
|
||||||
|
)
|
||||||
del self.pdf_base.pages[-1]
|
del self.pdf_base.pages[-1]
|
||||||
emplaced_page = True
|
emplaced_page = True
|
||||||
|
|
||||||
@@ -135,6 +144,8 @@ class OcrGrafter:
|
|||||||
)
|
)
|
||||||
|
|
||||||
if textpdf and self.font:
|
if textpdf and self.font:
|
||||||
|
if self.font_key is None:
|
||||||
|
raise ValueError("Font key is not set")
|
||||||
# Graft the text layer onto this page, whether new or old, possibly
|
# Graft the text layer onto this page, whether new or old, possibly
|
||||||
# rotating the text layer by the amount is misaligned.
|
# rotating the text layer by the amount is misaligned.
|
||||||
strip_old = self.context.options.redo_ocr
|
strip_old = self.context.options.redo_ocr
|
||||||
@@ -144,7 +155,6 @@ class OcrGrafter:
|
|||||||
font=self.font,
|
font=self.font,
|
||||||
font_key=self.font_key,
|
font_key=self.font_key,
|
||||||
text_rotation=text_misaligned,
|
text_rotation=text_misaligned,
|
||||||
procset=self.procset,
|
|
||||||
strip_old_text=strip_old,
|
strip_old_text=strip_old,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -159,7 +169,7 @@ class OcrGrafter:
|
|||||||
if self.emplacements % MAX_REPLACE_PAGES == 0:
|
if self.emplacements % MAX_REPLACE_PAGES == 0:
|
||||||
self.save_and_reload()
|
self.save_and_reload()
|
||||||
|
|
||||||
def save_and_reload(self):
|
def save_and_reload(self) -> None:
|
||||||
"""Save and reload the Pdf.
|
"""Save and reload the Pdf.
|
||||||
|
|
||||||
This will keep a lid on our memory usage for very large files. Attach
|
This will keep a lid on our memory usage for very large files. Attach
|
||||||
@@ -167,9 +177,7 @@ class OcrGrafter:
|
|||||||
back.
|
back.
|
||||||
"""
|
"""
|
||||||
page0 = self.pdf_base.pages[0]
|
page0 = self.pdf_base.pages[0]
|
||||||
_update_resources(
|
_update_resources(obj=page0.obj, font=self.font, font_key=self.font_key)
|
||||||
obj=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
|
||||||
)
|
|
||||||
|
|
||||||
# We cannot read and write the same file, that will corrupt it
|
# We cannot read and write the same file, that will corrupt it
|
||||||
# but we don't to keep more copies than we need to. Delete intermediates.
|
# but we don't to keep more copies than we need to. Delete intermediates.
|
||||||
@@ -188,7 +196,6 @@ class OcrGrafter:
|
|||||||
self.pdf_base.close()
|
self.pdf_base.close()
|
||||||
|
|
||||||
self.pdf_base = Pdf.open(next_file)
|
self.pdf_base = Pdf.open(next_file)
|
||||||
self.procset = self.pdf_base.pages[0].Resources.ProcSet
|
|
||||||
self.font, self.font_key = None, None # Ensure we reacquire this information
|
self.font, self.font_key = None, None # Ensure we reacquire this information
|
||||||
self.interim_count += 1
|
self.interim_count += 1
|
||||||
|
|
||||||
@@ -197,24 +204,32 @@ class OcrGrafter:
|
|||||||
self.pdf_base.close()
|
self.pdf_base.close()
|
||||||
return self.output_file
|
return self.output_file
|
||||||
|
|
||||||
def _find_font(self, text):
|
def _find_font(self, text: Path) -> tuple[Dictionary | None, Name | None]:
|
||||||
"""Copy a font from the filename text into pdf_base."""
|
"""Copy a font from the filename text into pdf_base."""
|
||||||
font, font_key = None, None
|
font, font_key = None, None
|
||||||
possible_font_names = ('/f-0-0', '/F1')
|
possible_font_names = ('/f-0-0', '/F1')
|
||||||
try:
|
try:
|
||||||
with Pdf.open(text) as pdf_text:
|
with Pdf.open(text) as pdf_text:
|
||||||
try:
|
try:
|
||||||
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
pdf_text_fonts = pdf_text.pages[0].Resources.get(
|
||||||
|
Name.Font, Dictionary()
|
||||||
|
)
|
||||||
except (AttributeError, IndexError, KeyError):
|
except (AttributeError, IndexError, KeyError):
|
||||||
return None, None
|
return None, None
|
||||||
|
if not isinstance(pdf_text_fonts, Dictionary):
|
||||||
|
log.warning("Page fonts are not stored in a dictionary")
|
||||||
|
return None, None
|
||||||
pdf_text_font = None
|
pdf_text_font = None
|
||||||
for f in possible_font_names:
|
for f in possible_font_names:
|
||||||
pdf_text_font = pdf_text_fonts.get(f, None)
|
pdf_text_font = pdf_text_fonts.get(f, None)
|
||||||
if pdf_text_font is not None:
|
if pdf_text_font is not None:
|
||||||
font_key = f
|
font_key = Name(f)
|
||||||
break
|
break
|
||||||
if pdf_text_font:
|
if pdf_text_font:
|
||||||
font = self.pdf_base.copy_foreign(pdf_text_font)
|
font = self.pdf_base.copy_foreign(pdf_text_font)
|
||||||
|
if not isinstance(font, Dictionary):
|
||||||
|
log.warning("Font is not a dictionary")
|
||||||
|
font, font_key = None, None
|
||||||
return font, font_key
|
return font, font_key
|
||||||
except (FileNotFoundError, PdfError):
|
except (FileNotFoundError, PdfError):
|
||||||
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
||||||
@@ -225,9 +240,8 @@ class OcrGrafter:
|
|||||||
*,
|
*,
|
||||||
page_num: int,
|
page_num: int,
|
||||||
textpdf: Path,
|
textpdf: Path,
|
||||||
font: Object,
|
font: Dictionary,
|
||||||
font_key: Object,
|
font_key: Name,
|
||||||
procset: Object,
|
|
||||||
text_rotation: int,
|
text_rotation: int,
|
||||||
strip_old_text: bool,
|
strip_old_text: bool,
|
||||||
):
|
):
|
||||||
@@ -254,13 +268,13 @@ class OcrGrafter:
|
|||||||
mediabox = base_page.mediabox
|
mediabox = base_page.mediabox
|
||||||
wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
||||||
|
|
||||||
translate = PdfMatrix().translated(-wt / 2, -ht / 2)
|
translate = Matrix().translated(-wt / 2, -ht / 2)
|
||||||
untranslate = PdfMatrix().translated(wp / 2, hp / 2)
|
untranslate = Matrix().translated(wp / 2, hp / 2)
|
||||||
corner = PdfMatrix().translated(mediabox[0], mediabox[1])
|
corner = Matrix().translated(mediabox[0], mediabox[1])
|
||||||
# -rotation because the input is a clockwise angle and this formula
|
# -rotation because the input is a clockwise angle and this formula
|
||||||
# uses CCW
|
# uses CCW
|
||||||
text_rotation = -text_rotation % 360
|
text_rotation = -text_rotation % 360
|
||||||
rotate = PdfMatrix().rotated(text_rotation)
|
rotate = Matrix().rotated(text_rotation)
|
||||||
|
|
||||||
# Because of rounding of DPI, we might get a text layer that is not
|
# Because of rounding of DPI, we might get a text layer that is not
|
||||||
# identically sized to the target page. Scale to adjust. Normally this
|
# identically sized to the target page. Scale to adjust. Normally this
|
||||||
@@ -271,14 +285,15 @@ class OcrGrafter:
|
|||||||
scale_y = hp / ht
|
scale_y = hp / ht
|
||||||
|
|
||||||
# log.debug('%r', scale_x, scale_y)
|
# log.debug('%r', scale_x, scale_y)
|
||||||
scale = PdfMatrix().scaled(scale_x, scale_y)
|
scale = Matrix().scaled(scale_x, scale_y)
|
||||||
|
|
||||||
# Translate the text so it is centered at (0, 0), rotate it there, adjust
|
# Translate the text so it is centered at (0, 0), rotate it there, adjust
|
||||||
# for a size different between initial and text PDF, then untranslate, and
|
# for a size different between initial and text PDF, then untranslate, and
|
||||||
# finally move the lower left corner to match the mediabox
|
# finally move the lower left corner to match the mediabox.
|
||||||
ctm = translate @ rotate @ scale @ untranslate @ corner
|
ctm = translate @ rotate @ scale @ untranslate @ corner
|
||||||
|
log.debug("Grafting with ctm %r", ctm)
|
||||||
|
|
||||||
base_resources = _ensure_dictionary(base_page, Name.Resources)
|
base_resources = _ensure_dictionary(base_page.obj, Name.Resources)
|
||||||
base_xobjs = _ensure_dictionary(base_resources, Name.XObject)
|
base_xobjs = _ensure_dictionary(base_resources, Name.XObject)
|
||||||
text_xobj_name = Name.random(prefix="OCR-")
|
text_xobj_name = Name.random(prefix="OCR-")
|
||||||
xobj = self.pdf_base.make_stream(pdf_text_contents)
|
xobj = self.pdf_base.make_stream(pdf_text_contents)
|
||||||
@@ -287,9 +302,7 @@ class OcrGrafter:
|
|||||||
xobj.Subtype = Name.Form
|
xobj.Subtype = Name.Form
|
||||||
xobj.FormType = 1
|
xobj.FormType = 1
|
||||||
xobj.BBox = mediabox
|
xobj.BBox = mediabox
|
||||||
_update_resources(
|
_update_resources(obj=xobj, font=font, font_key=font_key)
|
||||||
obj=xobj, font=font, font_key=font_key, procset=[Name.PDF]
|
|
||||||
)
|
|
||||||
|
|
||||||
pdf_draw_xobj = (
|
pdf_draw_xobj = (
|
||||||
(b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
(b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
||||||
@@ -298,9 +311,16 @@ class OcrGrafter:
|
|||||||
|
|
||||||
if strip_old_text:
|
if strip_old_text:
|
||||||
strip_invisible_text(self.pdf_base, base_page)
|
strip_invisible_text(self.pdf_base, base_page)
|
||||||
|
base_page.contents_coalesce()
|
||||||
base_page.contents_add(new_text_layer, prepend=True)
|
if self.render_mode == RenderMode.ON_TOP:
|
||||||
|
# Add q/Q to ensure content we append is drawn correctly
|
||||||
_update_resources(
|
# Strictly speaking this needs to trace the whole q/Q stack in case
|
||||||
obj=base_page, font=font, font_key=font_key, procset=procset
|
# stack is not balanced.
|
||||||
|
original = base_page.Contents.read_bytes()
|
||||||
|
base_page.Contents.write(b'q\n' + original + b'\nQ\n')
|
||||||
|
base_page.contents_add(
|
||||||
|
new_text_layer, prepend=self.render_mode == RenderMode.UNDERNEATH
|
||||||
)
|
)
|
||||||
|
base_page.contents_coalesce()
|
||||||
|
|
||||||
|
_update_resources(obj=base_page.obj, font=font, font_key=font_key)
|
||||||
|
|||||||
@@ -93,8 +93,8 @@ class PageContext:
|
|||||||
state = self.__dict__.copy()
|
state = self.__dict__.copy()
|
||||||
|
|
||||||
state['options'] = copy(self.options)
|
state['options'] = copy(self.options)
|
||||||
if not isinstance(state['options'].input_file, (str, bytes, os.PathLike)):
|
if not isinstance(state['options'].input_file, str | bytes | os.PathLike):
|
||||||
state['options'].input_file = 'stream'
|
state['options'].input_file = 'stream'
|
||||||
if not isinstance(state['options'].output_file, (str, bytes, os.PathLike)):
|
if not isinstance(state['options'].output_file, str | bytes | os.PathLike):
|
||||||
state['options'].output_file = 'stream'
|
state['options'].output_file = 'stream'
|
||||||
return state
|
return state
|
||||||
|
|||||||
@@ -15,8 +15,9 @@ from pikepdf import Dictionary, Name, Pdf
|
|||||||
from pikepdf import __version__ as PIKEPDF_VERSION
|
from pikepdf import __version__ as PIKEPDF_VERSION
|
||||||
from pikepdf.models.metadata import PdfMetadata, encode_pdf_date
|
from pikepdf.models.metadata import PdfMetadata, encode_pdf_date
|
||||||
|
|
||||||
|
from ocrmypdf._annots import remove_broken_goto_annotations
|
||||||
|
from ocrmypdf._defaults import PROGRAM_NAME
|
||||||
from ocrmypdf._jobcontext import PdfContext
|
from ocrmypdf._jobcontext import PdfContext
|
||||||
from ocrmypdf._version import PROGRAM_NAME
|
|
||||||
from ocrmypdf._version import __version__ as OCRMYPF_VERSION
|
from ocrmypdf._version import __version__ as OCRMYPF_VERSION
|
||||||
from ocrmypdf.languages import iso_639_2_from_3
|
from ocrmypdf.languages import iso_639_2_from_3
|
||||||
|
|
||||||
@@ -153,22 +154,52 @@ def _set_language(pdf: Pdf, languages: list[str]):
|
|||||||
pdf.Root.Lang = iso639_2
|
pdf.Root.Lang = iso639_2
|
||||||
|
|
||||||
|
|
||||||
|
class MetadataProgress:
|
||||||
|
def __init__(self, progressbar_class, enable: bool = True):
|
||||||
|
self.progressbar_class = progressbar_class
|
||||||
|
self.progressbar = self.progressbar_class(
|
||||||
|
total=100, desc="Linearizing", unit='%', disable=not enable
|
||||||
|
)
|
||||||
|
|
||||||
|
def __enter__(self):
|
||||||
|
self.progressbar.__enter__()
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
|
return self.progressbar.__exit__(exc_type, exc_value, traceback)
|
||||||
|
|
||||||
|
def __call__(self, percent: int):
|
||||||
|
if not self.progressbar_class:
|
||||||
|
return
|
||||||
|
self.progressbar.update(completed=percent)
|
||||||
|
|
||||||
|
|
||||||
def metadata_fixup(
|
def metadata_fixup(
|
||||||
working_file: Path, context: PdfContext, pdf_save_settings: dict[str, Any]
|
working_file: Path, context: PdfContext, pdf_save_settings: dict[str, Any]
|
||||||
) -> Path:
|
) -> Path:
|
||||||
"""Fix certain metadata fields after Ghostscript PDF/A conversion.
|
"""Fix certain metadata fields whether PDF or PDF/A.
|
||||||
|
|
||||||
|
Override some of Ghostscript's metadata choices.
|
||||||
|
|
||||||
Also report on metadata in the input file that was not retained during
|
Also report on metadata in the input file that was not retained during
|
||||||
PDF/A conversion.
|
conversion.
|
||||||
"""
|
"""
|
||||||
output_file = context.get_path('metafix.pdf')
|
output_file = context.get_path('metafix.pdf')
|
||||||
options = context.options
|
options = context.options
|
||||||
|
|
||||||
with Pdf.open(context.origin) as original, Pdf.open(working_file) as pdf:
|
pbar_class = context.plugin_manager.hook.get_progressbar_class()
|
||||||
|
with (
|
||||||
|
Pdf.open(context.origin) as original,
|
||||||
|
Pdf.open(working_file) as pdf,
|
||||||
|
MetadataProgress(pbar_class, options.progress_bar) as pbar,
|
||||||
|
):
|
||||||
docinfo = get_docinfo(original, context)
|
docinfo = get_docinfo(original, context)
|
||||||
with original.open_metadata(
|
with (
|
||||||
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
original.open_metadata(
|
||||||
) as meta_original, pdf.open_metadata() as meta_pdf:
|
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
||||||
|
) as meta_original,
|
||||||
|
pdf.open_metadata() as meta_pdf,
|
||||||
|
):
|
||||||
meta_pdf.load_from_docinfo(
|
meta_pdf.load_from_docinfo(
|
||||||
docinfo, delete_missing=False, raise_failure=False
|
docinfo, delete_missing=False, raise_failure=False
|
||||||
)
|
)
|
||||||
@@ -179,6 +210,6 @@ def metadata_fixup(
|
|||||||
report_on_metadata(options, meta_missing)
|
report_on_metadata(options, meta_missing)
|
||||||
|
|
||||||
_set_language(pdf, options.languages)
|
_set_language(pdf, options.languages)
|
||||||
pdf.save(output_file, **pdf_save_settings)
|
pdf.save(output_file, progress=pbar, **pdf_save_settings)
|
||||||
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|||||||
+108
-25
@@ -12,8 +12,9 @@ import re
|
|||||||
import sys
|
import sys
|
||||||
from collections.abc import Iterable, Iterator, Sequence
|
from collections.abc import Iterable, Iterator, Sequence
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
|
from io import BytesIO
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj, copystat
|
from shutil import copyfileobj
|
||||||
from typing import Any, BinaryIO, TypeVar, cast
|
from typing import Any, BinaryIO, TypeVar, cast
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
@@ -34,17 +35,29 @@ from ocrmypdf.exceptions import (
|
|||||||
UnsupportedImageFormatError,
|
UnsupportedImageFormatError,
|
||||||
)
|
)
|
||||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
|
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
|
||||||
from ocrmypdf.hocrtransform import HocrTransform
|
from ocrmypdf.hocrtransform import DebugRenderOptions, HocrTransform
|
||||||
|
from ocrmypdf.hocrtransform._font import Courier
|
||||||
from ocrmypdf.pdfa import generate_pdfa_ps
|
from ocrmypdf.pdfa import generate_pdfa_ps
|
||||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, PageInfo, PdfInfo
|
from ocrmypdf.pdfinfo import Colorspace, Encoding, PageInfo, PdfInfo
|
||||||
from ocrmypdf.pluginspec import OrientationConfidence
|
from ocrmypdf.pluginspec import OrientationConfidence
|
||||||
|
|
||||||
|
try:
|
||||||
|
from pi_heif import register_heif_opener
|
||||||
|
except ImportError:
|
||||||
|
|
||||||
|
def register_heif_opener():
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
T = TypeVar("T")
|
T = TypeVar("T")
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
VECTOR_PAGE_DPI = 400
|
VECTOR_PAGE_DPI = 400
|
||||||
|
|
||||||
|
|
||||||
|
register_heif_opener()
|
||||||
|
|
||||||
|
|
||||||
def triage_image_file(input_file: Path, output_file: Path, options) -> None:
|
def triage_image_file(input_file: Path, output_file: Path, options) -> None:
|
||||||
"""Triage the input image file.
|
"""Triage the input image file.
|
||||||
|
|
||||||
@@ -129,7 +142,7 @@ def _pdf_guess_version(input_file: Path, search_window=1024) -> str:
|
|||||||
"""
|
"""
|
||||||
with open(input_file, 'rb') as f:
|
with open(input_file, 'rb') as f:
|
||||||
signature = f.read(search_window)
|
signature = f.read(search_window)
|
||||||
m = re.search(br'%PDF-(\d\.\d)', signature)
|
m = re.search(rb'%PDF-(\d\.\d)', signature)
|
||||||
if m:
|
if m:
|
||||||
return m.group(1).decode('ascii')
|
return m.group(1).decode('ascii')
|
||||||
return ''
|
return ''
|
||||||
@@ -146,8 +159,13 @@ def triage(
|
|||||||
"Argument --image-dpi is being ignored because the "
|
"Argument --image-dpi is being ignored because the "
|
||||||
"input file is a PDF, not an image."
|
"input file is a PDF, not an image."
|
||||||
)
|
)
|
||||||
# Origin file is a pdf create a symlink with pdf extension
|
try:
|
||||||
safe_symlink(input_file, output_file)
|
with pikepdf.open(input_file) as pdf:
|
||||||
|
pdf.save(output_file)
|
||||||
|
except pikepdf.PdfError as e:
|
||||||
|
raise InputFileError() from e
|
||||||
|
except pikepdf.PasswordError as e:
|
||||||
|
raise EncryptedPdfError() from e
|
||||||
return output_file
|
return output_file
|
||||||
except OSError as e:
|
except OSError as e:
|
||||||
log.debug(f"Temporary file was at: {input_file}")
|
log.debug(f"Temporary file was at: {input_file}")
|
||||||
@@ -462,7 +480,7 @@ def calculate_raster_dpi(page_context: PageContext):
|
|||||||
page_dpi = get_page_square_dpi(page_context, image_dpi)
|
page_dpi = get_page_square_dpi(page_context, image_dpi)
|
||||||
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
|
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
|
||||||
log.warning(
|
log.warning(
|
||||||
"Weight average image DPI is %0.1f, max DPI is %0.1f. "
|
"Weighted average image DPI is %0.1f, max DPI is %0.1f. "
|
||||||
"The discrepancy may indicate a high detail region on this page, "
|
"The discrepancy may indicate a high detail region on this page, "
|
||||||
"but could also indicate a problem with the input PDF file. "
|
"but could also indicate a problem with the input PDF file. "
|
||||||
"Page image will be rendered at %0.1f DPI.",
|
"Page image will be rendered at %0.1f DPI.",
|
||||||
@@ -712,19 +730,29 @@ def create_pdf_page_from_image(
|
|||||||
pageinfo = page_context.pageinfo
|
pageinfo = page_context.pageinfo
|
||||||
pagesize = 72.0 * float(pageinfo.width_inches), 72.0 * float(pageinfo.height_inches)
|
pagesize = 72.0 * float(pageinfo.width_inches), 72.0 * float(pageinfo.height_inches)
|
||||||
effective_rotation = (pageinfo.rotation - orientation_correction) % 360
|
effective_rotation = (pageinfo.rotation - orientation_correction) % 360
|
||||||
if effective_rotation % 180 == 90:
|
swap_axis = effective_rotation % 180 == 90
|
||||||
|
if swap_axis:
|
||||||
pagesize = pagesize[1], pagesize[0]
|
pagesize = pagesize[1], pagesize[0]
|
||||||
|
|
||||||
# This create a single page PDF
|
# Create a new single page PDF to hold
|
||||||
with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf:
|
bio = BytesIO()
|
||||||
|
with open(image, 'rb') as imfile:
|
||||||
log.debug('convert')
|
log.debug('convert')
|
||||||
|
|
||||||
layout_fun = img2pdf.get_layout_fun(pagesize)
|
layout_fun = img2pdf.get_layout_fun(pagesize)
|
||||||
img2pdf.convert(
|
img2pdf.convert(
|
||||||
imfile, layout_fun=layout_fun, outputstream=pdf, **IMG2PDF_KWARGS
|
imfile,
|
||||||
|
layout_fun=layout_fun,
|
||||||
|
outputstream=bio,
|
||||||
|
engine=img2pdf.Engine.pikepdf,
|
||||||
|
rotation=img2pdf.Rotation.ifvalid,
|
||||||
)
|
)
|
||||||
log.debug('convert done')
|
log.debug('convert done')
|
||||||
|
|
||||||
|
# img2pdf does not generate boxes correctly, so we fix them
|
||||||
|
bio.seek(0)
|
||||||
|
fix_pagepdf_boxes(bio, output_file, page_context, swap_axis=swap_axis)
|
||||||
|
|
||||||
output_file = page_context.plugin_manager.hook.filter_pdf_page(
|
output_file = page_context.plugin_manager.hook.filter_pdf_page(
|
||||||
page=page_context, image_filename=image, output_pdf=output_file
|
page=page_context, image_filename=image, output_pdf=output_file
|
||||||
)
|
)
|
||||||
@@ -741,15 +769,27 @@ def render_hocr_page(hocr: Path, page_context: PageContext) -> Path:
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
dpi = get_page_square_dpi(page_context, calculate_image_dpi(page_context))
|
dpi = get_page_square_dpi(page_context, calculate_image_dpi(page_context))
|
||||||
debug_mode = options.pdf_renderer == 'hocrdebug'
|
debug_kwargs = {}
|
||||||
|
if options.pdf_renderer == 'hocrdebug':
|
||||||
hocrtransform = HocrTransform(hocr_filename=hocr, dpi=dpi.to_scalar()) # square
|
debug_kwargs = dict(
|
||||||
hocrtransform.to_pdf(
|
debug_render_options=DebugRenderOptions(
|
||||||
|
render_baseline=True,
|
||||||
|
render_triangle=True,
|
||||||
|
render_line_bbox=False,
|
||||||
|
render_word_bbox=True,
|
||||||
|
render_paragraph_bbox=False,
|
||||||
|
render_space_bbox=False,
|
||||||
|
),
|
||||||
|
font=Courier(),
|
||||||
|
)
|
||||||
|
HocrTransform(
|
||||||
|
hocr_filename=hocr,
|
||||||
|
dpi=dpi.to_scalar(),
|
||||||
|
**debug_kwargs, # square
|
||||||
|
).to_pdf(
|
||||||
out_filename=output_file,
|
out_filename=output_file,
|
||||||
image_filename=None,
|
image_filename=None,
|
||||||
show_bounding_boxes=False if not debug_mode else True,
|
invisible_text=True if not debug_kwargs else False,
|
||||||
invisible_text=True if not debug_mode else False,
|
|
||||||
interword_spaces=True,
|
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
@@ -769,7 +809,57 @@ def ocr_engine_textonly_pdf(
|
|||||||
output_text=output_text,
|
output_text=output_text,
|
||||||
options=options,
|
options=options,
|
||||||
)
|
)
|
||||||
return (output_pdf, output_text)
|
return output_pdf, output_text
|
||||||
|
|
||||||
|
|
||||||
|
def _offset_rect(rect: tuple[float, float, float, float], offset: tuple[float, float]):
|
||||||
|
"""Offset a rectangle by a given amount."""
|
||||||
|
return (
|
||||||
|
rect[0] + offset[0],
|
||||||
|
rect[1] + offset[1],
|
||||||
|
rect[2] + offset[0],
|
||||||
|
rect[3] + offset[1],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def fix_pagepdf_boxes(
|
||||||
|
infile: Path | BinaryIO,
|
||||||
|
out_file: Path,
|
||||||
|
page_context: PageContext,
|
||||||
|
swap_axis: bool = False,
|
||||||
|
) -> Path:
|
||||||
|
"""Fix the bounding boxes in a single page PDF.
|
||||||
|
|
||||||
|
The single page PDF is created with a normal MediaBox with its lower left corner
|
||||||
|
at (0, 0). infile is the single page PDF. page_context.mediabox has the original
|
||||||
|
file's mediabox, which may have a different origin. We needto adjust the other
|
||||||
|
boxes in the single page PDF to match the effect they had on the original page.
|
||||||
|
|
||||||
|
When correcting page rotation, we create a single page PDF that is correctly
|
||||||
|
rotated instead of an incorrectly rotated and then setting page.Rotate on it.
|
||||||
|
If rotation is either 90 or 270 degrees, then this function can be called
|
||||||
|
with swap_axis to swap the X and Y coordinates of all the boxes.
|
||||||
|
|
||||||
|
We are not concerned with solving degenerate cases where the boxes overlap or
|
||||||
|
or express invalid rectangles. We merely pass the boxes, producing a
|
||||||
|
transformation equivalent to the change made by constructing a new page image.
|
||||||
|
"""
|
||||||
|
with pikepdf.open(infile) as pdf:
|
||||||
|
for page in pdf.pages:
|
||||||
|
# page.BleedBox = page_context.pageinfo.bleedbox
|
||||||
|
# page.ArtBox = page_context.pageinfo.artbox
|
||||||
|
mediabox = page_context.pageinfo.mediabox
|
||||||
|
offset = mediabox[0], mediabox[1]
|
||||||
|
cropbox = _offset_rect(page_context.pageinfo.cropbox, offset)
|
||||||
|
trimbox = _offset_rect(page_context.pageinfo.trimbox, offset)
|
||||||
|
|
||||||
|
if swap_axis:
|
||||||
|
cropbox = cropbox[1], cropbox[0], cropbox[3], cropbox[2]
|
||||||
|
trimbox = trimbox[1], trimbox[0], trimbox[3], trimbox[2]
|
||||||
|
page.CropBox = cropbox
|
||||||
|
page.TrimBox = trimbox
|
||||||
|
pdf.save(out_file)
|
||||||
|
return out_file
|
||||||
|
|
||||||
|
|
||||||
def generate_postscript_stub(context: PdfContext) -> Path:
|
def generate_postscript_stub(context: PdfContext) -> Path:
|
||||||
@@ -992,10 +1082,3 @@ def copy_final(
|
|||||||
# get the appropriate umask, ownership, etc.
|
# get the appropriate umask, ownership, etc.
|
||||||
with open(output_file, 'w+b') as output_stream:
|
with open(output_file, 'w+b') as output_stream:
|
||||||
copyfileobj(input_stream, output_stream)
|
copyfileobj(input_stream, output_stream)
|
||||||
# Attempt to copy file attributes from input to output
|
|
||||||
if original_file:
|
|
||||||
with suppress(OSError):
|
|
||||||
# Copy original file's permissions, ownership, etc. if possible
|
|
||||||
copystat(original_file, output_file)
|
|
||||||
# Set output file's modification time to now
|
|
||||||
Path(output_file).touch(exist_ok=True)
|
|
||||||
|
|||||||
@@ -11,16 +11,18 @@ import os
|
|||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
from collections.abc import Sequence
|
from collections.abc import Callable, Sequence
|
||||||
from concurrent.futures.process import BrokenProcessPool
|
from concurrent.futures.process import BrokenProcessPool
|
||||||
from concurrent.futures.thread import BrokenThreadPool
|
from concurrent.futures.thread import BrokenThreadPool
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Callable, NamedTuple, cast
|
from typing import NamedTuple, cast
|
||||||
|
|
||||||
import PIL
|
import PIL
|
||||||
|
from pikepdf import Pdf
|
||||||
|
|
||||||
|
from ocrmypdf._annots import remove_broken_goto_annotations
|
||||||
from ocrmypdf._concurrent import Executor, setup_executor
|
from ocrmypdf._concurrent import Executor, setup_executor
|
||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._logging import PageNumberFilter
|
from ocrmypdf._logging import PageNumberFilter
|
||||||
@@ -33,6 +35,7 @@ from ocrmypdf._pipeline import (
|
|||||||
generate_postscript_stub,
|
generate_postscript_stub,
|
||||||
get_orientation_correction,
|
get_orientation_correction,
|
||||||
get_pdf_save_settings,
|
get_pdf_save_settings,
|
||||||
|
get_pdfinfo,
|
||||||
optimize_pdf,
|
optimize_pdf,
|
||||||
preprocess_clean,
|
preprocess_clean,
|
||||||
preprocess_deskew,
|
preprocess_deskew,
|
||||||
@@ -51,9 +54,12 @@ from ocrmypdf.helpers import (
|
|||||||
available_cpu_count,
|
available_cpu_count,
|
||||||
check_pdf,
|
check_pdf,
|
||||||
pikepdf_enable_mmap,
|
pikepdf_enable_mmap,
|
||||||
|
running_in_docker,
|
||||||
|
running_in_snap,
|
||||||
samefile,
|
samefile,
|
||||||
)
|
)
|
||||||
from ocrmypdf.pdfa import file_claims_pdfa
|
from ocrmypdf.pdfa import file_claims_pdfa
|
||||||
|
from ocrmypdf.pdfinfo import PdfInfo
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
tls = threading.local()
|
tls = threading.local()
|
||||||
@@ -100,6 +106,23 @@ class PageResult(NamedTuple):
|
|||||||
"""Orientation correction in degrees."""
|
"""Orientation correction in degrees."""
|
||||||
|
|
||||||
|
|
||||||
|
class HOCRResultEncoder(json.JSONEncoder):
|
||||||
|
def default(self, obj):
|
||||||
|
if isinstance(obj, Path):
|
||||||
|
return {'Path': str(obj)}
|
||||||
|
return super().default(obj)
|
||||||
|
|
||||||
|
|
||||||
|
class HOCRResultDecoder(json.JSONDecoder):
|
||||||
|
def __init__(self, *args, **kwargs):
|
||||||
|
super().__init__(object_hook=self.dict_to_object, *args, **kwargs)
|
||||||
|
|
||||||
|
def dict_to_object(self, d):
|
||||||
|
if 'Path' in d:
|
||||||
|
return Path(d['Path'])
|
||||||
|
return d
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class HOCRResult:
|
class HOCRResult:
|
||||||
"""Result when hOCR is finished processing."""
|
"""Result when hOCR is finished processing."""
|
||||||
@@ -119,45 +142,23 @@ class HOCRResult:
|
|||||||
orientation_correction: int = 0
|
orientation_correction: int = 0
|
||||||
"""Orientation correction in degrees."""
|
"""Orientation correction in degrees."""
|
||||||
|
|
||||||
def __getstate__(self):
|
|
||||||
"""Return state values to be pickled."""
|
|
||||||
return {
|
|
||||||
k: (
|
|
||||||
('Path://' + str(v))
|
|
||||||
if k in ('pdf_page_from_image', 'hocr', 'textpdf') and v is not None
|
|
||||||
else v
|
|
||||||
)
|
|
||||||
for k, v in self.__dict__.items()
|
|
||||||
}
|
|
||||||
|
|
||||||
def __setstate__(self, state):
|
|
||||||
"""Restore state from the unpickled state values."""
|
|
||||||
self.__dict__.update(
|
|
||||||
{
|
|
||||||
k: (
|
|
||||||
Path(v.removeprefix('Path://'))
|
|
||||||
if k in ('pdf_page_from_image', 'hocr', 'textpdf') and v is not None
|
|
||||||
else v
|
|
||||||
)
|
|
||||||
for k, v in state.items()
|
|
||||||
}
|
|
||||||
)
|
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_json(cls, json_str: str) -> HOCRResult:
|
def from_json(cls, json_str: str) -> HOCRResult:
|
||||||
"""Create an instance from a dict."""
|
"""Create an instance from a dict."""
|
||||||
return cls(**json.loads(json_str))
|
return cls(**json.loads(json_str, cls=HOCRResultDecoder))
|
||||||
|
|
||||||
def to_json(self) -> str:
|
def to_json(self) -> str:
|
||||||
"""Serialize to a JSON string."""
|
"""Serialize to a JSON string."""
|
||||||
return json.dumps(self.__getstate__())
|
return json.dumps(self.__dict__, cls=HOCRResultEncoder)
|
||||||
|
|
||||||
|
|
||||||
def configure_debug_logging(
|
def configure_debug_logging(
|
||||||
log_filename: Path, prefix: str = ''
|
log_filename: Path, prefix: str = ''
|
||||||
) -> logging.FileHandler:
|
) -> tuple[logging.FileHandler, Callable[[], None]]:
|
||||||
"""Create a debug log file at a specified location.
|
"""Create a debug log file at a specified location.
|
||||||
|
|
||||||
|
Returns the log handler, and a function to remove the handler.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
log_filename: Where to the put the log file.
|
log_filename: Where to the put the log file.
|
||||||
prefix: The logging domain prefix that should be sent to the log.
|
prefix: The logging domain prefix that should be sent to the log.
|
||||||
@@ -170,10 +171,18 @@ def configure_debug_logging(
|
|||||||
log_file_handler.setFormatter(formatter)
|
log_file_handler.setFormatter(formatter)
|
||||||
log_file_handler.addFilter(PageNumberFilter())
|
log_file_handler.addFilter(PageNumberFilter())
|
||||||
logging.getLogger(prefix).addHandler(log_file_handler)
|
logging.getLogger(prefix).addHandler(log_file_handler)
|
||||||
return log_file_handler
|
|
||||||
|
def remover():
|
||||||
|
try:
|
||||||
|
logging.getLogger(prefix).removeHandler(log_file_handler)
|
||||||
|
log_file_handler.close()
|
||||||
|
except OSError as e:
|
||||||
|
print(e, file=sys.stderr)
|
||||||
|
|
||||||
|
return log_file_handler, remover
|
||||||
|
|
||||||
|
|
||||||
def worker_init(max_pixels: int) -> None:
|
def worker_init(max_pixels: int | None) -> None:
|
||||||
"""Initialize a worker thread or process."""
|
"""Initialize a worker thread or process."""
|
||||||
# In Windows, child process will not inherit our change to this value in
|
# In Windows, child process will not inherit our change to this value in
|
||||||
# the parent process, so ensure workers get it set. Not needed when running
|
# the parent process, so ensure workers get it set. Not needed when running
|
||||||
@@ -188,25 +197,37 @@ def manage_debug_log_handler(
|
|||||||
options: argparse.Namespace,
|
options: argparse.Namespace,
|
||||||
work_folder: Path,
|
work_folder: Path,
|
||||||
):
|
):
|
||||||
debug_log_handler = None
|
remover = None
|
||||||
if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get(
|
if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get(
|
||||||
'PYTEST_CURRENT_TEST', ''
|
'PYTEST_CURRENT_TEST', ''
|
||||||
):
|
):
|
||||||
# Debug log for command line interface only with verbose output
|
# Debug log for command line interface only with verbose output
|
||||||
# See https://github.com/pytest-dev/pytest/issues/5502 for why we skip this
|
# See https://github.com/pytest-dev/pytest/issues/5502 for why we skip this
|
||||||
# when pytest is running
|
# when pytest is running
|
||||||
debug_log_handler = configure_debug_logging(
|
_debug_log_handler, remover = configure_debug_logging(
|
||||||
work_folder / "debug.log"
|
work_folder / "debug.log", prefix=""
|
||||||
) # pragma: no cover
|
) # pragma: no cover
|
||||||
try:
|
try:
|
||||||
yield
|
yield
|
||||||
finally:
|
finally:
|
||||||
if debug_log_handler:
|
if remover:
|
||||||
try:
|
remover()
|
||||||
debug_log_handler.close()
|
|
||||||
log.removeHandler(debug_log_handler)
|
|
||||||
except OSError as e:
|
def _print_temp_folder_location(work_folder: Path):
|
||||||
print(e, file=sys.stderr)
|
"""Print the location of the temporary work folder."""
|
||||||
|
msgs = [f"Temporary working files retained at:\n{work_folder}"]
|
||||||
|
if running_in_docker(): # pragma: no cover
|
||||||
|
msgs.append(
|
||||||
|
"OCRmyPDF is running in a Docker container, "
|
||||||
|
"so the files will be inside the container."
|
||||||
|
)
|
||||||
|
elif running_in_snap(): # pragma: no cover
|
||||||
|
msgs.append(
|
||||||
|
"OCRmyPDF is running in a Snap container, "
|
||||||
|
"so the files will be inside the container."
|
||||||
|
)
|
||||||
|
print('\n'.join(msgs), file=sys.stderr)
|
||||||
|
|
||||||
|
|
||||||
@contextmanager
|
@contextmanager
|
||||||
@@ -216,10 +237,7 @@ def manage_work_folder(*, work_folder: Path, retain: bool, print_location: bool)
|
|||||||
finally:
|
finally:
|
||||||
if retain:
|
if retain:
|
||||||
if print_location:
|
if print_location:
|
||||||
print(
|
_print_temp_folder_location(work_folder)
|
||||||
f"Temporary working files retained at:\n{work_folder}",
|
|
||||||
file=sys.stderr,
|
|
||||||
)
|
|
||||||
else:
|
else:
|
||||||
shutil.rmtree(work_folder, ignore_errors=True)
|
shutil.rmtree(work_folder, ignore_errors=True)
|
||||||
|
|
||||||
@@ -229,7 +247,17 @@ def cli_exception_handler(
|
|||||||
options: argparse.Namespace,
|
options: argparse.Namespace,
|
||||||
plugin_manager: OcrmypdfPluginManager,
|
plugin_manager: OcrmypdfPluginManager,
|
||||||
) -> ExitCode:
|
) -> ExitCode:
|
||||||
|
"""Convert exceptions into command line error messages and exit codes.
|
||||||
|
|
||||||
|
When known exceptions are raised, the exception message is printed to stderr
|
||||||
|
and the program exits with a non-zero exit code. When unknown exceptions are
|
||||||
|
raised, the exception traceback is printed to stderr and the program exits
|
||||||
|
with a non-zero exit code.
|
||||||
|
"""
|
||||||
try:
|
try:
|
||||||
|
# We cannot use a generator and yield here, as would be the usual pattern
|
||||||
|
# for exception handling context managers, because we need to return an exit
|
||||||
|
# code.
|
||||||
return fn(options, plugin_manager)
|
return fn(options, plugin_manager)
|
||||||
except KeyboardInterrupt:
|
except KeyboardInterrupt:
|
||||||
if options.verbose >= 1:
|
if options.verbose >= 1:
|
||||||
@@ -284,6 +312,20 @@ def setup_pipeline(
|
|||||||
return executor
|
return executor
|
||||||
|
|
||||||
|
|
||||||
|
def do_get_pdfinfo(
|
||||||
|
pdf_path: Path, executor: Executor, options: argparse.Namespace
|
||||||
|
) -> PdfInfo:
|
||||||
|
return get_pdfinfo(
|
||||||
|
pdf_path,
|
||||||
|
executor=executor,
|
||||||
|
detailed_analysis=options.redo_ocr,
|
||||||
|
progbar=options.progress_bar,
|
||||||
|
max_workers=options.jobs,
|
||||||
|
use_threads=options.use_threads,
|
||||||
|
check_pages=options.pages,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def preprocess(
|
def preprocess(
|
||||||
page_context: PageContext,
|
page_context: PageContext,
|
||||||
image: Path,
|
image: Path,
|
||||||
@@ -398,7 +440,14 @@ def postprocess(
|
|||||||
pdf_file: Path, context: PdfContext, executor: Executor
|
pdf_file: Path, context: PdfContext, executor: Executor
|
||||||
) -> tuple[Path, Sequence[str]]:
|
) -> tuple[Path, Sequence[str]]:
|
||||||
"""Postprocess the PDF file."""
|
"""Postprocess the PDF file."""
|
||||||
pdf_out = pdf_file
|
# pdf_out = pdf_file
|
||||||
|
with Pdf.open(pdf_file) as pdf:
|
||||||
|
fix_annots = context.get_path('fix_annots.pdf')
|
||||||
|
if remove_broken_goto_annotations(pdf):
|
||||||
|
pdf.save(fix_annots)
|
||||||
|
pdf_out = fix_annots
|
||||||
|
else:
|
||||||
|
pdf_out = pdf_file
|
||||||
if context.options.output_type.startswith('pdfa'):
|
if context.options.output_type.startswith('pdfa'):
|
||||||
ps_stub_out = generate_postscript_stub(context)
|
ps_stub_out = generate_postscript_stub(context)
|
||||||
pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context)
|
pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context)
|
||||||
@@ -425,7 +474,8 @@ def report_output_pdf(options, start_input_file, optimize_messages) -> ExitCode:
|
|||||||
log.info("Output file is a %s (as expected)", pdfa_info['conformance'])
|
log.info("Output file is a %s (as expected)", pdfa_info['conformance'])
|
||||||
else:
|
else:
|
||||||
log.warning(
|
log.warning(
|
||||||
"Output file is okay but is not PDF/A (seems to be %s)",
|
"Output file is a valid PDF, but conversion to PDF/A did not "
|
||||||
|
"succeed (issue: %s)",
|
||||||
pdfa_info['conformance'],
|
pdfa_info['conformance'],
|
||||||
)
|
)
|
||||||
return ExitCode.pdfa_conversion_failed
|
return ExitCode.pdfa_conversion_failed
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
|
|
||||||
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
@@ -20,11 +19,11 @@ from ocrmypdf._graft import OcrGrafter
|
|||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._pipeline import (
|
from ocrmypdf._pipeline import (
|
||||||
copy_final,
|
copy_final,
|
||||||
get_pdfinfo,
|
|
||||||
render_hocr_page,
|
render_hocr_page,
|
||||||
)
|
)
|
||||||
from ocrmypdf._pipelines._common import (
|
from ocrmypdf._pipelines._common import (
|
||||||
HOCRResult,
|
HOCRResult,
|
||||||
|
do_get_pdfinfo,
|
||||||
manage_work_folder,
|
manage_work_folder,
|
||||||
postprocess,
|
postprocess,
|
||||||
report_output_pdf,
|
report_output_pdf,
|
||||||
@@ -118,15 +117,7 @@ def run_hocr_to_ocr_pdf_pipeline(
|
|||||||
origin_pdf = work_folder / 'origin.pdf'
|
origin_pdf = work_folder / 'origin.pdf'
|
||||||
|
|
||||||
# Gather pdfinfo and create context
|
# Gather pdfinfo and create context
|
||||||
pdfinfo = get_pdfinfo(
|
pdfinfo = do_get_pdfinfo(origin_pdf, executor, options)
|
||||||
origin_pdf,
|
|
||||||
executor=executor,
|
|
||||||
detailed_analysis=options.redo_ocr,
|
|
||||||
progbar=options.progress_bar,
|
|
||||||
max_workers=options.jobs,
|
|
||||||
use_threads=options.use_threads,
|
|
||||||
check_pages=options.pages,
|
|
||||||
)
|
|
||||||
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
||||||
plugin_manager.hook.check_options(options=options)
|
plugin_manager.hook.check_options(options=options)
|
||||||
optimize_messages = exec_hocr_to_ocr_pdf(context, executor)
|
optimize_messages = exec_hocr_to_ocr_pdf(context, executor)
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
|
|
||||||
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
@@ -22,7 +21,6 @@ from ocrmypdf._graft import OcrGrafter
|
|||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._pipeline import (
|
from ocrmypdf._pipeline import (
|
||||||
copy_final,
|
copy_final,
|
||||||
get_pdfinfo,
|
|
||||||
is_ocr_required,
|
is_ocr_required,
|
||||||
merge_sidecars,
|
merge_sidecars,
|
||||||
ocr_engine_hocr,
|
ocr_engine_hocr,
|
||||||
@@ -34,6 +32,7 @@ from ocrmypdf._pipeline import (
|
|||||||
from ocrmypdf._pipelines._common import (
|
from ocrmypdf._pipelines._common import (
|
||||||
PageResult,
|
PageResult,
|
||||||
cli_exception_handler,
|
cli_exception_handler,
|
||||||
|
do_get_pdfinfo,
|
||||||
manage_debug_log_handler,
|
manage_debug_log_handler,
|
||||||
manage_work_folder,
|
manage_work_folder,
|
||||||
postprocess,
|
postprocess,
|
||||||
@@ -80,7 +79,6 @@ def _exec_page_sync(page_context: PageContext) -> PageResult:
|
|||||||
page_context
|
page_context
|
||||||
)
|
)
|
||||||
ocr_out, text_out = _image_to_ocr_text(page_context, ocr_image_out)
|
ocr_out, text_out = _image_to_ocr_text(page_context, ocr_image_out)
|
||||||
|
|
||||||
return PageResult(
|
return PageResult(
|
||||||
pageno=page_context.pageno,
|
pageno=page_context.pageno,
|
||||||
pdf_page_from_image=pdf_page_from_image_out,
|
pdf_page_from_image=pdf_page_from_image_out,
|
||||||
@@ -105,14 +103,14 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
|||||||
try:
|
try:
|
||||||
set_thread_pageno(result.pageno + 1)
|
set_thread_pageno(result.pageno + 1)
|
||||||
sidecars[result.pageno] = result.text
|
sidecars[result.pageno] = result.text
|
||||||
pbar.update()
|
pbar.update(0.5)
|
||||||
ocrgraft.graft_page(
|
ocrgraft.graft_page(
|
||||||
pageno=result.pageno,
|
pageno=result.pageno,
|
||||||
image=result.pdf_page_from_image,
|
image=result.pdf_page_from_image,
|
||||||
textpdf=result.ocr,
|
textpdf=result.ocr,
|
||||||
autorotate_correction=result.orientation_correction,
|
autorotate_correction=result.orientation_correction,
|
||||||
)
|
)
|
||||||
pbar.update()
|
pbar.update(0.5)
|
||||||
finally:
|
finally:
|
||||||
set_thread_pageno(None)
|
set_thread_pageno(None)
|
||||||
|
|
||||||
@@ -120,10 +118,9 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
|||||||
use_threads=options.use_threads,
|
use_threads=options.use_threads,
|
||||||
max_workers=max_workers,
|
max_workers=max_workers,
|
||||||
progress_kwargs=dict(
|
progress_kwargs=dict(
|
||||||
total=(2 * len(context.pdfinfo)),
|
total=len(context.pdfinfo),
|
||||||
desc='OCR' if options.tesseract_timeout > 0 else 'Image processing',
|
desc='OCR' if options.tesseract_timeout > 0 else 'Image processing',
|
||||||
unit='page',
|
unit='page',
|
||||||
unit_scale=0.5,
|
|
||||||
disable=not options.progress_bar,
|
disable=not options.progress_bar,
|
||||||
),
|
),
|
||||||
worker_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS),
|
worker_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS),
|
||||||
@@ -156,12 +153,13 @@ def _run_pipeline(
|
|||||||
options: argparse.Namespace,
|
options: argparse.Namespace,
|
||||||
plugin_manager: OcrmypdfPluginManager,
|
plugin_manager: OcrmypdfPluginManager,
|
||||||
) -> ExitCode:
|
) -> ExitCode:
|
||||||
with manage_work_folder(
|
with (
|
||||||
work_folder=Path(mkdtemp(prefix="ocrmypdf.io.")),
|
manage_work_folder(
|
||||||
retain=options.keep_temporary_files,
|
work_folder=Path(mkdtemp(prefix="ocrmypdf.io.")),
|
||||||
print_location=options.keep_temporary_files,
|
retain=options.keep_temporary_files,
|
||||||
) as work_folder, manage_debug_log_handler(
|
print_location=options.keep_temporary_files,
|
||||||
options=options, work_folder=work_folder
|
) as work_folder,
|
||||||
|
manage_debug_log_handler(options=options, work_folder=work_folder),
|
||||||
):
|
):
|
||||||
executor = setup_pipeline(options, plugin_manager)
|
executor = setup_pipeline(options, plugin_manager)
|
||||||
check_requested_output_file(options)
|
check_requested_output_file(options)
|
||||||
@@ -173,16 +171,7 @@ def _run_pipeline(
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Gather pdfinfo and create context
|
# Gather pdfinfo and create context
|
||||||
pdfinfo = get_pdfinfo(
|
pdfinfo = do_get_pdfinfo(origin_pdf, executor, options)
|
||||||
origin_pdf,
|
|
||||||
executor=executor,
|
|
||||||
detailed_analysis=options.redo_ocr,
|
|
||||||
progbar=options.progress_bar,
|
|
||||||
max_workers=options.jobs,
|
|
||||||
use_threads=options.use_threads,
|
|
||||||
check_pages=options.pages,
|
|
||||||
)
|
|
||||||
|
|
||||||
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
||||||
|
|
||||||
# Validate options are okay for this pdf
|
# Validate options are okay for this pdf
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
|
|
||||||
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
@@ -18,13 +17,13 @@ import PIL
|
|||||||
from ocrmypdf._concurrent import Executor
|
from ocrmypdf._concurrent import Executor
|
||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._pipeline import (
|
from ocrmypdf._pipeline import (
|
||||||
get_pdfinfo,
|
|
||||||
is_ocr_required,
|
is_ocr_required,
|
||||||
ocr_engine_hocr,
|
ocr_engine_hocr,
|
||||||
validate_pdfinfo_options,
|
validate_pdfinfo_options,
|
||||||
)
|
)
|
||||||
from ocrmypdf._pipelines._common import (
|
from ocrmypdf._pipelines._common import (
|
||||||
HOCRResult,
|
HOCRResult,
|
||||||
|
do_get_pdfinfo,
|
||||||
manage_work_folder,
|
manage_work_folder,
|
||||||
process_page,
|
process_page,
|
||||||
set_thread_pageno,
|
set_thread_pageno,
|
||||||
@@ -95,18 +94,11 @@ def run_hocr_pipeline(
|
|||||||
work_folder=options.output_folder, retain=True, print_location=False
|
work_folder=options.output_folder, retain=True, print_location=False
|
||||||
) as work_folder:
|
) as work_folder:
|
||||||
executor = setup_pipeline(options, plugin_manager)
|
executor = setup_pipeline(options, plugin_manager)
|
||||||
shutil.copy2(options.input_file, work_folder / 'origin.pdf')
|
origin_pdf = work_folder / 'origin.pdf'
|
||||||
|
shutil.copy2(options.input_file, origin_pdf)
|
||||||
|
|
||||||
# Gather pdfinfo and create context
|
# Gather pdfinfo and create context
|
||||||
pdfinfo = get_pdfinfo(
|
pdfinfo = do_get_pdfinfo(origin_pdf, executor, options)
|
||||||
options.input_file,
|
|
||||||
executor=executor,
|
|
||||||
detailed_analysis=options.redo_ocr,
|
|
||||||
progbar=options.progress_bar,
|
|
||||||
max_workers=options.jobs,
|
|
||||||
use_threads=options.use_threads,
|
|
||||||
check_pages=options.pages,
|
|
||||||
)
|
|
||||||
context = PdfContext(
|
context = PdfContext(
|
||||||
options, work_folder, options.input_file, pdfinfo, plugin_manager
|
options, work_folder, options.input_file, pdfinfo, plugin_manager
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -66,7 +66,7 @@ class ProgressBar(Protocol):
|
|||||||
def __exit__(self, *args):
|
def __exit__(self, *args):
|
||||||
"""Exit a progress bar context."""
|
"""Exit a progress bar context."""
|
||||||
|
|
||||||
def update(self, n=1):
|
def update(self, n=1, *, completed=None):
|
||||||
"""Update the progress bar by an increment.
|
"""Update the progress bar by an increment.
|
||||||
|
|
||||||
For use within a progress bar context.
|
For use within a progress bar context.
|
||||||
@@ -85,7 +85,7 @@ class NullProgressBar:
|
|||||||
def __exit__(self, exc_type, exc_value, traceback):
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def update(self, _arg=None):
|
def update(self, _arg=None, *, completed=None):
|
||||||
return
|
return
|
||||||
|
|
||||||
|
|
||||||
@@ -103,6 +103,7 @@ class RichProgressBar:
|
|||||||
disable: bool = False,
|
disable: bool = False,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
|
self._entered = False
|
||||||
self.progress = Progress(
|
self.progress = Progress(
|
||||||
TextColumn(
|
TextColumn(
|
||||||
"[progress.description]{task.description}",
|
"[progress.description]{task.description}",
|
||||||
@@ -130,6 +131,7 @@ class RichProgressBar:
|
|||||||
|
|
||||||
def __enter__(self):
|
def __enter__(self):
|
||||||
self.progress.start()
|
self.progress.start()
|
||||||
|
self._entered = True
|
||||||
return self
|
return self
|
||||||
|
|
||||||
def __exit__(self, exc_type, exc_value, traceback):
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
@@ -137,6 +139,10 @@ class RichProgressBar:
|
|||||||
self.progress.stop()
|
self.progress.stop()
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def update(self, value=None):
|
def update(self, n=1, *, completed=None):
|
||||||
advance = self.unit_scale if value is None else value
|
assert self._entered, "Progress bar must be entered before updating"
|
||||||
self.progress.update(self.progress_bar, advance=advance)
|
if completed is None:
|
||||||
|
advance = self.unit_scale if n is None else n
|
||||||
|
self.progress.update(self.progress_bar, advance=advance)
|
||||||
|
else:
|
||||||
|
self.progress.update(self.progress_bar, completed=completed)
|
||||||
|
|||||||
+16
-33
@@ -20,6 +20,7 @@ import pikepdf
|
|||||||
import PIL
|
import PIL
|
||||||
from pluggy import PluginManager
|
from pluggy import PluginManager
|
||||||
|
|
||||||
|
from ocrmypdf._defaults import DEFAULT_LANGUAGE, DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||||
from ocrmypdf._exec import unpaper
|
from ocrmypdf._exec import unpaper
|
||||||
from ocrmypdf.exceptions import (
|
from ocrmypdf.exceptions import (
|
||||||
BadArgsError,
|
BadArgsError,
|
||||||
@@ -27,21 +28,18 @@ from ocrmypdf.exceptions import (
|
|||||||
MissingDependencyError,
|
MissingDependencyError,
|
||||||
OutputFileAccessError,
|
OutputFileAccessError,
|
||||||
)
|
)
|
||||||
from ocrmypdf.helpers import is_file_writable, monotonic, safe_symlink
|
from ocrmypdf.helpers import (
|
||||||
from ocrmypdf.hocrtransform import HOCR_OK_LANGS
|
is_file_writable,
|
||||||
|
monotonic,
|
||||||
|
running_in_docker,
|
||||||
|
running_in_snap,
|
||||||
|
safe_symlink,
|
||||||
|
)
|
||||||
from ocrmypdf.subprocess import check_external_program
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
# -------------
|
|
||||||
# External dependencies
|
|
||||||
|
|
||||||
DEFAULT_LANGUAGE = 'eng' # Enforce English hegemony
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
# --------
|
|
||||||
|
|
||||||
|
|
||||||
def check_platform() -> None:
|
def check_platform() -> None:
|
||||||
if sys.maxsize <= 2**32: # pragma: no cover
|
if sys.maxsize <= 2**32: # pragma: no cover
|
||||||
log.warning(
|
log.warning(
|
||||||
@@ -62,6 +60,7 @@ def check_options_languages(
|
|||||||
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
||||||
if not ocr_engine_languages:
|
if not ocr_engine_languages:
|
||||||
return
|
return
|
||||||
|
|
||||||
missing_languages = set(options.languages) - set(ocr_engine_languages)
|
missing_languages = set(options.languages) - set(ocr_engine_languages)
|
||||||
if missing_languages:
|
if missing_languages:
|
||||||
lang_text = '\n'.join(lang for lang in missing_languages)
|
lang_text = '\n'.join(lang for lang in missing_languages)
|
||||||
@@ -83,15 +82,6 @@ def check_options_languages(
|
|||||||
|
|
||||||
|
|
||||||
def check_options_output(options: Namespace) -> None:
|
def check_options_output(options: Namespace) -> None:
|
||||||
is_latin = set(options.languages).issubset(HOCR_OK_LANGS)
|
|
||||||
|
|
||||||
if options.pdf_renderer.startswith('hocr') and not is_latin:
|
|
||||||
log.warning(
|
|
||||||
"The 'hocr' PDF renderer is known to cause problems with one "
|
|
||||||
"or more of the languages in your document. Use "
|
|
||||||
"`--pdf-renderer auto` (the default) to avoid this issue."
|
|
||||||
)
|
|
||||||
|
|
||||||
if options.output_type == 'none' and options.output_file not in (os.devnull, '-'):
|
if options.output_type == 'none' and options.output_file not in (os.devnull, '-'):
|
||||||
raise BadArgsError(
|
raise BadArgsError(
|
||||||
"Since you specified `--output-type none`, the output file "
|
"Since you specified `--output-type none`, the output file "
|
||||||
@@ -140,6 +130,11 @@ def check_options_preprocessing(options: Namespace) -> None:
|
|||||||
options.clean = True
|
options.clean = True
|
||||||
if options.unpaper_args and not options.clean:
|
if options.unpaper_args and not options.clean:
|
||||||
raise BadArgsError("--clean is required for --unpaper-args")
|
raise BadArgsError("--clean is required for --unpaper-args")
|
||||||
|
if (
|
||||||
|
options.rotate_pages_threshold != DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||||
|
and not options.rotate_pages
|
||||||
|
):
|
||||||
|
raise BadArgsError("--rotate-pages is required for --rotate-pages-threshold")
|
||||||
if options.clean:
|
if options.clean:
|
||||||
check_external_program(
|
check_external_program(
|
||||||
program='unpaper',
|
program='unpaper',
|
||||||
@@ -248,18 +243,6 @@ def check_options(options: Namespace, plugin_manager: PluginManager) -> None:
|
|||||||
_check_plugin_options(options, plugin_manager)
|
_check_plugin_options(options, plugin_manager)
|
||||||
|
|
||||||
|
|
||||||
def _in_docker():
|
|
||||||
return Path('/.dockerenv').exists()
|
|
||||||
|
|
||||||
|
|
||||||
def _in_snap():
|
|
||||||
try:
|
|
||||||
cgroup_text = Path('/proc/self/cgroup').read_text()
|
|
||||||
return 'snap.ocrmypdf' in cgroup_text
|
|
||||||
except FileNotFoundError:
|
|
||||||
return False
|
|
||||||
|
|
||||||
|
|
||||||
def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]:
|
def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]:
|
||||||
if options.input_file == '-':
|
if options.input_file == '-':
|
||||||
# stdin
|
# stdin
|
||||||
@@ -283,7 +266,7 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
|
|||||||
return target, os.fspath(options.input_file)
|
return target, os.fspath(options.input_file)
|
||||||
except FileNotFoundError as e:
|
except FileNotFoundError as e:
|
||||||
msg = f"File not found - {options.input_file}"
|
msg = f"File not found - {options.input_file}"
|
||||||
if _in_docker(): # pragma: no cover
|
if running_in_docker(): # pragma: no cover
|
||||||
msg += (
|
msg += (
|
||||||
"\nDocker cannot access your working directory unless you "
|
"\nDocker cannot access your working directory unless you "
|
||||||
"explicitly share it with the Docker container and set up"
|
"explicitly share it with the Docker container and set up"
|
||||||
@@ -293,7 +276,7 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
|
|||||||
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf"
|
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf"
|
||||||
"\n"
|
"\n"
|
||||||
)
|
)
|
||||||
elif _in_snap(): # pragma: no cover
|
elif running_in_snap(): # pragma: no cover
|
||||||
msg += (
|
msg += (
|
||||||
"\nSnap applications cannot access files outside of "
|
"\nSnap applications cannot access files outside of "
|
||||||
"your home directory unless you explicitly allow it. "
|
"your home directory unless you explicitly allow it. "
|
||||||
|
|||||||
@@ -1,16 +0,0 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
|
||||||
|
|
||||||
"""Get version by introspecting package information.
|
|
||||||
|
|
||||||
OCRmyPDF uses setuptools_scm to derive version from git tags.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from importlib.metadata import version as _package_version
|
|
||||||
|
|
||||||
PROGRAM_NAME = 'ocrmypdf'
|
|
||||||
|
|
||||||
# Official PEP 396
|
|
||||||
__version__ = _package_version('ocrmypdf')
|
|
||||||
+18
-13
@@ -14,7 +14,7 @@ from collections.abc import Iterable, Sequence
|
|||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
from io import IOBase
|
from io import IOBase
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import AnyStr, BinaryIO, Union
|
from typing import BinaryIO
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
import pluggy
|
import pluggy
|
||||||
@@ -28,8 +28,8 @@ from ocrmypdf._validation import check_options
|
|||||||
from ocrmypdf.cli import ArgumentParser, get_parser
|
from ocrmypdf.cli import ArgumentParser, get_parser
|
||||||
from ocrmypdf.helpers import is_iterable_notstr
|
from ocrmypdf.helpers import is_iterable_notstr
|
||||||
|
|
||||||
StrPath = Union[Path, AnyStr]
|
StrPath = Path | str | bytes
|
||||||
PathOrIO = Union[BinaryIO, StrPath]
|
PathOrIO = BinaryIO | StrPath
|
||||||
|
|
||||||
# Installing plugins affects the global state of the Python interpreter,
|
# Installing plugins affects the global state of the Python interpreter,
|
||||||
# so we need to use a lock to prevent multiple threads from installing
|
# so we need to use a lock to prevent multiple threads from installing
|
||||||
@@ -140,9 +140,9 @@ def configure_logging(
|
|||||||
|
|
||||||
def _kwargs_to_cmdline(
|
def _kwargs_to_cmdline(
|
||||||
*, defer_kwargs: set[str], **kwargs
|
*, defer_kwargs: set[str], **kwargs
|
||||||
) -> tuple[list[str], dict[str, AnyStr]]:
|
) -> tuple[list[str | bytes], dict[str, str | bytes]]:
|
||||||
"""Convert kwargs to command line arguments."""
|
"""Convert kwargs to command line arguments."""
|
||||||
cmdline = []
|
cmdline: list[str | bytes] = []
|
||||||
deferred = {}
|
deferred = {}
|
||||||
for arg, val in kwargs.items():
|
for arg, val in kwargs.items():
|
||||||
if val is None:
|
if val is None:
|
||||||
@@ -169,7 +169,7 @@ def _kwargs_to_cmdline(
|
|||||||
|
|
||||||
# We have a parameter
|
# We have a parameter
|
||||||
cmdline.append(f"--{cmd_style_arg}")
|
cmdline.append(f"--{cmd_style_arg}")
|
||||||
if isinstance(val, (int, float)):
|
if isinstance(val, int | float):
|
||||||
cmdline.append(str(val))
|
cmdline.append(str(val))
|
||||||
elif isinstance(val, str):
|
elif isinstance(val, str):
|
||||||
cmdline.append(val)
|
cmdline.append(val)
|
||||||
@@ -201,14 +201,17 @@ def create_options(
|
|||||||
defer_kwargs={'progress_bar', 'plugins', 'parser', 'input_file', 'output_file'},
|
defer_kwargs={'progress_bar', 'plugins', 'parser', 'input_file', 'output_file'},
|
||||||
**kwargs,
|
**kwargs,
|
||||||
)
|
)
|
||||||
if isinstance(input_file, (BinaryIO, IOBase)):
|
if isinstance(input_file, BinaryIO | IOBase):
|
||||||
cmdline.append('stream://input_file')
|
cmdline.append('stream://input_file')
|
||||||
else:
|
else:
|
||||||
cmdline.append(os.fspath(input_file))
|
cmdline.append(os.fspath(input_file))
|
||||||
if isinstance(output_file, (BinaryIO, IOBase)):
|
if isinstance(output_file, BinaryIO | IOBase):
|
||||||
cmdline.append('stream://output_file')
|
cmdline.append('stream://output_file')
|
||||||
else:
|
else:
|
||||||
cmdline.append(os.fspath(output_file))
|
cmdline.append(os.fspath(output_file))
|
||||||
|
if 'sidecar' in kwargs and isinstance(kwargs['sidecar'], BinaryIO | IOBase):
|
||||||
|
cmdline.append('--sidecar')
|
||||||
|
cmdline.append('stream://sidecar')
|
||||||
|
|
||||||
parser.enable_api_mode()
|
parser.enable_api_mode()
|
||||||
options = parser.parse_args(cmdline)
|
options = parser.parse_args(cmdline)
|
||||||
@@ -219,6 +222,8 @@ def create_options(
|
|||||||
options.input_file = input_file
|
options.input_file = input_file
|
||||||
if options.output_file == 'stream://output_file':
|
if options.output_file == 'stream://output_file':
|
||||||
options.output_file = output_file
|
options.output_file = output_file
|
||||||
|
if options.sidecar == 'stream://sidecar':
|
||||||
|
options.sidecar = kwargs['sidecar']
|
||||||
|
|
||||||
return options
|
return options
|
||||||
|
|
||||||
@@ -230,7 +235,7 @@ def ocr( # noqa: D417
|
|||||||
language: Iterable[str] | None = None,
|
language: Iterable[str] | None = None,
|
||||||
image_dpi: int | None = None,
|
image_dpi: int | None = None,
|
||||||
output_type: str | None = None,
|
output_type: str | None = None,
|
||||||
sidecar: StrPath | None = None,
|
sidecar: PathOrIO | None = None,
|
||||||
jobs: int | None = None,
|
jobs: int | None = None,
|
||||||
use_threads: bool | None = None,
|
use_threads: bool | None = None,
|
||||||
title: str | None = None,
|
title: str | None = None,
|
||||||
@@ -274,7 +279,7 @@ def ocr( # noqa: D417
|
|||||||
fast_web_view: float | None = None,
|
fast_web_view: float | None = None,
|
||||||
continue_on_soft_render_error: bool | None = None,
|
continue_on_soft_render_error: bool | None = None,
|
||||||
invalidate_digital_signatures: bool | None = None,
|
invalidate_digital_signatures: bool | None = None,
|
||||||
plugins: Iterable[StrPath] | None = None,
|
plugins: Iterable[Path | str] | None = None,
|
||||||
plugin_manager=None,
|
plugin_manager=None,
|
||||||
keep_temporary_files: bool | None = None,
|
keep_temporary_files: bool | None = None,
|
||||||
progress_bar: bool | None = None,
|
progress_bar: bool | None = None,
|
||||||
@@ -343,7 +348,7 @@ def ocr( # noqa: D417
|
|||||||
|
|
||||||
if not plugins:
|
if not plugins:
|
||||||
plugins = []
|
plugins = []
|
||||||
elif isinstance(plugins, (str, Path)):
|
elif isinstance(plugins, str | Path):
|
||||||
plugins = [plugins]
|
plugins = [plugins]
|
||||||
else:
|
else:
|
||||||
plugins = list(plugins)
|
plugins = list(plugins)
|
||||||
@@ -415,7 +420,7 @@ def _pdf_to_hocr( # noqa: D417
|
|||||||
continue_on_soft_render_error: bool | None = None,
|
continue_on_soft_render_error: bool | None = None,
|
||||||
invalidate_digital_signatures: bool | None = None,
|
invalidate_digital_signatures: bool | None = None,
|
||||||
plugin_manager=None,
|
plugin_manager=None,
|
||||||
plugins: Sequence[StrPath] | None = None,
|
plugins: Sequence[Path | str] | None = None,
|
||||||
keep_temporary_files: bool | None = None,
|
keep_temporary_files: bool | None = None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
@@ -486,7 +491,7 @@ def _hocr_to_ocr_pdf( # noqa: D417
|
|||||||
color_conversion_strategy: str | None = None,
|
color_conversion_strategy: str | None = None,
|
||||||
fast_web_view: float | None = None,
|
fast_web_view: float | None = None,
|
||||||
plugin_manager=None,
|
plugin_manager=None,
|
||||||
plugins: Sequence[StrPath] | None = None,
|
plugins: Sequence[Path | str] | None = None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
"""Run OCRmyPDF on a work folder and produce an output PDF.
|
"""Run OCRmyPDF on a work folder and produce an output PDF.
|
||||||
|
|||||||
@@ -12,10 +12,10 @@ import queue
|
|||||||
import signal
|
import signal
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
from collections.abc import Iterable
|
from collections.abc import Callable, Iterable
|
||||||
from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed
|
from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from typing import Callable, Union
|
from typing import Union
|
||||||
|
|
||||||
from rich.console import Console as RichConsole
|
from rich.console import Console as RichConsole
|
||||||
|
|
||||||
@@ -25,13 +25,13 @@ from ocrmypdf._progressbar import RichProgressBar
|
|||||||
from ocrmypdf.exceptions import InputFileError
|
from ocrmypdf.exceptions import InputFileError
|
||||||
from ocrmypdf.helpers import remove_all_log_handlers
|
from ocrmypdf.helpers import remove_all_log_handlers
|
||||||
|
|
||||||
FuturesExecutorClass = Union[type[ThreadPoolExecutor], type[ProcessPoolExecutor]]
|
FuturesExecutorClass = Union[ # noqa: UP007
|
||||||
Queue = Union[multiprocessing.Queue, queue.Queue]
|
type[ThreadPoolExecutor], type[ProcessPoolExecutor]
|
||||||
|
]
|
||||||
|
Queue = Union[multiprocessing.Queue, queue.Queue] # noqa: UP007
|
||||||
UserInit = Callable[[], None]
|
UserInit = Callable[[], None]
|
||||||
WorkerInit = Callable[[Queue, UserInit, int], None]
|
WorkerInit = Callable[[Queue, UserInit, int], None]
|
||||||
|
|
||||||
RichTqdmProgressAdapter = RichProgressBar # Deprecated shim; remove in OCRmyPDF 16
|
|
||||||
|
|
||||||
|
|
||||||
def log_listener(q: Queue):
|
def log_listener(q: Queue):
|
||||||
"""Listen to the worker processes and forward the messages to logging.
|
"""Listen to the worker processes and forward the messages to logging.
|
||||||
@@ -130,11 +130,14 @@ class StandardExecutor(Executor):
|
|||||||
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
||||||
listener.start()
|
listener.start()
|
||||||
|
|
||||||
with self.pbar_class(**progress_kwargs) as pbar, executor_class(
|
with (
|
||||||
max_workers=max_workers,
|
self.pbar_class(**progress_kwargs) as pbar,
|
||||||
initializer=initializer,
|
executor_class(
|
||||||
initargs=(log_queue, worker_initializer, logging.getLogger("").level),
|
max_workers=max_workers,
|
||||||
) as executor:
|
initializer=initializer,
|
||||||
|
initargs=(log_queue, worker_initializer, logging.getLogger("").level),
|
||||||
|
) as executor,
|
||||||
|
):
|
||||||
futures = [executor.submit(task, *args) for args in task_arguments]
|
futures = [executor.submit(task, *args) for args in task_arguments]
|
||||||
try:
|
try:
|
||||||
for future in as_completed(futures):
|
for future in as_completed(futures):
|
||||||
|
|||||||
@@ -8,7 +8,5 @@ from ocrmypdf import hookimpl
|
|||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def filter_pdf_page(
|
def filter_pdf_page(page, image_filename, output_pdf): # pylint: disable=unused-argument
|
||||||
page, image_filename, output_pdf
|
|
||||||
): # pylint: disable=unused-argument
|
|
||||||
return output_pdf
|
return output_pdf
|
||||||
|
|||||||
@@ -6,6 +6,8 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
|
from packaging.version import Version
|
||||||
|
|
||||||
from ocrmypdf import hookimpl
|
from ocrmypdf import hookimpl
|
||||||
from ocrmypdf._exec import ghostscript
|
from ocrmypdf._exec import ghostscript
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
@@ -15,7 +17,7 @@ log = logging.getLogger(__name__)
|
|||||||
|
|
||||||
# Currently all blacklisted versions are lower than 9.55, so none need to
|
# Currently all blacklisted versions are lower than 9.55, so none need to
|
||||||
# be added here. If a future version is blacklisted, add it here.
|
# be added here. If a future version is blacklisted, add it here.
|
||||||
BLACKLISTED_GS_VERSIONS: frozenset[str] = frozenset()
|
BLACKLISTED_GS_VERSIONS: frozenset[Version] = frozenset()
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
@@ -52,14 +54,24 @@ def check_options(options):
|
|||||||
program='gs',
|
program='gs',
|
||||||
package='ghostscript',
|
package='ghostscript',
|
||||||
version_checker=ghostscript.version,
|
version_checker=ghostscript.version,
|
||||||
need_version='9.55', # Ubuntu 22.04's version
|
need_version='9.54', # RHEL 9's version; Ubuntu 22.04 has 9.55
|
||||||
)
|
)
|
||||||
gs_version = ghostscript.version()
|
gs_version = ghostscript.version()
|
||||||
if gs_version in BLACKLISTED_GS_VERSIONS:
|
if gs_version in BLACKLISTED_GS_VERSIONS:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
f"Ghostscript {gs_version} contains serious regressions and is not "
|
f"Ghostscript {gs_version} contains serious regressions and is not "
|
||||||
"supported. Please upgrade to a newer version, or downgrade to the "
|
"supported. Please upgrade to a newer version."
|
||||||
"previous version."
|
)
|
||||||
|
if Version('10.0.0') <= gs_version < Version('10.02.1') and (
|
||||||
|
options.skip_text or options.redo_ocr
|
||||||
|
):
|
||||||
|
raise MissingDependencyError(
|
||||||
|
f"Ghostscript 10.0.0 through 10.02.0 (your version: {gs_version}) "
|
||||||
|
"contain serious regressions that corrupt PDFs with existing text, "
|
||||||
|
"such as those processed using --skip-text or --redo-ocr. "
|
||||||
|
"Please upgrade to a "
|
||||||
|
"newer version, or use --output-type pdf to avoid Ghostscript, or "
|
||||||
|
"use --force-ocr to discard existing text."
|
||||||
)
|
)
|
||||||
|
|
||||||
if options.output_type == 'pdfa':
|
if options.output_type == 'pdfa':
|
||||||
@@ -117,7 +129,7 @@ def generate_pdfa(
|
|||||||
):
|
):
|
||||||
"""Generate a PDF/A from the list of PDF pages and PDF/A metadata."""
|
"""Generate a PDF/A from the list of PDF pages and PDF/A metadata."""
|
||||||
ghostscript.generate_pdfa(
|
ghostscript.generate_pdfa(
|
||||||
pdf_pages=[*pdf_pages, pdfmark],
|
pdf_pages=[pdfmark, *pdf_pages],
|
||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
compression=context.options.pdfa_image_compression,
|
compression=context.options.pdfa_image_compression,
|
||||||
color_conversion_strategy=context.options.color_conversion_strategy,
|
color_conversion_strategy=context.options.color_conversion_strategy,
|
||||||
|
|||||||
@@ -2,9 +2,9 @@
|
|||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
"""Built-in plugin to implement OCR using Tesseract."""
|
"""Built-in plugin to implement OCR using Tesseract."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
|
||||||
@@ -14,6 +14,7 @@ from ocrmypdf import hookimpl
|
|||||||
from ocrmypdf._exec import tesseract
|
from ocrmypdf._exec import tesseract
|
||||||
from ocrmypdf._jobcontext import PageContext
|
from ocrmypdf._jobcontext import PageContext
|
||||||
from ocrmypdf.cli import numeric, str_to_int
|
from ocrmypdf.cli import numeric, str_to_int
|
||||||
|
from ocrmypdf.exceptions import BadArgsError, MissingDependencyError
|
||||||
from ocrmypdf.helpers import clamp
|
from ocrmypdf.helpers import clamp
|
||||||
from ocrmypdf.imageops import calculate_downsample, downsample_image
|
from ocrmypdf.imageops import calculate_downsample, downsample_image
|
||||||
from ocrmypdf.pluginspec import OcrEngine
|
from ocrmypdf.pluginspec import OcrEngine
|
||||||
@@ -94,7 +95,8 @@ def add_options(parser):
|
|||||||
)
|
)
|
||||||
tess.add_argument(
|
tess.add_argument(
|
||||||
'--tesseract-downsample-large-images',
|
'--tesseract-downsample-large-images',
|
||||||
action='store_true',
|
action=argparse.BooleanOptionalAction,
|
||||||
|
default=True,
|
||||||
help=(
|
help=(
|
||||||
"Downsample large images before OCR. Tesseract has an upper limit on the "
|
"Downsample large images before OCR. Tesseract has an upper limit on the "
|
||||||
"size images it will support. If this argument is given, OCRmyPDF will "
|
"size images it will support. If this argument is given, OCRmyPDF will "
|
||||||
@@ -143,10 +145,20 @@ def check_options(options):
|
|||||||
need_version='4.1.1', # Ubuntu 22.04 version (also 20.04)
|
need_version='4.1.1', # Ubuntu 22.04 version (also 20.04)
|
||||||
version_parser=tesseract.TesseractVersion,
|
version_parser=tesseract.TesseractVersion,
|
||||||
)
|
)
|
||||||
|
tess_version = tesseract.version()
|
||||||
|
if tess_version == tesseract.TesseractVersion('5.4.0'):
|
||||||
|
raise MissingDependencyError(
|
||||||
|
"Tesseract 5.4.0 is not supported due to regressions in this version. "
|
||||||
|
"Please upgrade to a newer or supported older version."
|
||||||
|
)
|
||||||
|
|
||||||
# Decide on what renderer to use
|
# Decide on what renderer to use
|
||||||
if options.pdf_renderer == 'auto':
|
if options.pdf_renderer == 'auto':
|
||||||
options.pdf_renderer = 'sandwich'
|
if {'ara', 'heb', 'fas', 'per'} & set(options.languages):
|
||||||
|
log.info("Using sandwich renderer since there is an RTL language")
|
||||||
|
options.pdf_renderer = 'sandwich'
|
||||||
|
else:
|
||||||
|
options.pdf_renderer = 'hocr'
|
||||||
|
|
||||||
if not tesseract.has_thresholding() and options.tesseract_thresholding != 0:
|
if not tesseract.has_thresholding() and options.tesseract_thresholding != 0:
|
||||||
log.warning(
|
log.warning(
|
||||||
@@ -159,6 +171,14 @@ def check_options(options):
|
|||||||
"The --tesseract-pagesegmode argument you select will disable OCR. "
|
"The --tesseract-pagesegmode argument you select will disable OCR. "
|
||||||
"This may cause processing to fail."
|
"This may cause processing to fail."
|
||||||
)
|
)
|
||||||
|
DENIED_LANGUAGES = {'equ', 'osd'}
|
||||||
|
if DENIED_LANGUAGES & set(options.languages):
|
||||||
|
raise BadArgsError(
|
||||||
|
"The following languages for Tesseract's internal use and should not "
|
||||||
|
"be issued explicitly: "
|
||||||
|
f"{', '.join(DENIED_LANGUAGES & set(options.languages))}\n"
|
||||||
|
"Remove them from the -l/--language argument."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
@@ -216,7 +236,7 @@ class TesseractOcrEngine(OcrEngine):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def creator_tag(options):
|
def creator_tag(options):
|
||||||
tag = '-PDF' if options.pdf_renderer == 'sandwich' else ''
|
tag = '-PDF' if options.pdf_renderer == 'sandwich' else '-hOCR'
|
||||||
return f"Tesseract OCR{tag} {TesseractOcrEngine.version()}"
|
return f"Tesseract OCR{tag} {TesseractOcrEngine.version()}"
|
||||||
|
|
||||||
def __str__(self):
|
def __str__(self):
|
||||||
|
|||||||
+6
-5
@@ -6,10 +6,11 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
from collections.abc import Mapping
|
from collections.abc import Callable, Mapping
|
||||||
from typing import Any, Callable, TypeVar
|
from typing import Any, TypeVar
|
||||||
|
|
||||||
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||||
|
from ocrmypdf._defaults import PROGRAM_NAME as _PROGRAM_NAME
|
||||||
from ocrmypdf._version import __version__ as _VERSION
|
from ocrmypdf._version import __version__ as _VERSION
|
||||||
|
|
||||||
T = TypeVar('T', int, float)
|
T = TypeVar('T', int, float)
|
||||||
@@ -390,7 +391,7 @@ Online documentation is located at:
|
|||||||
action='store',
|
action='store',
|
||||||
type=numeric(float, 0),
|
type=numeric(float, 0),
|
||||||
metavar='MPixels',
|
metavar='MPixels',
|
||||||
help="Set maximum number of pixels to unpack before treating an image as a "
|
help="Set maximum number of megapixels to unpack before treating an image as a "
|
||||||
"decompression bomb",
|
"decompression bomb",
|
||||||
default=250.0,
|
default=250.0,
|
||||||
)
|
)
|
||||||
@@ -403,7 +404,7 @@ Online documentation is located at:
|
|||||||
)
|
)
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--rotate-pages-threshold',
|
'--rotate-pages-threshold',
|
||||||
default=14.0,
|
default=DEFAULT_ROTATE_PAGES_THRESHOLD,
|
||||||
type=numeric(float, 0, 1000),
|
type=numeric(float, 0, 1000),
|
||||||
metavar='CONFIDENCE',
|
metavar='CONFIDENCE',
|
||||||
help="Only rotate pages when confidence is above this value (arbitrary "
|
help="Only rotate pages when confidence is above this value (arbitrary "
|
||||||
|
|||||||
Binary file not shown.
@@ -137,3 +137,16 @@ class TaggedPDFError(InputFileError):
|
|||||||
override this error.
|
override this error.
|
||||||
"""
|
"""
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class ColorConversionNeededError(BadArgsError):
|
||||||
|
"""PDF needs color conversion."""
|
||||||
|
|
||||||
|
message = dedent(
|
||||||
|
"""\
|
||||||
|
The input PDF has an unusual color space. Use
|
||||||
|
--color-conversion-strategy to convert to a common color space
|
||||||
|
such as RGB, or use --output-type pdf to skip PDF/A conversion
|
||||||
|
and retain the original color space.
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
|||||||
@@ -20,13 +20,12 @@ from __future__ import annotations
|
|||||||
import logging
|
import logging
|
||||||
import logging.handlers
|
import logging.handlers
|
||||||
import signal
|
import signal
|
||||||
from collections.abc import Iterable, Iterator
|
from collections.abc import Callable, Iterable, Iterator
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from enum import Enum, auto
|
from enum import Enum, auto
|
||||||
from itertools import islice, repeat, takewhile, zip_longest
|
from itertools import islice, repeat, takewhile, zip_longest
|
||||||
from multiprocessing import Pipe, Process
|
from multiprocessing import Pipe, Process
|
||||||
from multiprocessing.connection import Connection, wait
|
from multiprocessing.connection import Connection, wait
|
||||||
from typing import Callable
|
|
||||||
|
|
||||||
from ocrmypdf import Executor, hookimpl
|
from ocrmypdf import Executor, hookimpl
|
||||||
from ocrmypdf._concurrent import NullProgressBar
|
from ocrmypdf._concurrent import NullProgressBar
|
||||||
|
|||||||
+20
-3
@@ -10,7 +10,7 @@ import multiprocessing
|
|||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import warnings
|
import warnings
|
||||||
from collections.abc import Iterable, Sequence
|
from collections.abc import Callable, Iterable, Sequence
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from io import StringIO
|
from io import StringIO
|
||||||
@@ -19,13 +19,13 @@ from pathlib import Path
|
|||||||
from statistics import harmonic_mean
|
from statistics import harmonic_mean
|
||||||
from typing import (
|
from typing import (
|
||||||
Any,
|
Any,
|
||||||
Callable,
|
|
||||||
Generic,
|
Generic,
|
||||||
TypeVar,
|
TypeVar,
|
||||||
)
|
)
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
import pikepdf
|
import pikepdf
|
||||||
|
from deprecation import deprecated
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -136,6 +136,7 @@ class Resolution(Generic[T]):
|
|||||||
return self._isclose(self.x, other.x) and self._isclose(self.y, other.y)
|
return self._isclose(self.x, other.x) and self._isclose(self.y, other.y)
|
||||||
|
|
||||||
|
|
||||||
|
@deprecated(deprecated_in='15.4.0')
|
||||||
class NeverRaise(Exception):
|
class NeverRaise(Exception):
|
||||||
"""An exception that is never raised."""
|
"""An exception that is never raised."""
|
||||||
|
|
||||||
@@ -267,7 +268,9 @@ def check_pdf(input_file: Path) -> bool:
|
|||||||
return False
|
return False
|
||||||
else:
|
else:
|
||||||
with pdf:
|
with pdf:
|
||||||
messages = pdf.check()
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings('ignore', message=r'pikepdf.*JBIG2.*')
|
||||||
|
messages = pdf.check()
|
||||||
success = True
|
success = True
|
||||||
for msg in messages:
|
for msg in messages:
|
||||||
if 'error' in msg.lower():
|
if 'error' in msg.lower():
|
||||||
@@ -332,3 +335,17 @@ def pikepdf_enable_mmap() -> None:
|
|||||||
)
|
)
|
||||||
except AttributeError:
|
except AttributeError:
|
||||||
log.debug("pikepdf mmap not available")
|
log.debug("pikepdf mmap not available")
|
||||||
|
|
||||||
|
|
||||||
|
def running_in_docker() -> bool:
|
||||||
|
"""Returns True if we seem to be running in a Docker container."""
|
||||||
|
return Path('/.dockerenv').exists()
|
||||||
|
|
||||||
|
|
||||||
|
def running_in_snap() -> bool:
|
||||||
|
"""Returns True if we seem to be running in a Snap container."""
|
||||||
|
try:
|
||||||
|
cgroup_text = Path('/proc/self/cgroup').read_text()
|
||||||
|
return 'snap.ocrmypdf' in cgroup_text
|
||||||
|
except FileNotFoundError:
|
||||||
|
return False
|
||||||
|
|||||||
@@ -1,461 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
# SPDX-FileCopyrightText: 2010 Jonathan Brinley
|
|
||||||
# SPDX-FileCopyrightText: 2013-2014 Julien Pfefferkorn
|
|
||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
|
||||||
# SPDX-License-Identifier: MIT
|
|
||||||
|
|
||||||
"""Transform .hocr and page image to text PDF."""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
import os
|
|
||||||
import re
|
|
||||||
import warnings
|
|
||||||
from math import atan, cos, sin
|
|
||||||
from pathlib import Path
|
|
||||||
from typing import Any, NamedTuple
|
|
||||||
from xml.etree import ElementTree
|
|
||||||
|
|
||||||
with warnings.catch_warnings():
|
|
||||||
# reportlab uses deprecated load_module
|
|
||||||
# shim can be removed when we require reportlab >= 3.7
|
|
||||||
warnings.filterwarnings(
|
|
||||||
'ignore', category=DeprecationWarning, message=r".*load_module.*"
|
|
||||||
)
|
|
||||||
from reportlab.lib.colors import black, cyan, magenta, red
|
|
||||||
from reportlab.lib.units import inch
|
|
||||||
from reportlab.pdfgen.canvas import Canvas
|
|
||||||
|
|
||||||
# According to Wikipedia these languages are supported in the ISO-8859-1 character
|
|
||||||
# set, meaning reportlab can generate them and they are compatible with hocr,
|
|
||||||
# assuming Tesseract has the necessary languages installed. Note that there may
|
|
||||||
# not be language packs for them.
|
|
||||||
HOCR_OK_LANGS = frozenset(
|
|
||||||
[
|
|
||||||
# Languages fully covered by Latin-1:
|
|
||||||
'afr', # Afrikaans
|
|
||||||
'alb', # Albanian
|
|
||||||
'ast', # Leonese
|
|
||||||
'baq', # Basque
|
|
||||||
'bre', # Breton
|
|
||||||
'cos', # Corsican
|
|
||||||
'eng', # English
|
|
||||||
'eus', # Basque
|
|
||||||
'fao', # Faoese
|
|
||||||
'gla', # Scottish Gaelic
|
|
||||||
'glg', # Galician
|
|
||||||
'glv', # Manx
|
|
||||||
'ice', # Icelandic
|
|
||||||
'ind', # Indonesian
|
|
||||||
'isl', # Icelandic
|
|
||||||
'ita', # Italian
|
|
||||||
'ltz', # Luxembourgish
|
|
||||||
'mal', # Malay Rumi
|
|
||||||
'mga', # Irish
|
|
||||||
'nor', # Norwegian
|
|
||||||
'oci', # Occitan
|
|
||||||
'por', # Portugeuse
|
|
||||||
'roh', # Romansh
|
|
||||||
'sco', # Scots
|
|
||||||
'sma', # Sami
|
|
||||||
'spa', # Spanish
|
|
||||||
'sqi', # Albanian
|
|
||||||
'swa', # Swahili
|
|
||||||
'swe', # Swedish
|
|
||||||
'tgl', # Tagalog
|
|
||||||
'wln', # Walloon
|
|
||||||
# Languages supported by Latin-1 except for a few rare characters that OCR
|
|
||||||
# is probably not trained to recognize anyway:
|
|
||||||
'cat', # Catalan
|
|
||||||
'cym', # Welsh
|
|
||||||
'dan', # Danish
|
|
||||||
'deu', # German
|
|
||||||
'dut', # Dutch
|
|
||||||
'est', # Estonian
|
|
||||||
'fin', # Finnish
|
|
||||||
'fra', # French
|
|
||||||
'hun', # Hungarian
|
|
||||||
'kur', # Kurdish
|
|
||||||
'nld', # Dutch
|
|
||||||
'wel', # Welsh
|
|
||||||
]
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
Element = ElementTree.Element
|
|
||||||
|
|
||||||
|
|
||||||
class Rect(NamedTuple):
|
|
||||||
"""A rectangle for managing PDF coordinates."""
|
|
||||||
|
|
||||||
x1: Any
|
|
||||||
y1: Any
|
|
||||||
x2: Any
|
|
||||||
y2: Any
|
|
||||||
|
|
||||||
|
|
||||||
class HocrTransformError(Exception):
|
|
||||||
"""Error while applying hOCR transform."""
|
|
||||||
|
|
||||||
|
|
||||||
class HocrTransform:
|
|
||||||
"""A class for converting documents from the hOCR format.
|
|
||||||
|
|
||||||
For details of the hOCR format, see:
|
|
||||||
http://kba.cloud/hocr-spec/.
|
|
||||||
"""
|
|
||||||
|
|
||||||
box_pattern = re.compile(r'bbox((\s+\d+){4})')
|
|
||||||
baseline_pattern = re.compile(
|
|
||||||
r'''
|
|
||||||
baseline \s+
|
|
||||||
([\-\+]?\d*\.?\d*) \s+ # +/- decimal float
|
|
||||||
([\-\+]?\d+) # +/- int''',
|
|
||||||
re.VERBOSE,
|
|
||||||
)
|
|
||||||
ligatures = str.maketrans(
|
|
||||||
{'ff': 'ff', 'ffi': 'ffi', 'ffl': 'ffl', 'fi': 'fi', 'fl': 'fl'}
|
|
||||||
)
|
|
||||||
|
|
||||||
def __init__(self, *, hocr_filename: str | Path, dpi: float):
|
|
||||||
"""Initialize the HocrTransform object."""
|
|
||||||
self.dpi = dpi
|
|
||||||
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
|
||||||
|
|
||||||
# if the hOCR file has a namespace, ElementTree requires its use to
|
|
||||||
# find elements
|
|
||||||
matches = re.match(r'({.*})html', self.hocr.getroot().tag)
|
|
||||||
self.xmlns = ''
|
|
||||||
if matches:
|
|
||||||
self.xmlns = matches.group(1)
|
|
||||||
|
|
||||||
# get dimension in pt (not pixel!!!!) of the OCRed image
|
|
||||||
self.width, self.height = None, None
|
|
||||||
for div in self.hocr.findall(self._child_xpath('div', 'ocr_page')):
|
|
||||||
coords = self.element_coordinates(div)
|
|
||||||
pt_coords = self.pt_from_pixel(coords)
|
|
||||||
self.width = pt_coords.x2 - pt_coords.x1
|
|
||||||
self.height = pt_coords.y2 - pt_coords.y1
|
|
||||||
# there shouldn't be more than one, and if there is, we don't want
|
|
||||||
# it
|
|
||||||
break
|
|
||||||
if self.width is None or self.height is None:
|
|
||||||
raise HocrTransformError("hocr file is missing page dimensions")
|
|
||||||
|
|
||||||
def __str__(self): # pragma: no cover
|
|
||||||
"""Return the textual content of the HTML body."""
|
|
||||||
if self.hocr is None:
|
|
||||||
return ''
|
|
||||||
body = self.hocr.find(self._child_xpath('body'))
|
|
||||||
if body:
|
|
||||||
return self._get_element_text(body)
|
|
||||||
else:
|
|
||||||
return ''
|
|
||||||
|
|
||||||
def _get_element_text(self, element: Element):
|
|
||||||
"""Return the textual content of the element and its children."""
|
|
||||||
text = ''
|
|
||||||
if element.text is not None:
|
|
||||||
text += element.text
|
|
||||||
for child in element:
|
|
||||||
text += self._get_element_text(child)
|
|
||||||
if element.tail is not None:
|
|
||||||
text += element.tail
|
|
||||||
return text
|
|
||||||
|
|
||||||
@classmethod
|
|
||||||
def element_coordinates(cls, element: Element) -> Rect:
|
|
||||||
"""Get coordinates of the bounding box around an element."""
|
|
||||||
out = Rect._make(0 for _ in range(4))
|
|
||||||
if 'title' in element.attrib:
|
|
||||||
matches = cls.box_pattern.search(element.attrib['title'])
|
|
||||||
if matches:
|
|
||||||
coords = matches.group(1).split()
|
|
||||||
out = Rect._make(int(coords[n]) for n in range(4))
|
|
||||||
return out
|
|
||||||
|
|
||||||
@classmethod
|
|
||||||
def baseline(cls, element: Element) -> tuple[float, float]:
|
|
||||||
"""Get baseline's slope and intercept."""
|
|
||||||
if 'title' in element.attrib:
|
|
||||||
matches = cls.baseline_pattern.search(element.attrib['title'])
|
|
||||||
if matches:
|
|
||||||
return float(matches.group(1)), int(matches.group(2))
|
|
||||||
return (0.0, 0.0)
|
|
||||||
|
|
||||||
def pt_from_pixel(self, pxl) -> Rect:
|
|
||||||
"""Returns the quantity in PDF units (pt) given quantity in pixels."""
|
|
||||||
return Rect._make((c / self.dpi * inch) for c in pxl)
|
|
||||||
|
|
||||||
def _child_xpath(self, html_tag: str, html_class: str | None = None) -> str:
|
|
||||||
xpath = f".//{self.xmlns}{html_tag}"
|
|
||||||
if html_class:
|
|
||||||
xpath += f"[@class='{html_class}']"
|
|
||||||
return xpath
|
|
||||||
|
|
||||||
@classmethod
|
|
||||||
def replace_unsupported_chars(cls, s: str) -> str:
|
|
||||||
"""Replaces characters with those available in the Helvetica typeface."""
|
|
||||||
return s.translate(cls.ligatures)
|
|
||||||
|
|
||||||
def to_pdf(
|
|
||||||
self,
|
|
||||||
*,
|
|
||||||
out_filename: Path,
|
|
||||||
image_filename: Path | None = None,
|
|
||||||
show_bounding_boxes: bool = False,
|
|
||||||
fontname: str = "Helvetica",
|
|
||||||
invisible_text: bool = False,
|
|
||||||
interword_spaces: bool = False,
|
|
||||||
) -> None:
|
|
||||||
"""Creates a PDF file with an image superimposed on top of the text.
|
|
||||||
|
|
||||||
Text is positioned according to the bounding box of the lines in
|
|
||||||
the hOCR file.
|
|
||||||
The image need not be identical to the image used to create the hOCR
|
|
||||||
file.
|
|
||||||
It can have a lower resolution, different color mode, etc.
|
|
||||||
|
|
||||||
Arguments:
|
|
||||||
out_filename: Path of PDF to write.
|
|
||||||
image_filename: Image to use for this file. If omitted, the OCR text
|
|
||||||
is shown.
|
|
||||||
show_bounding_boxes: Show bounding boxes around various text regions,
|
|
||||||
for debugging.
|
|
||||||
fontname: Name of font to use.
|
|
||||||
invisible_text: If True, text is rendered invisible so that is
|
|
||||||
selectable but never drawn. If False, text is visible and may
|
|
||||||
be seen if the image is skipped or deleted in Acrobat.
|
|
||||||
interword_spaces: If True, insert spaces between words rather than
|
|
||||||
drawing each word without spaces. Generally this improves text
|
|
||||||
extraction.
|
|
||||||
"""
|
|
||||||
# create the PDF file
|
|
||||||
# page size in points (1/72 in.)
|
|
||||||
pdf = Canvas(
|
|
||||||
os.fspath(out_filename),
|
|
||||||
pagesize=(self.width, self.height),
|
|
||||||
pageCompression=1,
|
|
||||||
)
|
|
||||||
|
|
||||||
# draw bounding box for each paragraph
|
|
||||||
# light blue for bounding box of paragraph
|
|
||||||
pdf.setStrokeColor(cyan)
|
|
||||||
# light blue for bounding box of paragraph
|
|
||||||
pdf.setFillColor(cyan)
|
|
||||||
pdf.setLineWidth(0) # no line for bounding box
|
|
||||||
for elem in self.hocr.iterfind(self._child_xpath('p', 'ocr_par')):
|
|
||||||
elemtxt = self._get_element_text(elem).rstrip()
|
|
||||||
if len(elemtxt) == 0:
|
|
||||||
continue
|
|
||||||
|
|
||||||
pxl_coords = self.element_coordinates(elem)
|
|
||||||
pt = self.pt_from_pixel(pxl_coords) # pylint: disable=invalid-name
|
|
||||||
|
|
||||||
# draw the bbox border
|
|
||||||
if show_bounding_boxes: # pragma: no cover
|
|
||||||
pdf.rect(
|
|
||||||
pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, fill=1
|
|
||||||
)
|
|
||||||
|
|
||||||
found_lines = False
|
|
||||||
for line in (
|
|
||||||
element
|
|
||||||
for element in self.hocr.iterfind(self._child_xpath('span'))
|
|
||||||
if 'class' in element.attrib
|
|
||||||
and element.attrib['class'] in {'ocr_header', 'ocr_line', 'ocr_textfloat'}
|
|
||||||
):
|
|
||||||
found_lines = True
|
|
||||||
self._do_line(
|
|
||||||
pdf,
|
|
||||||
line,
|
|
||||||
"ocrx_word",
|
|
||||||
fontname,
|
|
||||||
invisible_text,
|
|
||||||
interword_spaces,
|
|
||||||
show_bounding_boxes,
|
|
||||||
)
|
|
||||||
|
|
||||||
if not found_lines:
|
|
||||||
# Tesseract did not report any lines (just words)
|
|
||||||
root = self.hocr.find(self._child_xpath('div', 'ocr_page'))
|
|
||||||
self._do_line(
|
|
||||||
pdf,
|
|
||||||
root,
|
|
||||||
"ocrx_word",
|
|
||||||
fontname,
|
|
||||||
invisible_text,
|
|
||||||
interword_spaces,
|
|
||||||
show_bounding_boxes,
|
|
||||||
)
|
|
||||||
# put the image on the page, scaled to fill the page
|
|
||||||
if image_filename is not None:
|
|
||||||
pdf.drawImage(
|
|
||||||
os.fspath(image_filename), 0, 0, width=self.width, height=self.height
|
|
||||||
)
|
|
||||||
|
|
||||||
# finish up the page and save it
|
|
||||||
pdf.showPage()
|
|
||||||
pdf.save()
|
|
||||||
|
|
||||||
@classmethod
|
|
||||||
def polyval(cls, poly, x): # pragma: no cover
|
|
||||||
"""Calculate the value of a polynomial at a point."""
|
|
||||||
return x * poly[0] + poly[1]
|
|
||||||
|
|
||||||
def _do_line(
|
|
||||||
self,
|
|
||||||
pdf: Canvas,
|
|
||||||
line: Element | None,
|
|
||||||
elemclass: str,
|
|
||||||
fontname: str,
|
|
||||||
invisible_text: bool,
|
|
||||||
interword_spaces: bool,
|
|
||||||
show_bounding_boxes: bool,
|
|
||||||
):
|
|
||||||
if line is None:
|
|
||||||
return
|
|
||||||
pxl_line_coords = self.element_coordinates(line)
|
|
||||||
line_box = self.pt_from_pixel(pxl_line_coords)
|
|
||||||
line_height = line_box.y2 - line_box.y1
|
|
||||||
|
|
||||||
slope, pxl_intercept = self.baseline(line)
|
|
||||||
if abs(slope) < 0.005:
|
|
||||||
slope = 0.0
|
|
||||||
angle = atan(slope)
|
|
||||||
cos_a, sin_a = cos(angle), sin(angle)
|
|
||||||
|
|
||||||
text = pdf.beginText()
|
|
||||||
intercept = pxl_intercept / self.dpi * inch
|
|
||||||
|
|
||||||
# Don't allow the font to break out of the bounding box. Division by
|
|
||||||
# cos_a accounts for extra clearance between the glyph's vertical axis
|
|
||||||
# on a sloped baseline and the edge of the bounding box.
|
|
||||||
fontsize = (line_height - abs(intercept)) / cos_a
|
|
||||||
text.setFont(fontname, fontsize)
|
|
||||||
if invisible_text:
|
|
||||||
text.setTextRenderMode(3) # Invisible (indicates OCR text)
|
|
||||||
|
|
||||||
# Intercept is normally negative, so this places it above the bottom
|
|
||||||
# of the line box
|
|
||||||
baseline_y2 = self.height - (line_box.y2 + intercept)
|
|
||||||
|
|
||||||
if show_bounding_boxes: # pragma: no cover
|
|
||||||
# draw the baseline in magenta, dashed
|
|
||||||
pdf.setDash()
|
|
||||||
pdf.setStrokeColor(magenta)
|
|
||||||
pdf.setLineWidth(0.5)
|
|
||||||
# negate slope because it is defined as a rise/run in pixel
|
|
||||||
# coordinates and page coordinates have the y axis flipped
|
|
||||||
pdf.line(
|
|
||||||
line_box.x1,
|
|
||||||
baseline_y2,
|
|
||||||
line_box.x2,
|
|
||||||
self.polyval((-slope, baseline_y2), line_box.x2 - line_box.x1),
|
|
||||||
)
|
|
||||||
# light green for bounding box of word/line
|
|
||||||
pdf.setDash(6, 3)
|
|
||||||
pdf.setStrokeColor(red)
|
|
||||||
|
|
||||||
text.setTextTransform(cos_a, -sin_a, sin_a, cos_a, line_box.x1, baseline_y2)
|
|
||||||
pdf.setFillColor(black) # text in black
|
|
||||||
|
|
||||||
elements = line.findall(self._child_xpath('span', elemclass))
|
|
||||||
for elem in elements:
|
|
||||||
elemtxt = self._get_element_text(elem).strip()
|
|
||||||
elemtxt = self.replace_unsupported_chars(elemtxt)
|
|
||||||
if elemtxt == '':
|
|
||||||
continue
|
|
||||||
|
|
||||||
pxl_coords = self.element_coordinates(elem)
|
|
||||||
box = self.pt_from_pixel(pxl_coords)
|
|
||||||
if interword_spaces:
|
|
||||||
# if `--interword-spaces` is true, append a space
|
|
||||||
# to the end of each text element to allow simpler PDF viewers
|
|
||||||
# such as PDF.js to better recognize words in search and copy
|
|
||||||
# and paste. Do not remove space from last word in line, even
|
|
||||||
# though it would look better, because it will interfere with
|
|
||||||
# naive text extraction. \n does not work either.
|
|
||||||
elemtxt += ' '
|
|
||||||
box = Rect._make(
|
|
||||||
(
|
|
||||||
box.x1,
|
|
||||||
line_box.y1,
|
|
||||||
box.x2 + pdf.stringWidth(' ', fontname, line_height),
|
|
||||||
line_box.y2,
|
|
||||||
)
|
|
||||||
)
|
|
||||||
box_width = box.x2 - box.x1
|
|
||||||
font_width = pdf.stringWidth(elemtxt, fontname, fontsize)
|
|
||||||
|
|
||||||
# draw the bbox border
|
|
||||||
if show_bounding_boxes: # pragma: no cover
|
|
||||||
pdf.rect(
|
|
||||||
box.x1, self.height - line_box.y2, box_width, line_height, fill=0
|
|
||||||
)
|
|
||||||
|
|
||||||
# Adjust relative position of cursor
|
|
||||||
# This is equivalent to:
|
|
||||||
# text.setTextOrigin(pt.x1, self.height - line_box.y2)
|
|
||||||
# but the former generates a full text reposition matrix (Tm) in the
|
|
||||||
# content stream while this issues a "offset" (Td) command.
|
|
||||||
# .moveCursor() is relative to start of the text line, where the
|
|
||||||
# "text line" means whatever reportlab defines it as. Do not use
|
|
||||||
# use .getCursor(), since moveCursor() rather unintuitively plans
|
|
||||||
# its moves relative to .getStartOfLine().
|
|
||||||
# For skewed lines, in the text transform we set up a rotated
|
|
||||||
# coordinate system, so we don't have to account for the
|
|
||||||
# incremental offset. Surprisingly most PDF viewers can handle this.
|
|
||||||
cursor = text.getStartOfLine()
|
|
||||||
dx = box.x1 - cursor[0]
|
|
||||||
dy = baseline_y2 - cursor[1]
|
|
||||||
text.moveCursor(dx, dy)
|
|
||||||
|
|
||||||
# If reportlab tells us this word is 0 units wide, our best seems
|
|
||||||
# to be to suppress this text
|
|
||||||
if font_width > 0:
|
|
||||||
text.setHorizScale(100 * box_width / font_width)
|
|
||||||
text.textOut(elemtxt)
|
|
||||||
pdf.drawText(text)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
parser = argparse.ArgumentParser(description='Convert hocr file to PDF')
|
|
||||||
parser.add_argument(
|
|
||||||
'-b',
|
|
||||||
'--boundingboxes',
|
|
||||||
action="store_true",
|
|
||||||
default=False,
|
|
||||||
help='Show bounding boxes borders',
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'-r',
|
|
||||||
'--resolution',
|
|
||||||
type=int,
|
|
||||||
default=300,
|
|
||||||
help='Resolution of the image that was OCRed',
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'-i',
|
|
||||||
'--image',
|
|
||||||
default=None,
|
|
||||||
help='Path to the image to be placed above the text',
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'--interword-spaces',
|
|
||||||
action='store_true',
|
|
||||||
default=False,
|
|
||||||
help='Add spaces between words',
|
|
||||||
)
|
|
||||||
parser.add_argument('hocrfile', help='Path to the hocr file to be parsed')
|
|
||||||
parser.add_argument('outputfile', help='Path to the PDF file to be generated')
|
|
||||||
args = parser.parse_args()
|
|
||||||
|
|
||||||
hocr = HocrTransform(hocr_filename=args.hocrfile, dpi=args.resolution)
|
|
||||||
hocr.to_pdf(
|
|
||||||
out_filename=args.outputfile,
|
|
||||||
image_filename=args.image,
|
|
||||||
show_bounding_boxes=args.boundingboxes,
|
|
||||||
interword_spaces=args.interword_spaces,
|
|
||||||
)
|
|
||||||
Executable
+18
@@ -0,0 +1,18 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
"""Transform .hocr and page image to text PDF."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from ocrmypdf.hocrtransform._hocr import (
|
||||||
|
DebugRenderOptions,
|
||||||
|
HocrTransform,
|
||||||
|
HocrTransformError,
|
||||||
|
)
|
||||||
|
|
||||||
|
__all__ = (
|
||||||
|
'HocrTransform',
|
||||||
|
'HocrTransformError',
|
||||||
|
'DebugRenderOptions',
|
||||||
|
)
|
||||||
@@ -0,0 +1,40 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
"""Simple CLI for testing HOCR."""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
|
||||||
|
from ocrmypdf.hocrtransform import HocrTransform
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
parser = argparse.ArgumentParser(description='Convert hocr file to PDF')
|
||||||
|
parser.add_argument(
|
||||||
|
'-b',
|
||||||
|
'--boundingboxes',
|
||||||
|
action="store_true",
|
||||||
|
default=False,
|
||||||
|
help='Show bounding boxes borders',
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
'-r',
|
||||||
|
'--resolution',
|
||||||
|
type=int,
|
||||||
|
default=300,
|
||||||
|
help='Resolution of the image that was OCRed',
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
'-i',
|
||||||
|
'--image',
|
||||||
|
default=None,
|
||||||
|
help='Path to the image to be placed above the text',
|
||||||
|
)
|
||||||
|
parser.add_argument('hocrfile', help='Path to the hocr file to be parsed')
|
||||||
|
parser.add_argument('outputfile', help='Path to the PDF file to be generated')
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
hocr = HocrTransform(hocr_filename=args.hocrfile, dpi=args.resolution)
|
||||||
|
hocr.to_pdf(
|
||||||
|
out_filename=args.outputfile,
|
||||||
|
image_filename=args.image,
|
||||||
|
)
|
||||||
@@ -0,0 +1,141 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import unicodedata
|
||||||
|
import zlib
|
||||||
|
from importlib.resources import files as package_files
|
||||||
|
|
||||||
|
from pikepdf import (
|
||||||
|
Dictionary,
|
||||||
|
Name,
|
||||||
|
Pdf,
|
||||||
|
)
|
||||||
|
from pikepdf.canvas import Font
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class EncodableFont(Font):
|
||||||
|
def text_encode(self, text: str) -> bytes:
|
||||||
|
raise NotImplementedError()
|
||||||
|
|
||||||
|
|
||||||
|
class GlyphlessFont(EncodableFont):
|
||||||
|
CID_TO_GID_DATA = zlib.compress(b"\x00\x01" * 65536)
|
||||||
|
GLYPHLESS_FONT_NAME = 'pdf.ttf'
|
||||||
|
GLYPHLESS_FONT = (package_files('ocrmypdf.data') / GLYPHLESS_FONT_NAME).read_bytes()
|
||||||
|
CHAR_ASPECT = 2
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
def text_width(self, text: str, fontsize: float) -> float:
|
||||||
|
"""Estimate the width of a text string when rendered with the given font."""
|
||||||
|
# NFKC: split ligatures, combine diacritics
|
||||||
|
return len(unicodedata.normalize("NFKC", text)) * (fontsize / self.CHAR_ASPECT)
|
||||||
|
|
||||||
|
def text_encode(self, text: str) -> bytes:
|
||||||
|
return text.encode('utf-16be')
|
||||||
|
|
||||||
|
def register(self, pdf: Pdf):
|
||||||
|
"""Register the glyphless font.
|
||||||
|
|
||||||
|
Create several data structures in the Pdf to describe the font. While it create
|
||||||
|
the data, a reference should be set in at least one page's /Resources dictionary
|
||||||
|
to retain the font in the output PDF and ensure it is usable on that page.
|
||||||
|
"""
|
||||||
|
PLACEHOLDER = Name.Placeholder
|
||||||
|
|
||||||
|
basefont = pdf.make_indirect(
|
||||||
|
Dictionary(
|
||||||
|
BaseFont=Name.GlyphLessFont,
|
||||||
|
DescendantFonts=[PLACEHOLDER],
|
||||||
|
Encoding=Name("/Identity-H"),
|
||||||
|
Subtype=Name.Type0,
|
||||||
|
ToUnicode=PLACEHOLDER,
|
||||||
|
Type=Name.Font,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
cid_font_type2 = pdf.make_indirect(
|
||||||
|
Dictionary(
|
||||||
|
BaseFont=Name.GlyphLessFont,
|
||||||
|
CIDToGIDMap=PLACEHOLDER,
|
||||||
|
CIDSystemInfo=Dictionary(
|
||||||
|
Ordering="Identity",
|
||||||
|
Registry="Adobe",
|
||||||
|
Supplement=0,
|
||||||
|
),
|
||||||
|
FontDescriptor=PLACEHOLDER,
|
||||||
|
Subtype=Name.CIDFontType2,
|
||||||
|
Type=Name.Font,
|
||||||
|
DW=1000 // self.CHAR_ASPECT,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
basefont.DescendantFonts = [cid_font_type2]
|
||||||
|
cid_font_type2.CIDToGIDMap = pdf.make_stream(
|
||||||
|
self.CID_TO_GID_DATA, Filter=Name.FlateDecode
|
||||||
|
)
|
||||||
|
basefont.ToUnicode = pdf.make_stream(
|
||||||
|
b"/CIDInit /ProcSet findresource begin\n"
|
||||||
|
b"12 dict begin\n"
|
||||||
|
b"begincmap\n"
|
||||||
|
b"/CIDSystemInfo\n"
|
||||||
|
b"<<\n"
|
||||||
|
b" /Registry (Adobe)\n"
|
||||||
|
b" /Ordering (UCS)\n"
|
||||||
|
b" /Supplement 0\n"
|
||||||
|
b">> def\n"
|
||||||
|
b"/CMapName /Adobe-Identify-UCS def\n"
|
||||||
|
b"/CMapType 2 def\n"
|
||||||
|
b"1 begincodespacerange\n"
|
||||||
|
b"<0000> <FFFF>\n"
|
||||||
|
b"endcodespacerange\n"
|
||||||
|
b"1 beginbfrange\n"
|
||||||
|
b"<0000> <FFFF> <0000>\n"
|
||||||
|
b"endbfrange\n"
|
||||||
|
b"endcmap\n"
|
||||||
|
b"CMapName currentdict /CMap defineresource pop\n"
|
||||||
|
b"end\n"
|
||||||
|
b"end\n"
|
||||||
|
)
|
||||||
|
font_descriptor = pdf.make_indirect(
|
||||||
|
Dictionary(
|
||||||
|
Ascent=1000,
|
||||||
|
CapHeight=1000,
|
||||||
|
Descent=-1,
|
||||||
|
Flags=5, # Fixed pitch and symbolic
|
||||||
|
FontBBox=[0, 0, 1000 // self.CHAR_ASPECT, 1000],
|
||||||
|
FontFile2=PLACEHOLDER,
|
||||||
|
FontName=Name.GlyphLessFont,
|
||||||
|
ItalicAngle=0,
|
||||||
|
StemV=80,
|
||||||
|
Type=Name.FontDescriptor,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
font_descriptor.FontFile2 = pdf.make_stream(self.GLYPHLESS_FONT)
|
||||||
|
cid_font_type2.FontDescriptor = font_descriptor
|
||||||
|
return basefont
|
||||||
|
|
||||||
|
|
||||||
|
class Courier(EncodableFont):
|
||||||
|
"""Courier font."""
|
||||||
|
|
||||||
|
def text_width(self, text: str, fontsize: float) -> float:
|
||||||
|
"""Estimate the width of a text string when rendered with the given font."""
|
||||||
|
return len(text) * fontsize
|
||||||
|
|
||||||
|
def text_encode(self, text: str) -> bytes:
|
||||||
|
return text.encode('pdfdoc', errors='ignore')
|
||||||
|
|
||||||
|
def register(self, pdf: Pdf) -> Dictionary:
|
||||||
|
"""Register the font."""
|
||||||
|
return pdf.make_indirect(
|
||||||
|
Dictionary(
|
||||||
|
BaseFont=Name.Courier,
|
||||||
|
Type=Name.Font,
|
||||||
|
Subtype=Name.Type1,
|
||||||
|
)
|
||||||
|
)
|
||||||
@@ -0,0 +1,501 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2010 Jonathan Brinley
|
||||||
|
# SPDX-FileCopyrightText: 2013-2014 Julien Pfefferkorn
|
||||||
|
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
"""hOCR transform implementation."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import unicodedata
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from itertools import pairwise
|
||||||
|
from math import atan, cos, pi
|
||||||
|
from pathlib import Path
|
||||||
|
from xml.etree import ElementTree
|
||||||
|
|
||||||
|
from pikepdf import Matrix, Name, Rectangle
|
||||||
|
from pikepdf.canvas import (
|
||||||
|
BLACK,
|
||||||
|
BLUE,
|
||||||
|
CYAN,
|
||||||
|
DARKGREEN,
|
||||||
|
GREEN,
|
||||||
|
MAGENTA,
|
||||||
|
RED,
|
||||||
|
Canvas,
|
||||||
|
Text,
|
||||||
|
TextDirection,
|
||||||
|
)
|
||||||
|
|
||||||
|
from ocrmypdf.hocrtransform._font import EncodableFont as Font
|
||||||
|
from ocrmypdf.hocrtransform._font import GlyphlessFont
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
INCH = 72.0
|
||||||
|
|
||||||
|
Element = ElementTree.Element
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass
|
||||||
|
class DebugRenderOptions:
|
||||||
|
"""A class for managing rendering options."""
|
||||||
|
|
||||||
|
render_paragraph_bbox: bool = False
|
||||||
|
render_baseline: bool = False
|
||||||
|
render_triangle: bool = False
|
||||||
|
render_line_bbox: bool = False
|
||||||
|
render_word_bbox: bool = False
|
||||||
|
render_space_bbox: bool = False
|
||||||
|
|
||||||
|
|
||||||
|
class HocrTransformError(Exception):
|
||||||
|
"""Error while applying hOCR transform."""
|
||||||
|
|
||||||
|
|
||||||
|
class HocrTransform:
|
||||||
|
"""A class for converting documents from the hOCR format.
|
||||||
|
|
||||||
|
For details of the hOCR format, see:
|
||||||
|
http://kba.github.io/hocr-spec/1.2/.
|
||||||
|
"""
|
||||||
|
|
||||||
|
box_pattern = re.compile(
|
||||||
|
r'''
|
||||||
|
bbox \s+
|
||||||
|
(\d+) \s+ # left: uint
|
||||||
|
(\d+) \s+ # top: uint
|
||||||
|
(\d+) \s+ # right: uint
|
||||||
|
(\d+) # bottom: uint
|
||||||
|
''',
|
||||||
|
re.VERBOSE,
|
||||||
|
)
|
||||||
|
baseline_pattern = re.compile(
|
||||||
|
r'''
|
||||||
|
baseline \s+
|
||||||
|
([\-\+]?\d*\.?\d*) \s+ # +/- decimal float
|
||||||
|
([\-\+]?\d+) # +/- int
|
||||||
|
''',
|
||||||
|
re.VERBOSE,
|
||||||
|
)
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
hocr_filename: str | Path,
|
||||||
|
dpi: float,
|
||||||
|
debug: bool = False,
|
||||||
|
fontname: Name = Name("/f-0-0"),
|
||||||
|
font: Font = GlyphlessFont(),
|
||||||
|
debug_render_options: DebugRenderOptions | None = None,
|
||||||
|
):
|
||||||
|
"""Initialize the HocrTransform object."""
|
||||||
|
if debug:
|
||||||
|
log.warning("Use debug_render_options instead", DeprecationWarning)
|
||||||
|
self.render_options = DebugRenderOptions(
|
||||||
|
render_baseline=debug,
|
||||||
|
render_triangle=debug,
|
||||||
|
render_line_bbox=False,
|
||||||
|
render_word_bbox=debug,
|
||||||
|
render_paragraph_bbox=False,
|
||||||
|
render_space_bbox=False,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
self.render_options = debug_render_options or DebugRenderOptions()
|
||||||
|
self.dpi = dpi
|
||||||
|
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
||||||
|
self._fontname = fontname
|
||||||
|
self._font = font
|
||||||
|
|
||||||
|
# if the hOCR file has a namespace, ElementTree requires its use to
|
||||||
|
# find elements
|
||||||
|
matches = re.match(r'({.*})html', self.hocr.getroot().tag)
|
||||||
|
self.xmlns = ''
|
||||||
|
if matches:
|
||||||
|
self.xmlns = matches.group(1)
|
||||||
|
|
||||||
|
for div in self.hocr.findall(self._child_xpath('div', 'ocr_page')):
|
||||||
|
coords = self.element_coordinates(div)
|
||||||
|
if not coords:
|
||||||
|
raise HocrTransformError("hocr file is missing page dimensions")
|
||||||
|
self.width = (coords.urx - coords.llx) / (self.dpi / INCH)
|
||||||
|
self.height = (coords.ury - coords.lly) / (self.dpi / INCH)
|
||||||
|
# Stop after first div that has page coordinates
|
||||||
|
break
|
||||||
|
|
||||||
|
def _get_element_text(self, element: Element) -> str:
|
||||||
|
"""Return the textual content of the element and its children."""
|
||||||
|
text = element.text if element.text is not None else ''
|
||||||
|
for child in element:
|
||||||
|
text += self._get_element_text(child)
|
||||||
|
text += element.tail if element.tail is not None else ''
|
||||||
|
return text
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def element_coordinates(cls, element: Element) -> Rectangle | None:
|
||||||
|
"""Get coordinates of the bounding box around an element."""
|
||||||
|
matches = cls.box_pattern.search(element.attrib.get('title', ''))
|
||||||
|
if not matches:
|
||||||
|
return None
|
||||||
|
return Rectangle(
|
||||||
|
float(matches.group(1)), # llx = left
|
||||||
|
float(matches.group(2)), # lly = top
|
||||||
|
float(matches.group(3)), # urx = right
|
||||||
|
float(matches.group(4)), # ury = bottom
|
||||||
|
)
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def baseline(cls, element: Element) -> tuple[float, float]:
|
||||||
|
"""Get baseline's slope and intercept."""
|
||||||
|
matches = cls.baseline_pattern.search(element.attrib.get('title', ''))
|
||||||
|
if not matches:
|
||||||
|
return (0.0, 0.0)
|
||||||
|
return float(matches.group(1)), int(matches.group(2))
|
||||||
|
|
||||||
|
def _child_xpath(self, html_tag: str, html_class: str | None = None) -> str:
|
||||||
|
xpath = f".//{self.xmlns}{html_tag}"
|
||||||
|
if html_class:
|
||||||
|
xpath += f"[@class='{html_class}']"
|
||||||
|
return xpath
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def normalize_text(cls, s: str) -> str:
|
||||||
|
"""Normalize the given text using the NFKC normalization form."""
|
||||||
|
return unicodedata.normalize("NFKC", s)
|
||||||
|
|
||||||
|
def to_pdf(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
out_filename: Path,
|
||||||
|
image_filename: Path | None = None,
|
||||||
|
invisible_text: bool = True,
|
||||||
|
) -> None:
|
||||||
|
"""Creates a PDF file with an image superimposed on top of the text.
|
||||||
|
|
||||||
|
Text is positioned according to the bounding box of the lines in
|
||||||
|
the hOCR file.
|
||||||
|
The image need not be identical to the image used to create the hOCR
|
||||||
|
file.
|
||||||
|
It can have a lower resolution, different color mode, etc.
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
out_filename: Path of PDF to write.
|
||||||
|
image_filename: Image to use for this file. If omitted, the OCR text
|
||||||
|
is shown.
|
||||||
|
invisible_text: If True, text is rendered invisible so that is
|
||||||
|
selectable but never drawn. If False, text is visible and may
|
||||||
|
be seen if the image is skipped or deleted in Acrobat.
|
||||||
|
"""
|
||||||
|
# create the PDF file
|
||||||
|
# page size in points (1/72 in.)
|
||||||
|
canvas = Canvas(page_size=(self.width, self.height))
|
||||||
|
canvas.add_font(self._fontname, self._font)
|
||||||
|
page_matrix = (
|
||||||
|
Matrix()
|
||||||
|
.translated(0, self.height)
|
||||||
|
.scaled(1, -1)
|
||||||
|
.scaled(INCH / self.dpi, INCH / self.dpi)
|
||||||
|
)
|
||||||
|
log.debug(page_matrix)
|
||||||
|
with canvas.do.save_state(cm=page_matrix):
|
||||||
|
self._debug_draw_paragraph_boxes(canvas)
|
||||||
|
found_lines = False
|
||||||
|
for par in self.hocr.iterfind(self._child_xpath('p', 'ocr_par')):
|
||||||
|
for line in (
|
||||||
|
element
|
||||||
|
for element in par.iterfind(self._child_xpath('span'))
|
||||||
|
if 'class' in element.attrib
|
||||||
|
and element.attrib['class']
|
||||||
|
in {'ocr_header', 'ocr_line', 'ocr_textfloat'}
|
||||||
|
):
|
||||||
|
found_lines = True
|
||||||
|
direction = self._get_text_direction(par)
|
||||||
|
inject_word_breaks = self._get_inject_word_breaks(par)
|
||||||
|
self._do_line(
|
||||||
|
canvas,
|
||||||
|
line,
|
||||||
|
"ocrx_word",
|
||||||
|
invisible_text,
|
||||||
|
direction,
|
||||||
|
inject_word_breaks,
|
||||||
|
)
|
||||||
|
|
||||||
|
if not found_lines:
|
||||||
|
# Tesseract did not report any lines (just words)
|
||||||
|
root = self.hocr.find(self._child_xpath('div', 'ocr_page'))
|
||||||
|
direction = self._get_text_direction(root)
|
||||||
|
self._do_line(
|
||||||
|
canvas,
|
||||||
|
root,
|
||||||
|
"ocrx_word",
|
||||||
|
invisible_text,
|
||||||
|
direction,
|
||||||
|
True,
|
||||||
|
)
|
||||||
|
# put the image on the page, scaled to fill the page
|
||||||
|
if image_filename is not None:
|
||||||
|
canvas.do.draw_image(
|
||||||
|
image_filename, 0, 0, width=self.width, height=self.height
|
||||||
|
)
|
||||||
|
|
||||||
|
# finish up the page and save it
|
||||||
|
canvas.to_pdf().save(out_filename)
|
||||||
|
|
||||||
|
def _get_text_direction(self, par):
|
||||||
|
"""Get the text direction of the paragraph.
|
||||||
|
|
||||||
|
Arabic, Hebrew, Persian, are right-to-left languages.
|
||||||
|
"""
|
||||||
|
return (
|
||||||
|
TextDirection.RTL
|
||||||
|
if par.attrib.get('dir', 'ltr') == 'rtl'
|
||||||
|
else TextDirection.LTR
|
||||||
|
)
|
||||||
|
|
||||||
|
def _get_inject_word_breaks(self, par):
|
||||||
|
"""Determine whether word breaks should be injected.
|
||||||
|
|
||||||
|
In Chinese, Japanese, and Korean, word breaks are not injected, because
|
||||||
|
words are usually one or two characters and separators are usually explicit.
|
||||||
|
In all other languages, we inject word breaks to help word segmentation.
|
||||||
|
"""
|
||||||
|
lang = par.attrib.get('lang', '')
|
||||||
|
log.debug(lang)
|
||||||
|
if lang in {'chi_sim', 'chi_tra', 'jpn', 'kor'}:
|
||||||
|
return False
|
||||||
|
return True
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def polyval(cls, poly, x): # pragma: no cover
|
||||||
|
"""Calculate the value of a polynomial at a point."""
|
||||||
|
return x * poly[0] + poly[1]
|
||||||
|
|
||||||
|
def _do_line(
|
||||||
|
self,
|
||||||
|
canvas: Canvas,
|
||||||
|
line: Element | None,
|
||||||
|
elemclass: str,
|
||||||
|
invisible_text: bool,
|
||||||
|
text_direction: TextDirection,
|
||||||
|
inject_word_breaks: bool,
|
||||||
|
):
|
||||||
|
"""Render the text for a given line.
|
||||||
|
|
||||||
|
The canvas's coordinate system must be configured so that hOCR pixel
|
||||||
|
coordinates are mapped to PDF coordinates.
|
||||||
|
"""
|
||||||
|
if line is None:
|
||||||
|
return
|
||||||
|
line_box = self.element_coordinates(line)
|
||||||
|
if not line_box:
|
||||||
|
return
|
||||||
|
if line_box.ury <= line_box.lly:
|
||||||
|
log.error(
|
||||||
|
"line box is invalid so we cannot render it: box=%s text=%s",
|
||||||
|
line_box,
|
||||||
|
self._get_element_text(line),
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
|
self._debug_draw_line_bbox(canvas, line_box)
|
||||||
|
|
||||||
|
# Baseline is a polynomial (usually straight line) that describes the
|
||||||
|
# text baseline relative to the bottom left corner of the line bounding
|
||||||
|
# box.
|
||||||
|
bottom_left_corner = line_box.llx, line_box.ury
|
||||||
|
slope, intercept = self.baseline(line)
|
||||||
|
if abs(slope) < 0.005:
|
||||||
|
slope = 0.0
|
||||||
|
angle = atan(slope)
|
||||||
|
|
||||||
|
# Setup a new coordinate system on the line box's intercept and rotated by
|
||||||
|
# its slope.
|
||||||
|
line_matrix = (
|
||||||
|
Matrix()
|
||||||
|
.translated(*bottom_left_corner)
|
||||||
|
.translated(0, intercept)
|
||||||
|
.rotated(angle / pi * 180)
|
||||||
|
)
|
||||||
|
log.debug(line_matrix)
|
||||||
|
with canvas.do.save_state(cm=line_matrix):
|
||||||
|
text = Text(direction=text_direction)
|
||||||
|
|
||||||
|
# Don't allow the font to break out of the bounding box. Division by
|
||||||
|
# cos_a accounts for extra clearance between the glyph's vertical axis
|
||||||
|
# on a sloped baseline and the edge of the bounding box.
|
||||||
|
line_box_height = abs(line_box.height) / cos(angle)
|
||||||
|
fontsize = line_box_height + intercept
|
||||||
|
text.font(self._fontname, fontsize)
|
||||||
|
text.render_mode(3 if invisible_text else 0)
|
||||||
|
|
||||||
|
self._debug_draw_baseline(
|
||||||
|
canvas, line_matrix.inverse().transform(line_box), 0
|
||||||
|
)
|
||||||
|
|
||||||
|
canvas.do.fill_color(BLACK) # text in black
|
||||||
|
elements = line.findall(self._child_xpath('span', elemclass))
|
||||||
|
for elem, next_elem in pairwise(elements + [None]):
|
||||||
|
self._do_line_word(
|
||||||
|
canvas,
|
||||||
|
line_matrix,
|
||||||
|
text,
|
||||||
|
fontsize,
|
||||||
|
elem,
|
||||||
|
next_elem,
|
||||||
|
text_direction,
|
||||||
|
inject_word_breaks,
|
||||||
|
)
|
||||||
|
canvas.do.draw_text(text)
|
||||||
|
|
||||||
|
def _do_line_word(
|
||||||
|
self,
|
||||||
|
canvas: Canvas,
|
||||||
|
line_matrix: Matrix,
|
||||||
|
text: Text,
|
||||||
|
fontsize: float,
|
||||||
|
elem: Element | None,
|
||||||
|
next_elem: Element | None,
|
||||||
|
text_direction: TextDirection,
|
||||||
|
inject_word_breaks: bool,
|
||||||
|
):
|
||||||
|
"""Render the text for a single word."""
|
||||||
|
if elem is None:
|
||||||
|
return
|
||||||
|
elemtxt = self.normalize_text(self._get_element_text(elem).strip())
|
||||||
|
if elemtxt == '':
|
||||||
|
return
|
||||||
|
|
||||||
|
hocr_box = self.element_coordinates(elem)
|
||||||
|
if hocr_box is None:
|
||||||
|
return
|
||||||
|
box = line_matrix.inverse().transform(hocr_box)
|
||||||
|
font_width = self._font.text_width(elemtxt, fontsize)
|
||||||
|
|
||||||
|
# Debug sketches
|
||||||
|
self._debug_draw_word_triangle(canvas, box)
|
||||||
|
self._debug_draw_word_bbox(canvas, box)
|
||||||
|
|
||||||
|
# If this word is 0 units wide, our best bet seems to be to suppress this text
|
||||||
|
if text_direction == TextDirection.RTL:
|
||||||
|
log.info("RTL: %s", elemtxt)
|
||||||
|
if font_width > 0:
|
||||||
|
if text_direction == TextDirection.LTR:
|
||||||
|
text.text_transform(Matrix(1, 0, 0, -1, box.llx, 0))
|
||||||
|
elif text_direction == TextDirection.RTL:
|
||||||
|
text.text_transform(Matrix(-1, 0, 0, -1, box.llx + box.width, 0))
|
||||||
|
text.horiz_scale(100 * box.width / font_width)
|
||||||
|
text.show(self._font.text_encode(elemtxt))
|
||||||
|
|
||||||
|
# Get coordinates of the next word (if there is one)
|
||||||
|
hocr_next_box = (
|
||||||
|
self.element_coordinates(next_elem) if next_elem is not None else None
|
||||||
|
)
|
||||||
|
if hocr_next_box is None:
|
||||||
|
return
|
||||||
|
# Render a space this word and the next word. The explicit space helps
|
||||||
|
# PDF viewers identify the word break, and horizontally scaling it to
|
||||||
|
# occupy the space the between the words helps the PDF viewer
|
||||||
|
# avoid combiningthewordstogether.
|
||||||
|
if not inject_word_breaks:
|
||||||
|
return
|
||||||
|
next_box = line_matrix.inverse().transform(hocr_next_box)
|
||||||
|
if text_direction == TextDirection.LTR:
|
||||||
|
space_box = Rectangle(box.urx, box.lly, next_box.llx, next_box.ury)
|
||||||
|
elif text_direction == TextDirection.RTL:
|
||||||
|
space_box = Rectangle(next_box.urx, box.lly, box.llx, next_box.ury)
|
||||||
|
self._debug_draw_space_bbox(canvas, space_box)
|
||||||
|
space_width = self._font.text_width(' ', fontsize)
|
||||||
|
if space_width > 0:
|
||||||
|
if text_direction == TextDirection.LTR:
|
||||||
|
text.text_transform(Matrix(1, 0, 0, -1, space_box.llx, 0))
|
||||||
|
elif text_direction == TextDirection.RTL:
|
||||||
|
text.text_transform(
|
||||||
|
Matrix(-1, 0, 0, -1, space_box.llx + space_box.width, 0)
|
||||||
|
)
|
||||||
|
text.horiz_scale(100 * space_box.width / space_width)
|
||||||
|
text.show(self._font.text_encode(' '))
|
||||||
|
|
||||||
|
def _debug_draw_paragraph_boxes(self, canvas: Canvas, color=CYAN):
|
||||||
|
"""Draw boxes around paragraphs in the document."""
|
||||||
|
if not self.render_options.render_paragraph_bbox: # pragma: no cover
|
||||||
|
return
|
||||||
|
with canvas.do.save_state():
|
||||||
|
# draw box around paragraph
|
||||||
|
canvas.do.stroke_color(color).line_width(0.1)
|
||||||
|
for elem in self.hocr.iterfind(self._child_xpath('p', 'ocr_par')):
|
||||||
|
elemtxt = self._get_element_text(elem).strip()
|
||||||
|
if len(elemtxt) == 0:
|
||||||
|
continue
|
||||||
|
ocr_par = self.element_coordinates(elem)
|
||||||
|
if ocr_par is None:
|
||||||
|
continue
|
||||||
|
canvas.do.rect(
|
||||||
|
ocr_par.llx, ocr_par.lly, ocr_par.width, ocr_par.height, fill=False
|
||||||
|
)
|
||||||
|
|
||||||
|
def _debug_draw_line_bbox(self, canvas: Canvas, line_box: Rectangle, color=BLUE):
|
||||||
|
"""Render the bounding box of a text line."""
|
||||||
|
if not self.render_options.render_line_bbox: # pragma: no cover
|
||||||
|
return
|
||||||
|
with canvas.do.save_state():
|
||||||
|
canvas.do.stroke_color(color).line_width(0.15).rect(
|
||||||
|
line_box.llx, line_box.lly, line_box.width, line_box.height, fill=False
|
||||||
|
)
|
||||||
|
|
||||||
|
def _debug_draw_word_triangle(
|
||||||
|
self, canvas: Canvas, box: Rectangle, color=RED, line_width=0.1
|
||||||
|
):
|
||||||
|
"""Render a triangle that conveys word height and drawing direction."""
|
||||||
|
if not self.render_options.render_triangle: # pragma: no cover
|
||||||
|
return
|
||||||
|
with canvas.do.save_state():
|
||||||
|
canvas.do.stroke_color(color).line_width(line_width).line(
|
||||||
|
box.llx, box.lly, box.urx, box.lly
|
||||||
|
).line(box.urx, box.lly, box.llx, box.ury).line(
|
||||||
|
box.llx, box.lly, box.llx, box.ury
|
||||||
|
)
|
||||||
|
|
||||||
|
def _debug_draw_word_bbox(
|
||||||
|
self, canvas: Canvas, box: Rectangle, color=GREEN, line_width=0.1
|
||||||
|
):
|
||||||
|
"""Render a box depicting the word."""
|
||||||
|
if not self.render_options.render_word_bbox: # pragma: no cover
|
||||||
|
return
|
||||||
|
with canvas.do.save_state():
|
||||||
|
canvas.do.stroke_color(color).line_width(line_width).rect(
|
||||||
|
box.llx, box.lly, box.width, box.height, fill=False
|
||||||
|
)
|
||||||
|
|
||||||
|
def _debug_draw_space_bbox(
|
||||||
|
self, canvas: Canvas, box: Rectangle, color=DARKGREEN, line_width=0.1
|
||||||
|
):
|
||||||
|
"""Render a box depicting the space between two words."""
|
||||||
|
if not self.render_options.render_space_bbox: # pragma: no cover
|
||||||
|
return
|
||||||
|
with canvas.do.save_state():
|
||||||
|
canvas.do.fill_color(color).line_width(line_width).rect(
|
||||||
|
box.llx, box.lly, box.width, box.height, fill=True
|
||||||
|
)
|
||||||
|
|
||||||
|
def _debug_draw_baseline(
|
||||||
|
self,
|
||||||
|
canvas: Canvas,
|
||||||
|
line_box: Rectangle,
|
||||||
|
baseline_lly,
|
||||||
|
color=MAGENTA,
|
||||||
|
line_width=0.25,
|
||||||
|
):
|
||||||
|
"""Render the text baseline."""
|
||||||
|
if not self.render_options.render_baseline:
|
||||||
|
return
|
||||||
|
with canvas.do.save_state():
|
||||||
|
canvas.do.stroke_color(color).line_width(line_width).line(
|
||||||
|
line_box.llx,
|
||||||
|
baseline_lly,
|
||||||
|
line_box.urx,
|
||||||
|
baseline_lly,
|
||||||
|
)
|
||||||
@@ -7,17 +7,9 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import logging
|
import logging
|
||||||
from math import floor, sqrt
|
from math import floor, sqrt
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
# While from __future__ import annotations, we use singledispatch here, which
|
|
||||||
# does not support annotations. Disable check about using old-style typing
|
|
||||||
# until Python 3.10, OR when drop singledispatch in ocrmypdf 15.
|
|
||||||
# ruff: noqa: UP006
|
|
||||||
# ruff: noqa: UP007
|
|
||||||
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
@@ -38,9 +30,9 @@ def _calculate_downsample(
|
|||||||
image_size: tuple[int, int],
|
image_size: tuple[int, int],
|
||||||
bytes_per_pixel: int,
|
bytes_per_pixel: int,
|
||||||
*,
|
*,
|
||||||
max_size: Optional[tuple[int, int]] = None,
|
max_size: tuple[int, int] | None = None,
|
||||||
max_pixels: Optional[int] = None,
|
max_pixels: int | None = None,
|
||||||
max_bytes: Optional[int] = None,
|
max_bytes: int | None = None,
|
||||||
) -> tuple[int, int]:
|
) -> tuple[int, int]:
|
||||||
"""Calculate image size required to downsample an image to fit limits.
|
"""Calculate image size required to downsample an image to fit limits.
|
||||||
|
|
||||||
@@ -98,9 +90,9 @@ def _calculate_downsample(
|
|||||||
def calculate_downsample(
|
def calculate_downsample(
|
||||||
image: Image.Image,
|
image: Image.Image,
|
||||||
*,
|
*,
|
||||||
max_size: Optional[tuple[int, int]] = None,
|
max_size: tuple[int, int] | None = None,
|
||||||
max_pixels: Optional[int] = None,
|
max_pixels: int | None = None,
|
||||||
max_bytes: Optional[int] = None,
|
max_bytes: int | None = None,
|
||||||
) -> tuple[int, int]:
|
) -> tuple[int, int]:
|
||||||
"""Calculate image size required to downsample an image to fit limits.
|
"""Calculate image size required to downsample an image to fit limits.
|
||||||
|
|
||||||
|
|||||||
@@ -7,12 +7,12 @@ Derived from
|
|||||||
https://www.loc.gov/standards/iso639-2/ascii_8bits.html
|
https://www.loc.gov/standards/iso639-2/ascii_8bits.html
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
from typing import NamedTuple
|
from typing import NamedTuple
|
||||||
|
|
||||||
|
|
||||||
class ISOCodeData(NamedTuple):
|
class ISOCodeData(NamedTuple):
|
||||||
"""Data for a single ISO 639 code."""
|
"""Data for a single ISO 639 code."""
|
||||||
|
|
||||||
alt: str
|
alt: str
|
||||||
alpha_2: str
|
alpha_2: str
|
||||||
english: str
|
english: str
|
||||||
@@ -168,8 +168,10 @@ ISO_639_3 = {
|
|||||||
'chu': ISOCodeData(
|
'chu': ISOCodeData(
|
||||||
'',
|
'',
|
||||||
'cu',
|
'cu',
|
||||||
('Church Slavic; Old Slavonic; Church Slavonic;'
|
(
|
||||||
' Old Bulgarian; Old Church Slavonic'),
|
'Church Slavic; Old Slavonic; Church Slavonic;'
|
||||||
|
' Old Bulgarian; Old Church Slavonic'
|
||||||
|
),
|
||||||
"slavon d'église; vieux slave; slavon liturgique; vieux bulgare",
|
"slavon d'église; vieux slave; slavon liturgique; vieux bulgare",
|
||||||
),
|
),
|
||||||
'chv': ISOCodeData('', 'cv', 'Chuvash', 'tchouvache'),
|
'chv': ISOCodeData('', 'cv', 'Chuvash', 'tchouvache'),
|
||||||
|
|||||||
+29
-18
@@ -3,7 +3,6 @@
|
|||||||
|
|
||||||
"""Post-processing image optimization of OCR PDFs."""
|
"""Post-processing image optimization of OCR PDFs."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
@@ -11,11 +10,10 @@ import sys
|
|||||||
import tempfile
|
import tempfile
|
||||||
import threading
|
import threading
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from collections.abc import Iterator, MutableSet, Sequence
|
from collections.abc import Callable, Iterator, MutableSet, Sequence
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, NamedTuple, NewType
|
from typing import Any, NamedTuple, NewType
|
||||||
from warnings import warn
|
|
||||||
from zlib import compress
|
from zlib import compress
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
@@ -30,6 +28,7 @@ from pikepdf import (
|
|||||||
Stream,
|
Stream,
|
||||||
UnsupportedImageTypeError,
|
UnsupportedImageTypeError,
|
||||||
)
|
)
|
||||||
|
from pikepdf.models.image import HifiPrintImageNotTranscodableError
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||||
@@ -71,21 +70,20 @@ def jpg_name(root: Path, xref: Xref) -> Path:
|
|||||||
|
|
||||||
|
|
||||||
def extract_image_filter(
|
def extract_image_filter(
|
||||||
image: Stream, xref: Xref, *args
|
image: Stream, xref: Xref
|
||||||
) -> tuple[PdfImage, tuple[Name, Object]] | None:
|
) -> tuple[PdfImage, tuple[Name, Object]] | None:
|
||||||
"""Determine if an image is extractable."""
|
"""Determine if an image is extractable."""
|
||||||
if isinstance(image, Pdf):
|
|
||||||
# Support deprecated old function signature
|
|
||||||
# TODO Remove for v16 and drop *args from current function signature
|
|
||||||
image, xref = args[0], args[1]
|
|
||||||
warn("extract_image_filter: pdf, root parameters ignored", DeprecationWarning)
|
|
||||||
|
|
||||||
if image.Subtype != Name.Image:
|
if image.Subtype != Name.Image:
|
||||||
return None
|
return None
|
||||||
if image.Length < 100:
|
if not isinstance(image.Length, int) or image.Length < 100:
|
||||||
log.debug(f"xref {xref}: skipping image with small stream size")
|
log.debug(f"xref {xref}: skipping image with small stream size")
|
||||||
return None
|
return None
|
||||||
if image.Width < 8 or image.Height < 8: # Issue 732
|
if (
|
||||||
|
not isinstance(image.Width, int)
|
||||||
|
or not isinstance(image.Height, int)
|
||||||
|
or image.Width < 8
|
||||||
|
or image.Height < 8
|
||||||
|
): # Issue 732
|
||||||
log.debug(f"xref {xref}: skipping image with unusually small dimensions")
|
log.debug(f"xref {xref}: skipping image with unusually small dimensions")
|
||||||
return None
|
return None
|
||||||
|
|
||||||
@@ -97,6 +95,7 @@ def extract_image_filter(
|
|||||||
if (
|
if (
|
||||||
len(pim.filter_decodeparms) == 2
|
len(pim.filter_decodeparms) == 2
|
||||||
and first_filtdp[0] == Name.FlateDecode
|
and first_filtdp[0] == Name.FlateDecode
|
||||||
|
and first_filtdp[1] is not None
|
||||||
and first_filtdp[1].get(Name.Predictor, 1) == 1
|
and first_filtdp[1].get(Name.Predictor, 1) == 1
|
||||||
and second_filtdp[0] == Name.DCTDecode
|
and second_filtdp[0] == Name.DCTDecode
|
||||||
and not second_filtdp[1]
|
and not second_filtdp[1]
|
||||||
@@ -160,7 +159,17 @@ def extract_image_jbig2(
|
|||||||
imgname = root / f'{xref:08d}'
|
imgname = root / f'{xref:08d}'
|
||||||
with imgname.open('wb') as f:
|
with imgname.open('wb') as f:
|
||||||
ext = pim.extract_to(stream=f)
|
ext = pim.extract_to(stream=f)
|
||||||
imgname.rename(imgname.with_suffix(ext))
|
# Rename the file so it has .prejbig2.ext extension
|
||||||
|
# Making it unique avoids problems with Windows if the
|
||||||
|
# same image is extracted multiple times
|
||||||
|
imgname.rename(imgname.with_suffix(".prejbig2" + ext))
|
||||||
|
except NotImplementedError as e:
|
||||||
|
if '/Decode' in str(e):
|
||||||
|
log.debug(
|
||||||
|
f"xref {xref}: skipping image with unsupported Decode table"
|
||||||
|
)
|
||||||
|
return None
|
||||||
|
raise
|
||||||
except UnsupportedImageTypeError:
|
except UnsupportedImageTypeError:
|
||||||
return None
|
return None
|
||||||
finally:
|
finally:
|
||||||
@@ -169,7 +178,7 @@ def extract_image_jbig2(
|
|||||||
pim.obj.ColorSpace = colorspace
|
pim.obj.ColorSpace = colorspace
|
||||||
else:
|
else:
|
||||||
del pim.obj.ColorSpace
|
del pim.obj.ColorSpace
|
||||||
return XrefExt(xref, ext)
|
return XrefExt(xref, ".prejbig2" + ext)
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
@@ -200,7 +209,7 @@ def extract_image_generic(
|
|||||||
with imgname.open('wb') as f:
|
with imgname.open('wb') as f:
|
||||||
ext = pim.extract_to(stream=f)
|
ext = pim.extract_to(stream=f)
|
||||||
imgname.rename(imgname.with_suffix(ext))
|
imgname.rename(imgname.with_suffix(ext))
|
||||||
except UnsupportedImageTypeError:
|
except (UnsupportedImageTypeError, HifiPrintImageNotTranscodableError):
|
||||||
return None
|
return None
|
||||||
return XrefExt(xref, ext)
|
return XrefExt(xref, ext)
|
||||||
elif (
|
elif (
|
||||||
@@ -256,6 +265,9 @@ def _find_image_xrefs_container(
|
|||||||
for _imname, image in dict(xobjs).items():
|
for _imname, image in dict(xobjs).items():
|
||||||
if image.objgen[1] != 0:
|
if image.objgen[1] != 0:
|
||||||
continue # Ignore images in an incremental PDF
|
continue # Ignore images in an incremental PDF
|
||||||
|
xref = Xref(image.objgen[0])
|
||||||
|
if xref in include_xrefs or xref in exclude_xrefs:
|
||||||
|
continue # Already processed
|
||||||
if Name.Subtype in image and image.Subtype == Name.Form:
|
if Name.Subtype in image and image.Subtype == Name.Form:
|
||||||
# Recurse into Form XObjects
|
# Recurse into Form XObjects
|
||||||
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
||||||
@@ -269,7 +281,6 @@ def _find_image_xrefs_container(
|
|||||||
depth + 1,
|
depth + 1,
|
||||||
)
|
)
|
||||||
continue
|
continue
|
||||||
xref = Xref(image.objgen[0])
|
|
||||||
if Name.SMask in image:
|
if Name.SMask in image:
|
||||||
# Ignore soft masks
|
# Ignore soft masks
|
||||||
smask_xref = Xref(image.SMask.objgen[0])
|
smask_xref = Xref(image.SMask.objgen[0])
|
||||||
@@ -376,7 +387,7 @@ def _produce_jbig2_images(
|
|||||||
options.jbig2_threshold,
|
options.jbig2_threshold,
|
||||||
)
|
)
|
||||||
|
|
||||||
def jbig2_single_args(root, groups: dict[int, list[XrefExt]]):
|
def jbig2_single_args(root: Path, groups: dict[int, list[XrefExt]]):
|
||||||
for group, xref_exts in groups.items():
|
for group, xref_exts in groups.items():
|
||||||
prefix = f'group{group:08d}'
|
prefix = f'group{group:08d}'
|
||||||
# Second loop is to ensure multiple images per page are unpacked
|
# Second loop is to ensure multiple images per page are unpacked
|
||||||
|
|||||||
@@ -90,7 +90,7 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
|||||||
icc: ICC identifier such as 'sRGB'
|
icc: ICC identifier such as 'sRGB'
|
||||||
References:
|
References:
|
||||||
Adobe PDFMARK Reference:
|
Adobe PDFMARK Reference:
|
||||||
https://www.adobe.com/content/dam/acom/en/devnet/acrobat/pdfs/pdfmark_reference.pdf
|
https://opensource.adobe.com/dc-acrobat-sdk-docs/library/pdfmark/
|
||||||
"""
|
"""
|
||||||
if icc != 'sRGB':
|
if icc != 'sRGB':
|
||||||
raise NotImplementedError("Only supporting sRGB")
|
raise NotImplementedError("Only supporting sRGB")
|
||||||
|
|||||||
+122
-45
@@ -10,28 +10,28 @@ import atexit
|
|||||||
import logging
|
import logging
|
||||||
import re
|
import re
|
||||||
import statistics
|
import statistics
|
||||||
import sys
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from collections.abc import Container, Iterable, Iterator, Mapping, Sequence
|
from collections.abc import Callable, Container, Iterable, Iterator, Mapping, Sequence
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager, nullcontext
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from enum import Enum, auto
|
from enum import Enum, auto
|
||||||
from functools import partial
|
from functools import partial
|
||||||
from math import hypot, inf, isclose
|
from math import hypot, inf, isclose
|
||||||
from os import PathLike
|
from os import PathLike
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Callable, NamedTuple
|
from typing import NamedTuple
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
from pdfminer.layout import LTPage, LTTextBox
|
from pdfminer.layout import LTPage, LTTextBox
|
||||||
from pikepdf import (
|
from pikepdf import (
|
||||||
|
Dictionary,
|
||||||
|
Matrix,
|
||||||
Name,
|
Name,
|
||||||
Object,
|
Object,
|
||||||
Page,
|
Page,
|
||||||
Pdf,
|
Pdf,
|
||||||
PdfImage,
|
PdfImage,
|
||||||
PdfInlineImage,
|
PdfInlineImage,
|
||||||
PdfMatrix,
|
|
||||||
Stream,
|
Stream,
|
||||||
UnsupportedImageTypeError,
|
UnsupportedImageTypeError,
|
||||||
parse_content_stream,
|
parse_content_stream,
|
||||||
@@ -41,7 +41,12 @@ from ocrmypdf._concurrent import Executor, SerialExecutor
|
|||||||
from ocrmypdf._progressbar import ProgressBar
|
from ocrmypdf._progressbar import ProgressBar
|
||||||
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
||||||
from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap
|
from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap
|
||||||
from ocrmypdf.pdfinfo.layout import LTStateAwareChar, get_page_analysis, get_text_boxes
|
from ocrmypdf.pdfinfo.layout import (
|
||||||
|
LTStateAwareChar,
|
||||||
|
PdfMinerState,
|
||||||
|
get_page_analysis,
|
||||||
|
get_text_boxes,
|
||||||
|
)
|
||||||
|
|
||||||
logger = logging.getLogger()
|
logger = logging.getLogger()
|
||||||
|
|
||||||
@@ -209,7 +214,7 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
|||||||
CTM unchanged.
|
CTM unchanged.
|
||||||
"""
|
"""
|
||||||
stack = []
|
stack = []
|
||||||
ctm = PdfMatrix(initial_shorthand)
|
ctm = Matrix(initial_shorthand)
|
||||||
xobject_settings: list[XobjectSettings] = []
|
xobject_settings: list[XobjectSettings] = []
|
||||||
inline_images: list[InlineSettings] = []
|
inline_images: list[InlineSettings] = []
|
||||||
name_index = defaultdict(lambda: [])
|
name_index = defaultdict(lambda: [])
|
||||||
@@ -240,7 +245,13 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
|||||||
# to do. Just pretend nothing happened, keep calm and carry on.
|
# to do. Just pretend nothing happened, keep calm and carry on.
|
||||||
warn("PDF graphics stack underflowed - PDF may be malformed")
|
warn("PDF graphics stack underflowed - PDF may be malformed")
|
||||||
elif operator == 'cm':
|
elif operator == 'cm':
|
||||||
ctm = PdfMatrix(operands) @ ctm
|
try:
|
||||||
|
ctm = Matrix(operands) @ ctm
|
||||||
|
except ValueError:
|
||||||
|
raise InputFileError(
|
||||||
|
"PDF content stream is corrupt - this PDF is malformed. "
|
||||||
|
"Use a PDF editor that is capable of visually inspecting the PDF."
|
||||||
|
)
|
||||||
elif operator == 'Do':
|
elif operator == 'Do':
|
||||||
image_name = operands[0]
|
image_name = operands[0]
|
||||||
settings = XobjectSettings(
|
settings = XobjectSettings(
|
||||||
@@ -364,8 +375,26 @@ class ImageInfo:
|
|||||||
pim = PdfImage(pdfimage)
|
pim = PdfImage(pdfimage)
|
||||||
else:
|
else:
|
||||||
raise ValueError("Either pdfimage or inline must be set")
|
raise ValueError("Either pdfimage or inline must be set")
|
||||||
|
|
||||||
self._width = pim.width
|
self._width = pim.width
|
||||||
self._height = pim.height
|
self._height = pim.height
|
||||||
|
if (smask := pim.obj.get(Name.SMask, None)) is not None:
|
||||||
|
# SMask is pretty much an alpha channel, but in PDF it's possible
|
||||||
|
# for channel to have different dimensions than the image
|
||||||
|
# itself. Some PDF writers use this to create a grayscale stencil
|
||||||
|
# mask. For our purposes, the effective size is the size of the
|
||||||
|
# larger component (image or smask).
|
||||||
|
if isinstance(smask, Stream | Dictionary):
|
||||||
|
self._width = max(smask.get(Name.Width, 0), self._width)
|
||||||
|
self._height = max(smask.get(Name.Height, 0), self._height)
|
||||||
|
if (mask := pim.obj.get(Name.Mask, None)) is not None:
|
||||||
|
# If the image has a /Mask entry, it has an explicit mask.
|
||||||
|
# /Mask can be a Stream or an Array. If it's a Stream,
|
||||||
|
# use its /Width and /Height if they are larger than the main
|
||||||
|
# image's.
|
||||||
|
if isinstance(mask, Stream | Dictionary):
|
||||||
|
self._width = max(mask.get(Name.Width, 0), self._width)
|
||||||
|
self._height = max(mask.get(Name.Height, 0), self._height)
|
||||||
|
|
||||||
# If /ImageMask is true, then this image is a stencil mask
|
# If /ImageMask is true, then this image is a stencil mask
|
||||||
# (Images that draw with this stencil mask will have a reference to
|
# (Images that draw with this stencil mask will have a reference to
|
||||||
@@ -469,9 +498,18 @@ class ImageInfo:
|
|||||||
def renderable(self) -> bool:
|
def renderable(self) -> bool:
|
||||||
"""Whether the image is renderable.
|
"""Whether the image is renderable.
|
||||||
|
|
||||||
Some PDFs in the wild have invalid images that are not renderable.
|
Some PDFs in the wild have invalid images that are not renderable,
|
||||||
|
due to unusual dimensions.
|
||||||
|
|
||||||
|
Stencil masks are not also not renderable, since they are not
|
||||||
|
drawn, but rather they control how rendering happens.
|
||||||
"""
|
"""
|
||||||
return self.dpi.is_finite and self.width >= 0 and self.height >= 0
|
return (
|
||||||
|
self.dpi.is_finite
|
||||||
|
and self.width >= 0
|
||||||
|
and self.height >= 0
|
||||||
|
and self.type_ != 'stencil'
|
||||||
|
)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def dpi(self) -> Resolution:
|
def dpi(self) -> Resolution:
|
||||||
@@ -486,7 +524,7 @@ class ImageInfo:
|
|||||||
"""Physical area of the image in square inches."""
|
"""Physical area of the image in square inches."""
|
||||||
if not self.renderable:
|
if not self.renderable:
|
||||||
return 0.0
|
return 0.0
|
||||||
return float(self.width * self.dpi.x * self.height * self.dpi.y)
|
return float((self.width / self.dpi.x) * (self.height / self.dpi.y))
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
"""Return a string representation of the image."""
|
"""Return a string representation of the image."""
|
||||||
@@ -568,7 +606,7 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
|||||||
xobjs = resources[Name.XObject].as_dict()
|
xobjs = resources[Name.XObject].as_dict()
|
||||||
for xobj in xobjs:
|
for xobj in xobjs:
|
||||||
candidate = xobjs[xobj]
|
candidate = xobjs[xobj]
|
||||||
if candidate is None or candidate[Name.Subtype] != Name.Form:
|
if candidate is None or candidate.get(Name.Subtype) != Name.Form:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
form_xobject = candidate
|
form_xobject = candidate
|
||||||
@@ -614,12 +652,12 @@ def _process_content_streams(
|
|||||||
):
|
):
|
||||||
# Set the CTM to the state it was when the "Do" operator was
|
# Set the CTM to the state it was when the "Do" operator was
|
||||||
# encountered that is drawing this instance of the Form XObject
|
# encountered that is drawing this instance of the Form XObject
|
||||||
ctm = PdfMatrix(shorthand) if shorthand else PdfMatrix.identity()
|
ctm = Matrix(shorthand) if shorthand else Matrix()
|
||||||
|
|
||||||
# A Form XObject may provide its own matrix to map form space into
|
# A Form XObject may provide its own matrix to map form space into
|
||||||
# user space. Get this if one exists
|
# user space. Get this if one exists
|
||||||
form_shorthand = container.get(Name.Matrix, PdfMatrix.identity())
|
form_shorthand = container.get(Name.Matrix, Matrix())
|
||||||
form_matrix = PdfMatrix(form_shorthand)
|
form_matrix = Matrix(form_shorthand)
|
||||||
|
|
||||||
# Concatenate form matrix with CTM to ensure CTM is correct for
|
# Concatenate form matrix with CTM to ensure CTM is correct for
|
||||||
# drawing this instance of the XObject
|
# drawing this instance of the XObject
|
||||||
@@ -669,13 +707,13 @@ def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) ->
|
|||||||
|
|
||||||
|
|
||||||
def simplify_textboxes(
|
def simplify_textboxes(
|
||||||
miner: LTPage, textbox_getter: Callable[[LTPage], Iterator[LTTextBox]]
|
miner_page: LTPage, textbox_getter: Callable[[LTPage], Iterator[LTTextBox]]
|
||||||
) -> Iterator[TextboxInfo]:
|
) -> Iterator[TextboxInfo]:
|
||||||
"""Extract only limited content from text boxes.
|
"""Extract only limited content from text boxes.
|
||||||
|
|
||||||
We do this to save memory and ensure that our objects are pickleable.
|
We do this to save memory and ensure that our objects are pickleable.
|
||||||
"""
|
"""
|
||||||
for box in textbox_getter(miner):
|
for box in textbox_getter(miner_page):
|
||||||
first_line = box._objs[0] # pylint: disable=protected-access
|
first_line = box._objs[0] # pylint: disable=protected-access
|
||||||
first_char = first_line._objs[0] # pylint: disable=protected-access
|
first_char = first_line._objs[0] # pylint: disable=protected-access
|
||||||
if not isinstance(first_char, LTStateAwareChar):
|
if not isinstance(first_char, LTStateAwareChar):
|
||||||
@@ -722,9 +760,12 @@ def _pdf_pageinfo_sync(
|
|||||||
infile: Path,
|
infile: Path,
|
||||||
check_pages: Container[int],
|
check_pages: Container[int],
|
||||||
detailed_analysis: bool,
|
detailed_analysis: bool,
|
||||||
|
miner_state: PdfMinerState | None,
|
||||||
) -> PageInfo:
|
) -> PageInfo:
|
||||||
with _pdf_pageinfo_sync_pdf(thread_pdf, infile) as pdf:
|
with _pdf_pageinfo_sync_pdf(thread_pdf, infile) as pdf:
|
||||||
return PageInfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
return PageInfo(
|
||||||
|
pdf, pageno, infile, check_pages, detailed_analysis, miner_state
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _pdf_pageinfo_concurrent(
|
def _pdf_pageinfo_concurrent(
|
||||||
@@ -736,6 +777,7 @@ def _pdf_pageinfo_concurrent(
|
|||||||
progbar,
|
progbar,
|
||||||
check_pages,
|
check_pages,
|
||||||
detailed_analysis: bool = False,
|
detailed_analysis: bool = False,
|
||||||
|
miner_state: PdfMinerState | None = None,
|
||||||
) -> Sequence[PageInfo | None]:
|
) -> Sequence[PageInfo | None]:
|
||||||
pages: list[PageInfo | None] = [None] * len(pdf.pages)
|
pages: list[PageInfo | None] = [None] * len(pdf.pages)
|
||||||
|
|
||||||
@@ -767,7 +809,8 @@ def _pdf_pageinfo_concurrent(
|
|||||||
initial_pdf = pdf if use_threads else None
|
initial_pdf = pdf if use_threads else None
|
||||||
|
|
||||||
contexts = (
|
contexts = (
|
||||||
(n, initial_pdf, infile, check_pages, detailed_analysis) for n in range(total)
|
(n, initial_pdf, infile, check_pages, detailed_analysis, miner_state)
|
||||||
|
for n in range(total)
|
||||||
)
|
)
|
||||||
assert n_workers == 1 if use_threads else n_workers >= 1, "Not multithreadable"
|
assert n_workers == 1 if use_threads else n_workers >= 1, "Not multithreadable"
|
||||||
logger.debug(
|
logger.debug(
|
||||||
@@ -834,12 +877,15 @@ class PageInfo:
|
|||||||
infile: PathLike,
|
infile: PathLike,
|
||||||
check_pages: Container[int],
|
check_pages: Container[int],
|
||||||
detailed_analysis: bool = False,
|
detailed_analysis: bool = False,
|
||||||
|
miner_state: PdfMinerState | None = None,
|
||||||
):
|
):
|
||||||
"""Initialize a PageInfo object."""
|
"""Initialize a PageInfo object."""
|
||||||
self._pageno = pageno
|
self._pageno = pageno
|
||||||
self._infile = infile
|
self._infile = infile
|
||||||
self._detailed_analysis = detailed_analysis
|
self._detailed_analysis = detailed_analysis
|
||||||
self._gather_pageinfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
self._gather_pageinfo(
|
||||||
|
pdf, pageno, infile, check_pages, detailed_analysis, miner_state
|
||||||
|
)
|
||||||
|
|
||||||
def _gather_pageinfo(
|
def _gather_pageinfo(
|
||||||
self,
|
self,
|
||||||
@@ -848,19 +894,27 @@ class PageInfo:
|
|||||||
infile: PathLike,
|
infile: PathLike,
|
||||||
check_pages: Container[int],
|
check_pages: Container[int],
|
||||||
detailed_analysis: bool,
|
detailed_analysis: bool,
|
||||||
|
miner_state: PdfMinerState | None,
|
||||||
):
|
):
|
||||||
page: Page = pdf.pages[pageno]
|
page: Page = pdf.pages[pageno]
|
||||||
mediabox = [Decimal(d) for d in page.mediabox.as_list()]
|
mediabox = [Decimal(d) for d in page.mediabox.as_list()]
|
||||||
width_pt = mediabox[2] - mediabox[0]
|
width_pt = mediabox[2] - mediabox[0]
|
||||||
height_pt = mediabox[3] - mediabox[1]
|
height_pt = mediabox[3] - mediabox[1]
|
||||||
|
|
||||||
|
# self._artbox = [float(d) for d in page.artbox.as_list()]
|
||||||
|
# self._bleedbox = [float(d) for d in page.bleedbox.as_list()]
|
||||||
|
self._cropbox = [float(d) for d in page.cropbox.as_list()]
|
||||||
|
self._mediabox = [float(d) for d in page.mediabox.as_list()]
|
||||||
|
self._trimbox = [float(d) for d in page.trimbox.as_list()]
|
||||||
|
|
||||||
check_this_page = pageno in check_pages
|
check_this_page = pageno in check_pages
|
||||||
|
|
||||||
if check_this_page and detailed_analysis:
|
if check_this_page and detailed_analysis:
|
||||||
pscript5_mode = str(pdf.docinfo.get(Name.Creator)).startswith('PScript5')
|
page_analysis = miner_state.get_page_analysis(pageno)
|
||||||
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
if page_analysis is not None:
|
||||||
if miner is not None:
|
self._textboxes = list(
|
||||||
self._textboxes = list(simplify_textboxes(miner, get_text_boxes))
|
simplify_textboxes(page_analysis, get_text_boxes)
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
self._textboxes = []
|
self._textboxes = []
|
||||||
bboxes = (box.bbox for box in self._textboxes)
|
bboxes = (box.bbox for box in self._textboxes)
|
||||||
@@ -970,6 +1024,21 @@ class PageInfo:
|
|||||||
else:
|
else:
|
||||||
raise ValueError("rotation must be a cardinal angle")
|
raise ValueError("rotation must be a cardinal angle")
|
||||||
|
|
||||||
|
@property
|
||||||
|
def cropbox(self) -> FloatRect:
|
||||||
|
"""Return cropbox of page in PDF coordinates."""
|
||||||
|
return self._cropbox
|
||||||
|
|
||||||
|
@property
|
||||||
|
def mediabox(self) -> FloatRect:
|
||||||
|
"""Return mediabox of page in PDF coordinates."""
|
||||||
|
return self._mediabox
|
||||||
|
|
||||||
|
@property
|
||||||
|
def trimbox(self) -> FloatRect:
|
||||||
|
"""Return trimbox of page in PDF coordinates."""
|
||||||
|
return self._trimbox
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def images(self) -> list[ImageInfo]:
|
def images(self) -> list[ImageInfo]:
|
||||||
"""Return images."""
|
"""Return images."""
|
||||||
@@ -1029,28 +1098,26 @@ class PageInfo:
|
|||||||
|
|
||||||
Returns None if there is no meaningful DPI for the page.
|
Returns None if there is no meaningful DPI for the page.
|
||||||
"""
|
"""
|
||||||
image_dpis = [
|
image_dpis = []
|
||||||
image.dpi.to_scalar() for image in self._images if image.renderable
|
image_areas = []
|
||||||
]
|
for image in self._images:
|
||||||
image_areas = [image.printed_area for image in self._images if image.renderable]
|
if not image.renderable:
|
||||||
|
continue
|
||||||
|
image_dpis.append(image.dpi.to_scalar())
|
||||||
|
image_areas.append(image.printed_area)
|
||||||
|
|
||||||
total_drawn_area = sum(image_areas)
|
total_drawn_area = sum(image_areas)
|
||||||
if total_drawn_area == 0:
|
if total_drawn_area == 0:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
weights = [area / total_drawn_area for area in image_areas]
|
weights = [area / total_drawn_area for area in image_areas]
|
||||||
# Calculate harmonic mean of DPIs weighted by area
|
# Calculate harmonic mean of DPIs weighted by area
|
||||||
if sys.version_info >= (3, 10):
|
weighted_dpi = statistics.harmonic_mean(image_dpis, weights)
|
||||||
weighted_dpi = statistics.harmonic_mean(image_dpis, weights)
|
|
||||||
else:
|
|
||||||
weighted_dpi = sum(weights) / sum(
|
|
||||||
weight / dpi for weight, dpi in zip(weights, image_dpis)
|
|
||||||
)
|
|
||||||
max_dpi = max(image_dpis)
|
max_dpi = max(image_dpis)
|
||||||
dpi_average_max_ratio = weighted_dpi / max_dpi
|
dpi_average_max_ratio = weighted_dpi / max_dpi
|
||||||
|
|
||||||
arg_max_dpi = image_dpis.index(max_dpi)
|
arg_max_dpi = image_dpis.index(max_dpi)
|
||||||
max_area_ratio = image_areas[arg_max_dpi] / total_drawn_area
|
max_area_ratio = image_areas[arg_max_dpi] / total_drawn_area
|
||||||
|
|
||||||
return PageResolutionProfile(
|
return PageResolutionProfile(
|
||||||
weighted_dpi,
|
weighted_dpi,
|
||||||
max_dpi,
|
max_dpi,
|
||||||
@@ -1100,16 +1167,26 @@ class PdfInfo:
|
|||||||
with Pdf.open(infile) as pdf:
|
with Pdf.open(infile) as pdf:
|
||||||
if pdf.is_encrypted:
|
if pdf.is_encrypted:
|
||||||
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
||||||
self._pages = _pdf_pageinfo_concurrent(
|
pscript5_mode = str(pdf.docinfo.get(Name.Creator, "")).startswith(
|
||||||
pdf,
|
'PScript5'
|
||||||
executor,
|
|
||||||
max_workers,
|
|
||||||
use_threads,
|
|
||||||
infile,
|
|
||||||
progbar,
|
|
||||||
check_pages=check_pages,
|
|
||||||
detailed_analysis=detailed_analysis,
|
|
||||||
)
|
)
|
||||||
|
self._miner_state = (
|
||||||
|
PdfMinerState(infile, pscript5_mode)
|
||||||
|
if detailed_analysis
|
||||||
|
else nullcontext()
|
||||||
|
)
|
||||||
|
with self._miner_state as miner_state:
|
||||||
|
self._pages = _pdf_pageinfo_concurrent(
|
||||||
|
pdf,
|
||||||
|
executor,
|
||||||
|
max_workers,
|
||||||
|
use_threads,
|
||||||
|
infile,
|
||||||
|
progbar,
|
||||||
|
check_pages=check_pages,
|
||||||
|
detailed_analysis=detailed_analysis,
|
||||||
|
miner_state=miner_state,
|
||||||
|
)
|
||||||
self._needs_rendering = pdf.Root.get(Name.NeedsRendering, False)
|
self._needs_rendering = pdf.Root.get(Name.NeedsRendering, False)
|
||||||
if Name.AcroForm in pdf.Root:
|
if Name.AcroForm in pdf.Root:
|
||||||
if len(pdf.Root.AcroForm.get(Name.Fields, [])) > 0:
|
if len(pdf.Root.AcroForm.get(Name.Fields, [])) > 0:
|
||||||
@@ -1155,7 +1232,7 @@ class PdfInfo:
|
|||||||
@property
|
@property
|
||||||
def filename(self) -> str | Path:
|
def filename(self) -> str | Path:
|
||||||
"""Return filename of PDF."""
|
"""Return filename of PDF."""
|
||||||
if not isinstance(self._infile, (str, Path)):
|
if not isinstance(self._infile, str | Path):
|
||||||
raise NotImplementedError("can't get filename from stream")
|
raise NotImplementedError("can't get filename from stream")
|
||||||
return self._infile
|
return self._infile
|
||||||
|
|
||||||
|
|||||||
@@ -5,18 +5,20 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import re
|
import re
|
||||||
from collections.abc import Mapping
|
from collections.abc import Iterator, Mapping
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from math import copysign
|
from math import copysign
|
||||||
from os import PathLike
|
from os import PathLike
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Iterator
|
from typing import Any
|
||||||
from unittest.mock import patch
|
from unittest.mock import patch
|
||||||
|
|
||||||
import pdfminer
|
import pdfminer
|
||||||
import pdfminer.encodingdb
|
import pdfminer.encodingdb
|
||||||
import pdfminer.pdfdevice
|
import pdfminer.pdfdevice
|
||||||
import pdfminer.pdfinterp
|
import pdfminer.pdfinterp
|
||||||
|
import pdfminer.psparser
|
||||||
|
from deprecation import deprecated
|
||||||
from pdfminer.converter import PDFLayoutAnalyzer
|
from pdfminer.converter import PDFLayoutAnalyzer
|
||||||
from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox
|
from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox
|
||||||
from pdfminer.pdfcolor import PDFColorSpace
|
from pdfminer.pdfcolor import PDFColorSpace
|
||||||
@@ -58,9 +60,10 @@ def pdfsimplefont__init__(
|
|||||||
|
|
||||||
setattr(PDFSimpleFont, '__init__', pdfsimplefont__init__)
|
setattr(PDFSimpleFont, '__init__', pdfsimplefont__init__)
|
||||||
|
|
||||||
#
|
# Patch pdfminer.six buffer size
|
||||||
# pdfminer patches when creator is PScript5.dll
|
# The parser doesn't properly handle keyword tokens are split across the end of the
|
||||||
#
|
# buffer, so increase the buffer size something far larger than will ever be seen.
|
||||||
|
pdfminer.psparser.PSBaseParser.BUFSIZ = 256 * 1024 * 1024
|
||||||
|
|
||||||
|
|
||||||
def pdftype3font__pscript5_get_height(self):
|
def pdftype3font__pscript5_get_height(self):
|
||||||
@@ -287,6 +290,7 @@ def patch_pdfminer(pscript5_mode: bool):
|
|||||||
yield
|
yield
|
||||||
|
|
||||||
|
|
||||||
|
@deprecated(deprecated_in='16.6.0', details='Use PdfMinerState instead.')
|
||||||
def get_page_analysis(
|
def get_page_analysis(
|
||||||
infile: PathLike, pageno: int, pscript5_mode: bool
|
infile: PathLike, pageno: int, pscript5_mode: bool
|
||||||
) -> LTPage | None:
|
) -> LTPage | None:
|
||||||
@@ -317,6 +321,73 @@ def get_page_analysis(
|
|||||||
return dev.get_result()
|
return dev.get_result()
|
||||||
|
|
||||||
|
|
||||||
|
class PdfMinerState:
|
||||||
|
"""Provide a context manager for using pdfminer.six.
|
||||||
|
|
||||||
|
This ensures that the file is closed. It also provides a cache of pages
|
||||||
|
from the PDF so that they can be reused if needed, to improve performance.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, infile: Path, pscript5_mode: bool) -> None:
|
||||||
|
"""Initialize the context manager.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
infile: The path to the PDF file to be analyzed.
|
||||||
|
pscript5_mode: Whether the PDF was generated by PScript5.dll.
|
||||||
|
"""
|
||||||
|
self.infile = infile
|
||||||
|
self.rman = pdfminer.pdfinterp.PDFResourceManager(caching=True)
|
||||||
|
self.disable_boxes_flow = None
|
||||||
|
self.page_cache: list[PDFPage] = []
|
||||||
|
self.pscript5_mode = pscript5_mode
|
||||||
|
self.file = None
|
||||||
|
|
||||||
|
def __enter__(self):
|
||||||
|
"""Enter the context manager."""
|
||||||
|
self.file = Path(self.infile).open('rb')
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
|
"""Exit the context manager."""
|
||||||
|
if self.file:
|
||||||
|
self.file.close()
|
||||||
|
return True
|
||||||
|
|
||||||
|
def _load_page_cache(self):
|
||||||
|
"""Load the page cache."""
|
||||||
|
try:
|
||||||
|
self.page_cache = list(PDFPage.get_pages(self.file))
|
||||||
|
if not self.page_cache:
|
||||||
|
raise InputFileError(
|
||||||
|
"pdfminer did not find any pages in the input file."
|
||||||
|
)
|
||||||
|
for n, page in enumerate(self.page_cache):
|
||||||
|
if page is None:
|
||||||
|
raise InputFileError(
|
||||||
|
f"pdfminer could not process page {n} (counting from 0)."
|
||||||
|
)
|
||||||
|
except PDFTextExtractionNotAllowed as e:
|
||||||
|
raise EncryptedPdfError() from e
|
||||||
|
|
||||||
|
def get_page_analysis(self, pageno: int):
|
||||||
|
"""Get the page analysis for a given page."""
|
||||||
|
if not self.page_cache:
|
||||||
|
self._load_page_cache()
|
||||||
|
page = self.page_cache[pageno]
|
||||||
|
dev = TextPositionTracker(
|
||||||
|
self.rman,
|
||||||
|
laparams=LAParams(
|
||||||
|
all_texts=True, detect_vertical=True, boxes_flow=self.disable_boxes_flow
|
||||||
|
),
|
||||||
|
)
|
||||||
|
interp = pdfminer.pdfinterp.PDFPageInterpreter(self.rman, dev)
|
||||||
|
|
||||||
|
with patch_pdfminer(self.pscript5_mode):
|
||||||
|
interp.process_page(page)
|
||||||
|
|
||||||
|
return dev.get_result()
|
||||||
|
|
||||||
|
|
||||||
def get_text_boxes(obj) -> Iterator[LTTextBox]:
|
def get_text_boxes(obj) -> Iterator[LTTextBox]:
|
||||||
"""Get the text boxes attached to the current node."""
|
"""Get the text boxes attached to the current node."""
|
||||||
for child in obj:
|
for child in obj:
|
||||||
|
|||||||
@@ -3,7 +3,6 @@
|
|||||||
|
|
||||||
"""Utilities to measure OCR quality."""
|
"""Utilities to measure OCR quality."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import re
|
import re
|
||||||
|
|||||||
@@ -8,12 +8,11 @@ import logging
|
|||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
from collections.abc import Mapping, Sequence
|
from collections.abc import Callable, Mapping, Sequence
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||||
from subprocess import run as subprocess_run
|
from subprocess import run as subprocess_run
|
||||||
from typing import Callable, Union
|
|
||||||
|
|
||||||
from packaging.version import Version
|
from packaging.version import Version
|
||||||
|
|
||||||
@@ -23,7 +22,7 @@ from ocrmypdf.exceptions import MissingDependencyError
|
|||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
Args = Sequence[Union[Path, str]]
|
Args = Sequence[Path | str]
|
||||||
OsEnviron = os._Environ # pylint: disable=protected-access
|
OsEnviron = os._Environ # pylint: disable=protected-access
|
||||||
|
|
||||||
|
|
||||||
@@ -172,6 +171,7 @@ def get_version(
|
|||||||
) from e
|
) from e
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
if e.returncode != 0:
|
if e.returncode != 0:
|
||||||
|
log.exception(e)
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
f"Ran program '{program}' but it exited with an error:\n{e.output}"
|
f"Ran program '{program}' but it exited with an error:\n{e.output}"
|
||||||
) from e
|
) from e
|
||||||
@@ -215,8 +215,10 @@ to have {found_version}. Please update this program.
|
|||||||
|
|
||||||
OLD_VERSION_REQUIRED_FOR = '''
|
OLD_VERSION_REQUIRED_FOR = '''
|
||||||
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
||||||
{required_for} arguments. If you omit these arguments, OCRmyPDF may be able to
|
{required_for} arguments. {program} {found_version} is installed.
|
||||||
proceed. For best results, install the program.
|
|
||||||
|
If you omit these arguments, OCRmyPDF may be able to
|
||||||
|
proceed. For best results, update the program.
|
||||||
'''
|
'''
|
||||||
|
|
||||||
OSX_INSTALL_ADVICE = '''
|
OSX_INSTALL_ADVICE = '''
|
||||||
|
|||||||
@@ -9,18 +9,13 @@ import os
|
|||||||
import re
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
from collections.abc import Iterable, Iterator
|
from collections.abc import Callable, Iterable, Iterator
|
||||||
from itertools import chain
|
from itertools import chain
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, TypeVar
|
from typing import Any, TypeAlias, TypeVar
|
||||||
|
|
||||||
from packaging.version import InvalidVersion, Version
|
from packaging.version import InvalidVersion, Version
|
||||||
|
|
||||||
if sys.version_info >= (3, 10):
|
|
||||||
from typing import TypeAlias
|
|
||||||
else:
|
|
||||||
from typing_extensions import TypeAlias # pragma: no cover
|
|
||||||
|
|
||||||
if sys.platform == 'win32':
|
if sys.platform == 'win32':
|
||||||
# mypy understands 'if sys.platform' better than try/except ModuleNotFoundError
|
# mypy understands 'if sys.platform' better than try/except ModuleNotFoundError
|
||||||
import winreg # pylint: disable=import-error
|
import winreg # pylint: disable=import-error
|
||||||
@@ -84,7 +79,7 @@ def registry_path_ghostscript(env=None) -> Iterator[Path]:
|
|||||||
registry_subkeys(k), key=ghostscript_version_key, default=(0, 0, 0)
|
registry_subkeys(k), key=ghostscript_version_key, default=(0, 0, 0)
|
||||||
)
|
)
|
||||||
with winreg.OpenKey(
|
with winreg.OpenKey(
|
||||||
winreg.HKEY_LOCAL_MACHINE, fr"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
|
winreg.HKEY_LOCAL_MACHINE, rf"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
|
||||||
) as k:
|
) as k:
|
||||||
for _, gs_path, _ in registry_values(k):
|
for _, gs_path, _ in registry_values(k):
|
||||||
yield Path(gs_path) / 'bin'
|
yield Path(gs_path) / 'bin'
|
||||||
|
|||||||
+177
@@ -0,0 +1,177 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||||
|
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||||
|
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||||
|
<head>
|
||||||
|
<title></title>
|
||||||
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||||
|
<meta name='ocr-system' content='tesseract 5.3.2' />
|
||||||
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.jcdzf4up/000001_ocr.png"; bbox 0 0 4000 2864; ppageno 0; scan_res 2400 2400'>
|
||||||
|
<div class='ocr_carea' id='block_1_1' title="bbox 251 146 2173 237">
|
||||||
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 251 146 2173 237">
|
||||||
|
<span class='ocr_line' id='line_1_1' title="bbox 251 146 2173 237; baseline -0.006 5; x_size 99.418808; x_descenders 9.4188042; x_ascenders 32">
|
||||||
|
<span class='ocrx_word' id='word_1_1' title='bbox 251 154 274 176; x_wconf 89'>i</span>
|
||||||
|
<span class='ocrx_word' id='word_1_2' title='bbox 1080 146 1152 237; x_wconf 89'>a</span>
|
||||||
|
<span class='ocrx_word' id='word_1_3' title='bbox 1254 153 1406 235; x_wconf 92'>la</span>
|
||||||
|
<span class='ocrx_word' id='word_1_4' title='bbox 1500 153 2173 235; x_wconf 95'>Waterman</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_separator' id='block_1_2' title="bbox 135 202 2180 295"></div>
|
||||||
|
<div class='ocr_carea' id='block_1_3' title="bbox 145 414 2929 1221">
|
||||||
|
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 145 414 1154 516">
|
||||||
|
<span class='ocr_line' id='line_1_2' title="bbox 145 414 1154 516; baseline -0.005 -4; x_size 122.38636; x_descenders 24.386362; x_ascenders 40">
|
||||||
|
<span class='ocrx_word' id='word_1_5' title='bbox 145 414 211 512; x_wconf 60'>4</span>
|
||||||
|
<span class='ocrx_word' id='word_1_6' title='bbox 318 453 551 516; x_wconf 93'>ons</span>
|
||||||
|
<span class='ocrx_word' id='word_1_7' title='bbox 660 430 1154 512; x_wconf 91'>linzen</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_3' lang='eng' title="bbox 146 568 1239 667">
|
||||||
|
<span class='ocr_line' id='line_1_3' title="bbox 146 568 1239 667; baseline -0.004 -16; x_size 99; x_descenders 17; x_ascenders 24">
|
||||||
|
<span class='ocrx_word' id='word_1_8' title='bbox 146 569 209 667; x_wconf 94'>3</span>
|
||||||
|
<span class='ocrx_word' id='word_1_9' title='bbox 323 568 729 652; x_wconf 83'>liter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_10' title='bbox 821 569 1239 650; x_wconf 96'>water</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_4' lang='eng' title="bbox 148 705 645 804">
|
||||||
|
<span class='ocr_line' id='line_1_4' title="bbox 148 705 645 804; baseline -0.016 -14; x_size 103; x_descenders 19; x_ascenders 27">
|
||||||
|
<span class='ocrx_word' id='word_1_11' title='bbox 148 706 211 804; x_wconf 88'>3</span>
|
||||||
|
<span class='ocrx_word' id='word_1_12' title='bbox 311 705 645 789; x_wconf 52'>uien</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_5' lang='eng' title="bbox 147 832 1154 950">
|
||||||
|
<span class='ocr_line' id='line_1_5' title="bbox 147 832 1154 950; baseline -0.004 -27; x_size 118; x_descenders 28; x_ascenders 30">
|
||||||
|
<span class='ocrx_word' id='word_1_13' title='bbox 147 832 623 950; x_wconf 91'>bloem,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_14' title='bbox 737 843 1154 924; x_wconf 91'>boter</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_6' lang='eng' title="bbox 148 979 1241 1087">
|
||||||
|
<span class='ocr_line' id='line_1_6' title="bbox 148 979 1241 1087; baseline -0.005 -21; x_size 107; x_descenders 24; x_ascenders 24">
|
||||||
|
<span class='ocrx_word' id='word_1_15' title='bbox 148 983 215 1066; x_wconf 88'>2</span>
|
||||||
|
<span class='ocrx_word' id='word_1_16' title='bbox 312 983 807 1087; x_wconf 88'>kopjes</span>
|
||||||
|
<span class='ocrx_word' id='word_1_17' title='bbox 905 979 1241 1062; x_wconf 92'>melk</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_7' lang='eng' title="bbox 154 1109 2929 1221">
|
||||||
|
<span class='ocr_line' id='line_1_7' title="bbox 154 1109 2929 1221; baseline -0.003 -21; x_size 112; x_descenders 28; x_ascenders 26">
|
||||||
|
<span class='ocrx_word' id='word_1_18' title='bbox 154 1117 791 1221; x_wconf 92'>laurier,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_19' title='bbox 906 1111 1810 1219; x_wconf 90'>kruidnagel,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_20' title='bbox 1927 1109 2490 1219; x_wconf 90'>kerrie,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_21' title='bbox 2615 1110 2929 1195; x_wconf 91'>zout</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_carea' id='block_1_4' title="bbox 147 1383 3706 2731">
|
||||||
|
<p class='ocr_par' id='par_1_8' lang='eng' title="bbox 147 1383 3706 2731">
|
||||||
|
<span class='ocr_line' id='line_1_8' title="bbox 149 1383 3612 1474; baseline -0.003 -2; x_size 107.01524; x_descenders 25.01524; x_ascenders 21">
|
||||||
|
<span class='ocrx_word' id='word_1_22' title='bbox 149 1395 303 1474; x_wconf 93'>De</span>
|
||||||
|
<span class='ocrx_word' id='word_1_23' title='bbox 411 1390 902 1473; x_wconf 80'>linzgen</span>
|
||||||
|
<span class='ocrx_word' id='word_1_24' title='bbox 996 1409 1497 1470; x_wconf 89'>wassen</span>
|
||||||
|
<span class='ocrx_word' id='word_1_25' title='bbox 1595 1409 1752 1470; x_wconf 88'>en</span>
|
||||||
|
<span class='ocrx_word' id='word_1_26' title='bbox 1853 1385 2166 1467; x_wconf 75'>in-l</span>
|
||||||
|
<span class='ocrx_word' id='word_1_27' title='bbox 2275 1383 2684 1466; x_wconf 93'>liter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_28' title='bbox 2775 1383 3278 1464; x_wconf 91'>kokend</span>
|
||||||
|
<span class='ocrx_word' id='word_1_29' title='bbox 3368 1401 3612 1462; x_wconf 90'>wa-</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_9' title="bbox 157 1516 3520 1632; baseline -0.004 -21; x_size 110; x_descenders 25; x_ascenders 25">
|
||||||
|
<span class='ocrx_word' id='word_1_30' title='bbox 157 1531 394 1611; x_wconf 94'>ter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_31' title='bbox 495 1527 558 1609; x_wconf 93'>1</span>
|
||||||
|
<span class='ocrx_word' id='word_1_32' title='bbox 658 1530 903 1632; x_wconf 84'>dag</span>
|
||||||
|
<span class='ocrx_word' id='word_1_33' title='bbox 1002 1527 1415 1609; x_wconf 90'>laten</span>
|
||||||
|
<span class='ocrx_word' id='word_1_34' title='bbox 1505 1525 1979 1611; x_wconf 58'>weken,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_35' title='bbox 2103 1521 2167 1603; x_wconf 96'>2</span>
|
||||||
|
<span class='ocrx_word' id='word_1_36' title='bbox 2275 1518 2683 1603; x_wconf 83'>liter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_37' title='bbox 2777 1519 3194 1601; x_wconf 96'>water</span>
|
||||||
|
<span class='ocrx_word' id='word_1_38' title='bbox 3286 1516 3520 1599; x_wconf 89'>bij</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_10' title="bbox 152 1651 3616 1767; baseline -0.004 -21; x_size 110; x_descenders 27; x_ascenders 23">
|
||||||
|
<span class='ocrx_word' id='word_1_39' title='bbox 152 1668 302 1747; x_wconf 92'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_40' title='bbox 407 1662 905 1747; x_wconf 91'>linzen</span>
|
||||||
|
<span class='ocrx_word' id='word_1_41' title='bbox 996 1682 1559 1767; x_wconf 92'>voegen,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_42' title='bbox 1680 1659 2176 1741; x_wconf 96'>zonder</span>
|
||||||
|
<span class='ocrx_word' id='word_1_43' title='bbox 2267 1655 2506 1737; x_wconf 96'>het</span>
|
||||||
|
<span class='ocrx_word' id='word_1_44' title='bbox 2606 1655 3023 1737; x_wconf 92'>water</span>
|
||||||
|
<span class='ocrx_word' id='word_1_45' title='bbox 3116 1651 3616 1735; x_wconf 91'>waarin</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_11' title="bbox 153 1782 3704 1905; baseline -0.004 -20; x_size 117; x_descenders 23; x_ascenders 34">
|
||||||
|
<span class='ocrx_word' id='word_1_46' title='bbox 153 1824 305 1885; x_wconf 91'>ze</span>
|
||||||
|
<span class='ocrx_word' id='word_1_47' title='bbox 407 1800 979 1905; x_wconf 85'>geweekt</span>
|
||||||
|
<span class='ocrx_word' id='word_1_48' title='bbox 1089 1797 1412 1900; x_wconf 96'>zijn</span>
|
||||||
|
<span class='ocrx_word' id='word_1_49' title='bbox 1510 1796 1672 1878; x_wconf 96'>af</span>
|
||||||
|
<span class='ocrx_word' id='word_1_50' title='bbox 1770 1782 1914 1876; x_wconf 93'>te</span>
|
||||||
|
<span class='ocrx_word' id='word_1_51' title='bbox 2019 1792 2576 1899; x_wconf 54'>gieten,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_52' title='bbox 2690 1790 2850 1874; x_wconf 93'>De</span>
|
||||||
|
<span class='ocrx_word' id='word_1_53' title='bbox 2948 1791 3357 1872; x_wconf 89'>helft</span>
|
||||||
|
<span class='ocrx_word' id='word_1_54' title='bbox 3452 1811 3704 1873; x_wconf 96'>van</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_12' title="bbox 151 1928 3593 2035; baseline -0.003 -13; x_size 109; x_descenders 25; x_ascenders 25">
|
||||||
|
<span class='ocrx_word' id='word_1_55' title='bbox 151 1942 305 2024; x_wconf 92'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_56' title='bbox 403 1940 735 2021; x_wconf 89'>uien</span>
|
||||||
|
<span class='ocrx_word' id='word_1_57' title='bbox 829 1938 1330 2022; x_wconf 91'>bakken</span>
|
||||||
|
<span class='ocrx_word' id='word_1_58' title='bbox 1419 1939 1659 2018; x_wconf 92'>met</span>
|
||||||
|
<span class='ocrx_word' id='word_1_59' title='bbox 1770 1933 2346 2016; x_wconf 91'>laurier</span>
|
||||||
|
<span class='ocrx_word' id='word_1_60' title='bbox 2447 1953 2603 2014; x_wconf 91'>en</span>
|
||||||
|
<span class='ocrx_word' id='word_1_61' title='bbox 2691 1928 3593 2035; x_wconf 63'>Kruidnagel.</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_13' title="bbox 151 2067 3451 2180; baseline -0.004 -21; x_size 108; x_descenders 25; x_ascenders 24">
|
||||||
|
<span class='ocrx_word' id='word_1_62' title='bbox 151 2076 473 2159; x_wconf 92'>Alle</span>
|
||||||
|
<span class='ocrx_word' id='word_1_63' title='bbox 569 2076 965 2180; x_wconf 88'>uien,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_64' title='bbox 1079 2072 1578 2156; x_wconf 90'>kerrie</span>
|
||||||
|
<span class='ocrx_word' id='word_1_65' title='bbox 1685 2092 1837 2153; x_wconf 93'>en</span>
|
||||||
|
<span class='ocrx_word' id='word_1_66' title='bbox 1938 2072 2254 2153; x_wconf 81'>zgout</span>
|
||||||
|
<span class='ocrx_word' id='word_1_67' title='bbox 2355 2068 2593 2172; x_wconf 43'>bij</span>
|
||||||
|
<span class='ocrx_word' id='word_1_68' title='bbox 2696 2071 2850 2150; x_wconf 91'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_69' title='bbox 2957 2067 3451 2150; x_wconf 85'>linzen</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_14' title="bbox 147 2205 3614 2318; baseline -0.003 -23; x_size 106; x_descenders 22; x_ascenders 25">
|
||||||
|
<span class='ocrx_word' id='word_1_70' title='bbox 147 2234 711 2318; x_wconf 88'>voegen,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_71' title='bbox 826 2210 1234 2295; x_wconf 95'>Alles</span>
|
||||||
|
<span class='ocrx_word' id='word_1_72' title='bbox 1342 2232 1588 2295; x_wconf 95'>aan</span>
|
||||||
|
<span class='ocrx_word' id='word_1_73' title='bbox 1679 2212 1831 2291; x_wconf 96'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_74' title='bbox 1928 2208 2263 2290; x_wconf 93'>kook</span>
|
||||||
|
<span class='ocrx_word' id='word_1_75' title='bbox 2355 2206 3000 2308; x_wconf 54'>brengen,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_76' title='bbox 3117 2205 3367 2285; x_wconf 95'>Van</span>
|
||||||
|
<span class='ocrx_word' id='word_1_77' title='bbox 3462 2206 3614 2287; x_wconf 95'>de</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_15' title="bbox 152 2341 3706 2447; baseline -0.003 -18; x_size 107; x_descenders 24; x_ascenders 24">
|
||||||
|
<span class='ocrx_word' id='word_1_78' title='bbox 152 2352 565 2434; x_wconf 91'>bloem</span>
|
||||||
|
<span class='ocrx_word' id='word_1_79' title='bbox 655 2351 896 2431; x_wconf 92'>met</span>
|
||||||
|
<span class='ocrx_word' id='word_1_80' title='bbox 997 2349 1669 2431; x_wconf 90'>boter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_81' title='bbox 1514 2337 1683 2455; x_wconf 91'>en</span>
|
||||||
|
<span class='ocrx_word' id='word_1_82' title='bbox 1757 2343 2095 2426; x_wconf 88'>melk</span>
|
||||||
|
<span class='ocrx_word' id='word_1_83' title='bbox 2193 2364 2432 2427; x_wconf 93'>een</span>
|
||||||
|
<span class='ocrx_word' id='word_1_84' title='bbox 2527 2341 2935 2447; x_wconf 90'>papje</span>
|
||||||
|
<span class='ocrx_word' id='word_1_85' title='bbox 3029 2341 3453 2422; x_wconf 96'>maken</span>
|
||||||
|
<span class='ocrx_word' id='word_1_86' title='bbox 3549 2362 3706 2422; x_wconf 95'>en</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_16' title="bbox 149 2477 3619 2586; baseline -0.003 -16; x_size 107; x_descenders 24; x_ascenders 23">
|
||||||
|
<span class='ocrx_word' id='word_1_87' title='bbox 149 2489 650 2571; x_wconf 86'>verder</span>
|
||||||
|
<span class='ocrx_word' id='word_1_88' title='bbox 750 2486 1330 2570; x_wconf 90'>afmaken</span>
|
||||||
|
<span class='ocrx_word' id='word_1_89' title='bbox 1420 2485 1660 2567; x_wconf 96'>met</span>
|
||||||
|
<span class='ocrx_word' id='word_1_90' title='bbox 1765 2485 1917 2566; x_wconf 93'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_91' title='bbox 2025 2505 2409 2586; x_wconf 86'>soep,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_92' title='bbox 2521 2479 2764 2564; x_wconf 96'>Als</span>
|
||||||
|
<span class='ocrx_word' id='word_1_93' title='bbox 2868 2480 3021 2561; x_wconf 92'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_94' title='bbox 3127 2477 3619 2562; x_wconf 91'>linzen</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_17' title="bbox 155 2619 2412 2731; baseline -0.005 -20; x_size 98; x_descenders 15; x_ascenders 23">
|
||||||
|
<span class='ocrx_word' id='word_1_95' title='bbox 155 2647 481 2731; x_wconf 35'>gfgaar</span>
|
||||||
|
<span class='ocrx_word' id='word_1_96' title='bbox 581 2624 909 2728; x_wconf 87'>Zijn</span>
|
||||||
|
<span class='ocrx_word' id='word_1_97' title='bbox 1005 2623 1153 2707; x_wconf 95'>is</span>
|
||||||
|
<span class='ocrx_word' id='word_1_98' title='bbox 1255 2624 1409 2706; x_wconf 93'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_99' title='bbox 1522 2642 1840 2726; x_wconf 91'>soep</span>
|
||||||
|
<span class='ocrx_word' id='word_1_100' title='bbox 1929 2619 2412 2709; x_wconf 89'>klaar.</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
+24
@@ -0,0 +1,24 @@
|
|||||||
|
i a la Waterman
|
||||||
|
|
||||||
|
4 ons linzen
|
||||||
|
|
||||||
|
3 liter water
|
||||||
|
|
||||||
|
3 uien
|
||||||
|
|
||||||
|
bloem, boter
|
||||||
|
|
||||||
|
2 kopjes melk
|
||||||
|
|
||||||
|
laurier, kruidnagel, kerrie, zout
|
||||||
|
|
||||||
|
De linzgen wassen en in-l liter kokend wa-
|
||||||
|
ter 1 dag laten weken, 2 liter water bij
|
||||||
|
de linzen voegen, zonder het water waarin
|
||||||
|
ze geweekt zijn af te gieten, De helft van
|
||||||
|
de uien bakken met laurier en Kruidnagel.
|
||||||
|
Alle uien, kerrie en zgout bij de linzen
|
||||||
|
voegen, Alles aan de kook brengen, Van de
|
||||||
|
bloem met boter en melk een papje maken en
|
||||||
|
verder afmaken met de soep, Als de linzen
|
||||||
|
gfgaar Zijn is de soep klaar.
|
||||||
BIN
Binary file not shown.
+89
@@ -0,0 +1,89 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||||
|
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||||
|
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||||
|
<head>
|
||||||
|
<title></title>
|
||||||
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||||
|
<meta name='ocr-system' content='tesseract 5.3.2' />
|
||||||
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.vh2to5iv/000001_ocr.png"; bbox 0 0 640 682; ppageno 0; scan_res 230 230'>
|
||||||
|
<div class='ocr_carea' id='block_1_1' title="bbox 365 15 429 29">
|
||||||
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 365 15 429 29">
|
||||||
|
<span class='ocr_line' id='line_1_1' title="bbox 365 15 429 29; baseline 0 0; x_size 19.238094; x_descenders 5.2380953; x_ascenders 4">
|
||||||
|
<span class='ocrx_word' id='word_1_1' title='bbox 365 15 429 29; x_wconf 92'>Tarnose</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_photo' id='block_1_2' title="bbox 255 38 616 72"></div>
|
||||||
|
<div class='ocr_photo' id='block_1_3' title="bbox 273 38 349 57"></div>
|
||||||
|
<div class='ocr_carea' id='block_1_4' title="bbox 186 20 244 49">
|
||||||
|
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 186 20 345 49">
|
||||||
|
<span class='ocr_line' id='line_1_2' title="bbox 186 20 244 49; baseline -0.006 -10; x_size 20.454546; x_descenders 5.4545455; x_ascenders 5">
|
||||||
|
<span class='ocrx_word' id='word_1_2' title='bbox 186 20 244 49; x_wconf 92'>Bokale</span>
|
||||||
|
<span class='ocrx_word' id='word_1_3' title='bbox 299 28 345 46; x_wconf 42'>oa</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_photo' id='block_1_5' title="bbox 64 38 186 72"></div>
|
||||||
|
<div class='ocr_carea' id='block_1_6' title="bbox 537 276 612 291">
|
||||||
|
<p class='ocr_par' id='par_1_3' lang='eng' title="bbox 537 276 612 291">
|
||||||
|
<span class='ocr_line' id='line_1_3' title="bbox 537 276 612 291; baseline 0 0; x_size 20.238094; x_descenders 5.2380953; x_ascenders 5">
|
||||||
|
<span class='ocrx_word' id='word_1_4' title='bbox 537 276 612 291; x_wconf 92'>Lehuntze</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_photo' id='block_1_7' title="bbox 20 269 616 314"></div>
|
||||||
|
<div class='ocr_carea' id='block_1_8' title="bbox 480 323 550 341">
|
||||||
|
<p class='ocr_par' id='par_1_4' lang='eng' title="bbox 480 323 550 341">
|
||||||
|
<span class='ocr_line' id='line_1_4' title="bbox 480 323 550 341; baseline 0 -4; x_size 18; x_descenders 4; x_ascenders 4">
|
||||||
|
<span class='ocrx_word' id='word_1_5' title='bbox 480 323 550 341; x_wconf 91'>Mugerre</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_photo' id='block_1_9' title="bbox 44 574 197 623"></div>
|
||||||
|
<div class='ocr_carea' id='block_1_10' title="bbox 204 542 603 574">
|
||||||
|
<p class='ocr_par' id='par_1_5' lang='eng' title="bbox 204 542 603 574">
|
||||||
|
<span class='ocr_line' id='line_1_5' title="bbox 204 542 603 574; baseline 0.018 -11; x_size 23.26087; x_descenders 5.2608695; x_ascenders 7">
|
||||||
|
<span class='ocrx_word' id='word_1_6' title='bbox 204 542 295 574; x_wconf 91'>Milafranga</span>
|
||||||
|
<span class='ocrx_word' id='word_1_7' title='bbox 439 552 603 569; x_wconf 90'>Komunikabideak</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_carea' id='block_1_11' title="bbox 220 584 616 619">
|
||||||
|
<p class='ocr_par' id='par_1_6' lang='eng' title="bbox 220 584 616 625">
|
||||||
|
<span class='ocr_line' id='line_1_6' title="bbox 220 584 616 619; baseline -0.005 -1; x_size 43.666668; x_descenders 10.916667; x_ascenders 10.916667">
|
||||||
|
<span class='ocrx_word' id='word_1_8' title='bbox 220 585 404 619; x_wconf 2'>BAIONA</span>
|
||||||
|
<span class='ocrx_word' id='word_1_9' title='bbox 468 584 576 625; x_wconf 0'> zeiteninsiie</span>
|
||||||
|
<span class='ocrx_word' id='word_1_10' title='bbox 585 588 616 610; x_wconf 86'>—</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_carea' id='block_1_12' title="bbox 393 623 598 634">
|
||||||
|
<p class='ocr_par' id='par_1_7' lang='eng' title="bbox 393 623 598 634">
|
||||||
|
<span class='ocr_line' id='line_1_7' title="bbox 393 623 598 634; baseline 0 0; x_size 22.75; x_descenders 5.5; x_ascenders 5.75">
|
||||||
|
<span class='ocrx_word' id='word_1_11' title='bbox 393 629 396 631; x_wconf 54'>7</span>
|
||||||
|
<span class='ocrx_word' id='word_1_12' title='bbox 470 623 539 634; x_wconf 24'>Trenbideak</span>
|
||||||
|
<span class='ocrx_word' id='word_1_13' title='bbox 550 628 598 630; x_wconf 24'>-----</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_carea' id='block_1_13' title="bbox 81 633 613 667">
|
||||||
|
<p class='ocr_par' id='par_1_8' lang='eng' title="bbox 81 633 613 667">
|
||||||
|
<span class='ocr_line' id='line_1_8' title="bbox 81 633 613 667; baseline -0.002 -13; x_size 21; x_descenders 4; x_ascenders 7">
|
||||||
|
<span class='ocrx_word' id='word_1_14' title='bbox 81 637 111 653; x_wconf 0'>t\</span>
|
||||||
|
<span class='ocrx_word' id='word_1_15' title='bbox 123 633 201 663; x_wconf 22'>Basusarri</span>
|
||||||
|
<span class='ocrx_word' id='word_1_16' title='bbox 214 633 222 663; x_wconf 0'>—</span>
|
||||||
|
<span class='ocrx_word' id='word_1_17' title='bbox 230 647 325 656; x_wconf 0'>spmeans:20141004</span>
|
||||||
|
<span class='ocrx_word' id='word_1_18' title='bbox 373 633 415 667; x_wconf 42'>ae:</span>
|
||||||
|
<span class='ocrx_word' id='word_1_19' title='bbox 441 650 444 653; x_wconf 25'>.</span>
|
||||||
|
<span class='ocrx_word' id='word_1_20' title='bbox 521 649 544 657; x_wconf 17'>_</span>
|
||||||
|
<span class='ocrx_word' id='word_1_21' title='bbox 595 650 613 659; x_wconf 7'>~</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
+15
@@ -0,0 +1,15 @@
|
|||||||
|
Tarnose
|
||||||
|
|
||||||
|
Bokale oa
|
||||||
|
|
||||||
|
Lehuntze
|
||||||
|
|
||||||
|
Mugerre
|
||||||
|
|
||||||
|
Milafranga Komunikabideak
|
||||||
|
|
||||||
|
BAIONA zeiteninsiie —
|
||||||
|
|
||||||
|
7 Trenbideak -----
|
||||||
|
|
||||||
|
t\ Basusarri — spmeans:20141004 ae: . _ ~
|
||||||
BIN
Binary file not shown.
+27
@@ -0,0 +1,27 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||||
|
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||||
|
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||||
|
<head>
|
||||||
|
<title></title>
|
||||||
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||||
|
<meta name='ocr-system' content='tesseract 5.3.2' />
|
||||||
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.vh2to5iv/000002_ocr.png"; bbox 0 0 400 50; ppageno 0; scan_res 200 200'>
|
||||||
|
<div class='ocr_carea' id='block_1_1' title="bbox 54 16 344 33">
|
||||||
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 54 16 344 33">
|
||||||
|
<span class='ocr_line' id='line_1_1' title="bbox 54 16 344 33; baseline 0 -4; x_size 25.75; x_descenders 5.5; x_ascenders 6.75">
|
||||||
|
<span class='ocrx_word' id='word_1_1' title='bbox 54 16 110 29; x_wconf 84'>Covfefe</span>
|
||||||
|
<span class='ocrx_word' id='word_1_2' title='bbox 115 17 127 29; x_wconf 95'>is</span>
|
||||||
|
<span class='ocrx_word' id='word_1_3' title='bbox 132 20 140 29; x_wconf 95'>a</span>
|
||||||
|
<span class='ocrx_word' id='word_1_4' title='bbox 145 16 212 33; x_wconf 85'>perfectly</span>
|
||||||
|
<span class='ocrx_word' id='word_1_5' title='bbox 217 16 296 29; x_wconf 85'>cromulent</span>
|
||||||
|
<span class='ocrx_word' id='word_1_6' title='bbox 300 16 344 29; x_wconf 95'>word.</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
+1
@@ -0,0 +1 @@
|
|||||||
|
Covfefe is a perfectly cromulent word.
|
||||||
BIN
Binary file not shown.
+177
@@ -0,0 +1,177 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8"?>
|
||||||
|
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||||
|
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||||
|
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||||
|
<head>
|
||||||
|
<title></title>
|
||||||
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||||
|
<meta name='ocr-system' content='tesseract 5.3.2' />
|
||||||
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
|
</head>
|
||||||
|
<body>
|
||||||
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.vh2to5iv/000003_ocr.png"; bbox 0 0 4000 2864; ppageno 0; scan_res 1440 1440'>
|
||||||
|
<div class='ocr_carea' id='block_1_1' title="bbox 142 146 2173 258">
|
||||||
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 142 146 2173 258">
|
||||||
|
<span class='ocr_line' id='line_1_1' title="bbox 142 146 2173 258; baseline -0.004 -19; x_size 113; x_descenders 23; x_ascenders 32">
|
||||||
|
<span class='ocrx_word' id='word_1_1' title='bbox 142 154 982 258; x_wconf 90'>Linzensoep</span>
|
||||||
|
<span class='ocrx_word' id='word_1_2' title='bbox 1080 146 1152 237; x_wconf 96'>a</span>
|
||||||
|
<span class='ocrx_word' id='word_1_3' title='bbox 1254 153 1406 235; x_wconf 96'>la</span>
|
||||||
|
<span class='ocrx_word' id='word_1_4' title='bbox 1500 153 2173 235; x_wconf 96'>Waterman</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_separator' id='block_1_2' title="bbox 136 253 2180 275"></div>
|
||||||
|
<div class='ocr_carea' id='block_1_3' title="bbox 145 414 2929 1221">
|
||||||
|
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 145 414 1154 516">
|
||||||
|
<span class='ocr_line' id='line_1_2' title="bbox 145 414 1154 516; baseline -0.005 -4; x_size 122.38636; x_descenders 24.386362; x_ascenders 40">
|
||||||
|
<span class='ocrx_word' id='word_1_5' title='bbox 145 414 211 512; x_wconf 60'>4</span>
|
||||||
|
<span class='ocrx_word' id='word_1_6' title='bbox 318 453 551 516; x_wconf 93'>ons</span>
|
||||||
|
<span class='ocrx_word' id='word_1_7' title='bbox 660 430 1154 512; x_wconf 91'>linzen</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_3' lang='eng' title="bbox 146 568 1239 667">
|
||||||
|
<span class='ocr_line' id='line_1_3' title="bbox 146 568 1239 667; baseline -0.004 -16; x_size 99; x_descenders 17; x_ascenders 24">
|
||||||
|
<span class='ocrx_word' id='word_1_8' title='bbox 146 569 209 667; x_wconf 94'>3</span>
|
||||||
|
<span class='ocrx_word' id='word_1_9' title='bbox 323 568 729 652; x_wconf 83'>liter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_10' title='bbox 821 569 1239 650; x_wconf 96'>water</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_4' lang='eng' title="bbox 148 705 645 804">
|
||||||
|
<span class='ocr_line' id='line_1_4' title="bbox 148 705 645 804; baseline -0.016 -14; x_size 103; x_descenders 19; x_ascenders 27">
|
||||||
|
<span class='ocrx_word' id='word_1_11' title='bbox 148 706 211 804; x_wconf 88'>3</span>
|
||||||
|
<span class='ocrx_word' id='word_1_12' title='bbox 311 705 645 789; x_wconf 52'>uien</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_5' lang='eng' title="bbox 147 832 1154 950">
|
||||||
|
<span class='ocr_line' id='line_1_5' title="bbox 147 832 1154 950; baseline -0.004 -27; x_size 118; x_descenders 28; x_ascenders 30">
|
||||||
|
<span class='ocrx_word' id='word_1_13' title='bbox 147 832 623 950; x_wconf 91'>bloem,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_14' title='bbox 737 843 1154 924; x_wconf 91'>boter</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_6' lang='eng' title="bbox 148 979 1241 1087">
|
||||||
|
<span class='ocr_line' id='line_1_6' title="bbox 148 979 1241 1087; baseline -0.005 -21; x_size 107; x_descenders 24; x_ascenders 24">
|
||||||
|
<span class='ocrx_word' id='word_1_15' title='bbox 148 983 215 1066; x_wconf 88'>2</span>
|
||||||
|
<span class='ocrx_word' id='word_1_16' title='bbox 312 983 807 1087; x_wconf 88'>kopjes</span>
|
||||||
|
<span class='ocrx_word' id='word_1_17' title='bbox 905 979 1241 1062; x_wconf 92'>melk</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
|
||||||
|
<p class='ocr_par' id='par_1_7' lang='eng' title="bbox 154 1109 2929 1221">
|
||||||
|
<span class='ocr_line' id='line_1_7' title="bbox 154 1109 2929 1221; baseline -0.003 -21; x_size 112; x_descenders 28; x_ascenders 26">
|
||||||
|
<span class='ocrx_word' id='word_1_18' title='bbox 154 1117 791 1221; x_wconf 92'>laurier,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_19' title='bbox 906 1111 1810 1219; x_wconf 90'>kruidnagel,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_20' title='bbox 1927 1109 2490 1219; x_wconf 90'>kerrie,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_21' title='bbox 2615 1110 2929 1195; x_wconf 91'>zout</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
<div class='ocr_carea' id='block_1_4' title="bbox 147 1383 3706 2731">
|
||||||
|
<p class='ocr_par' id='par_1_8' lang='eng' title="bbox 147 1383 3706 2731">
|
||||||
|
<span class='ocr_line' id='line_1_8' title="bbox 149 1383 3612 1474; baseline -0.003 -2; x_size 107.01524; x_descenders 25.01524; x_ascenders 21">
|
||||||
|
<span class='ocrx_word' id='word_1_22' title='bbox 149 1395 303 1474; x_wconf 93'>De</span>
|
||||||
|
<span class='ocrx_word' id='word_1_23' title='bbox 411 1390 902 1473; x_wconf 80'>linzgen</span>
|
||||||
|
<span class='ocrx_word' id='word_1_24' title='bbox 996 1409 1497 1470; x_wconf 89'>wassen</span>
|
||||||
|
<span class='ocrx_word' id='word_1_25' title='bbox 1595 1409 1752 1470; x_wconf 88'>en</span>
|
||||||
|
<span class='ocrx_word' id='word_1_26' title='bbox 1853 1385 2166 1467; x_wconf 75'>in-l</span>
|
||||||
|
<span class='ocrx_word' id='word_1_27' title='bbox 2275 1383 2684 1466; x_wconf 93'>liter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_28' title='bbox 2775 1383 3278 1464; x_wconf 91'>kokend</span>
|
||||||
|
<span class='ocrx_word' id='word_1_29' title='bbox 3368 1401 3612 1462; x_wconf 90'>wa-</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_9' title="bbox 157 1516 3521 1632; baseline -0.004 -21; x_size 110; x_descenders 25; x_ascenders 25">
|
||||||
|
<span class='ocrx_word' id='word_1_30' title='bbox 157 1531 394 1611; x_wconf 93'>ter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_31' title='bbox 495 1527 558 1609; x_wconf 94'>1</span>
|
||||||
|
<span class='ocrx_word' id='word_1_32' title='bbox 658 1530 903 1632; x_wconf 75'>dag</span>
|
||||||
|
<span class='ocrx_word' id='word_1_33' title='bbox 1002 1527 1415 1609; x_wconf 85'>laten</span>
|
||||||
|
<span class='ocrx_word' id='word_1_34' title='bbox 1505 1525 1979 1611; x_wconf 86'>weken,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_35' title='bbox 2103 1521 2167 1603; x_wconf 96'>2</span>
|
||||||
|
<span class='ocrx_word' id='word_1_36' title='bbox 2275 1518 2683 1603; x_wconf 93'>liter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_37' title='bbox 2777 1519 3194 1601; x_wconf 96'>water</span>
|
||||||
|
<span class='ocrx_word' id='word_1_38' title='bbox 3286 1516 3521 1620; x_wconf 93'>bij</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_10' title="bbox 152 1651 3616 1767; baseline -0.004 -21; x_size 110; x_descenders 27; x_ascenders 23">
|
||||||
|
<span class='ocrx_word' id='word_1_39' title='bbox 152 1668 302 1747; x_wconf 92'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_40' title='bbox 407 1662 905 1747; x_wconf 91'>linzen</span>
|
||||||
|
<span class='ocrx_word' id='word_1_41' title='bbox 996 1682 1559 1767; x_wconf 92'>voegen,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_42' title='bbox 1680 1659 2176 1741; x_wconf 96'>zonder</span>
|
||||||
|
<span class='ocrx_word' id='word_1_43' title='bbox 2267 1655 2506 1737; x_wconf 96'>het</span>
|
||||||
|
<span class='ocrx_word' id='word_1_44' title='bbox 2606 1655 3023 1737; x_wconf 92'>water</span>
|
||||||
|
<span class='ocrx_word' id='word_1_45' title='bbox 3116 1651 3616 1735; x_wconf 91'>waarin</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_11' title="bbox 153 1782 3704 1905; baseline -0.004 -20; x_size 117; x_descenders 23; x_ascenders 34">
|
||||||
|
<span class='ocrx_word' id='word_1_46' title='bbox 153 1824 305 1885; x_wconf 91'>ze</span>
|
||||||
|
<span class='ocrx_word' id='word_1_47' title='bbox 407 1800 979 1905; x_wconf 85'>geweekt</span>
|
||||||
|
<span class='ocrx_word' id='word_1_48' title='bbox 1089 1797 1412 1900; x_wconf 96'>zijn</span>
|
||||||
|
<span class='ocrx_word' id='word_1_49' title='bbox 1510 1796 1672 1878; x_wconf 96'>af</span>
|
||||||
|
<span class='ocrx_word' id='word_1_50' title='bbox 1770 1782 1914 1876; x_wconf 93'>te</span>
|
||||||
|
<span class='ocrx_word' id='word_1_51' title='bbox 2019 1792 2576 1899; x_wconf 54'>gieten,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_52' title='bbox 2690 1790 2850 1874; x_wconf 93'>De</span>
|
||||||
|
<span class='ocrx_word' id='word_1_53' title='bbox 2948 1791 3357 1872; x_wconf 89'>helft</span>
|
||||||
|
<span class='ocrx_word' id='word_1_54' title='bbox 3452 1811 3704 1873; x_wconf 96'>van</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_12' title="bbox 151 1928 3593 2035; baseline -0.003 -13; x_size 109; x_descenders 25; x_ascenders 25">
|
||||||
|
<span class='ocrx_word' id='word_1_55' title='bbox 151 1942 305 2024; x_wconf 92'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_56' title='bbox 403 1940 735 2021; x_wconf 89'>uien</span>
|
||||||
|
<span class='ocrx_word' id='word_1_57' title='bbox 829 1938 1330 2022; x_wconf 91'>bakken</span>
|
||||||
|
<span class='ocrx_word' id='word_1_58' title='bbox 1419 1939 1659 2018; x_wconf 92'>met</span>
|
||||||
|
<span class='ocrx_word' id='word_1_59' title='bbox 1770 1933 2346 2016; x_wconf 91'>laurier</span>
|
||||||
|
<span class='ocrx_word' id='word_1_60' title='bbox 2447 1953 2603 2014; x_wconf 91'>en</span>
|
||||||
|
<span class='ocrx_word' id='word_1_61' title='bbox 2691 1928 3593 2035; x_wconf 63'>Kruidnagel.</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_13' title="bbox 151 2067 3451 2180; baseline -0.004 -21; x_size 108; x_descenders 25; x_ascenders 24">
|
||||||
|
<span class='ocrx_word' id='word_1_62' title='bbox 151 2076 473 2159; x_wconf 92'>Alle</span>
|
||||||
|
<span class='ocrx_word' id='word_1_63' title='bbox 569 2076 965 2180; x_wconf 88'>uien,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_64' title='bbox 1079 2072 1578 2156; x_wconf 90'>kerrie</span>
|
||||||
|
<span class='ocrx_word' id='word_1_65' title='bbox 1685 2092 1837 2153; x_wconf 93'>en</span>
|
||||||
|
<span class='ocrx_word' id='word_1_66' title='bbox 1938 2072 2254 2153; x_wconf 81'>zgout</span>
|
||||||
|
<span class='ocrx_word' id='word_1_67' title='bbox 2355 2068 2593 2172; x_wconf 43'>bij</span>
|
||||||
|
<span class='ocrx_word' id='word_1_68' title='bbox 2696 2071 2850 2150; x_wconf 91'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_69' title='bbox 2957 2067 3451 2150; x_wconf 85'>linzen</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_14' title="bbox 147 2205 3614 2318; baseline -0.003 -23; x_size 106; x_descenders 22; x_ascenders 25">
|
||||||
|
<span class='ocrx_word' id='word_1_70' title='bbox 147 2234 711 2318; x_wconf 88'>voegen,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_71' title='bbox 826 2210 1234 2295; x_wconf 95'>Alles</span>
|
||||||
|
<span class='ocrx_word' id='word_1_72' title='bbox 1342 2232 1588 2295; x_wconf 95'>aan</span>
|
||||||
|
<span class='ocrx_word' id='word_1_73' title='bbox 1679 2212 1831 2291; x_wconf 96'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_74' title='bbox 1928 2208 2263 2290; x_wconf 93'>kook</span>
|
||||||
|
<span class='ocrx_word' id='word_1_75' title='bbox 2355 2206 3000 2308; x_wconf 54'>brengen,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_76' title='bbox 3117 2205 3367 2285; x_wconf 95'>Van</span>
|
||||||
|
<span class='ocrx_word' id='word_1_77' title='bbox 3462 2206 3614 2287; x_wconf 95'>de</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_15' title="bbox 152 2341 3706 2447; baseline -0.003 -18; x_size 107; x_descenders 24; x_ascenders 24">
|
||||||
|
<span class='ocrx_word' id='word_1_78' title='bbox 152 2352 565 2434; x_wconf 91'>bloem</span>
|
||||||
|
<span class='ocrx_word' id='word_1_79' title='bbox 655 2351 896 2431; x_wconf 92'>met</span>
|
||||||
|
<span class='ocrx_word' id='word_1_80' title='bbox 997 2349 1669 2431; x_wconf 90'>boter</span>
|
||||||
|
<span class='ocrx_word' id='word_1_81' title='bbox 1514 2337 1683 2455; x_wconf 91'>en</span>
|
||||||
|
<span class='ocrx_word' id='word_1_82' title='bbox 1757 2343 2095 2426; x_wconf 88'>melk</span>
|
||||||
|
<span class='ocrx_word' id='word_1_83' title='bbox 2193 2364 2432 2427; x_wconf 93'>een</span>
|
||||||
|
<span class='ocrx_word' id='word_1_84' title='bbox 2527 2341 2935 2447; x_wconf 90'>papje</span>
|
||||||
|
<span class='ocrx_word' id='word_1_85' title='bbox 3029 2341 3453 2422; x_wconf 96'>maken</span>
|
||||||
|
<span class='ocrx_word' id='word_1_86' title='bbox 3549 2362 3706 2422; x_wconf 95'>en</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_16' title="bbox 149 2477 3619 2586; baseline -0.003 -16; x_size 107; x_descenders 24; x_ascenders 23">
|
||||||
|
<span class='ocrx_word' id='word_1_87' title='bbox 149 2489 650 2571; x_wconf 86'>verder</span>
|
||||||
|
<span class='ocrx_word' id='word_1_88' title='bbox 750 2486 1330 2570; x_wconf 90'>afmaken</span>
|
||||||
|
<span class='ocrx_word' id='word_1_89' title='bbox 1420 2485 1660 2567; x_wconf 96'>met</span>
|
||||||
|
<span class='ocrx_word' id='word_1_90' title='bbox 1765 2485 1917 2566; x_wconf 93'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_91' title='bbox 2025 2505 2409 2586; x_wconf 86'>soep,</span>
|
||||||
|
<span class='ocrx_word' id='word_1_92' title='bbox 2521 2479 2764 2564; x_wconf 96'>Als</span>
|
||||||
|
<span class='ocrx_word' id='word_1_93' title='bbox 2868 2480 3021 2561; x_wconf 92'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_94' title='bbox 3127 2477 3619 2562; x_wconf 91'>linzen</span>
|
||||||
|
</span>
|
||||||
|
<span class='ocr_line' id='line_1_17' title="bbox 155 2619 2412 2731; baseline -0.005 -20; x_size 98; x_descenders 15; x_ascenders 23">
|
||||||
|
<span class='ocrx_word' id='word_1_95' title='bbox 155 2647 481 2731; x_wconf 35'>gfgaar</span>
|
||||||
|
<span class='ocrx_word' id='word_1_96' title='bbox 581 2624 909 2728; x_wconf 87'>Zijn</span>
|
||||||
|
<span class='ocrx_word' id='word_1_97' title='bbox 1005 2623 1153 2707; x_wconf 95'>is</span>
|
||||||
|
<span class='ocrx_word' id='word_1_98' title='bbox 1255 2624 1409 2706; x_wconf 93'>de</span>
|
||||||
|
<span class='ocrx_word' id='word_1_99' title='bbox 1522 2642 1840 2726; x_wconf 91'>soep</span>
|
||||||
|
<span class='ocrx_word' id='word_1_100' title='bbox 1929 2619 2412 2709; x_wconf 89'>klaar.</span>
|
||||||
|
</span>
|
||||||
|
</p>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
</body>
|
||||||
|
</html>
|
||||||
+24
@@ -0,0 +1,24 @@
|
|||||||
|
Linzensoep a la Waterman
|
||||||
|
|
||||||
|
4 ons linzen
|
||||||
|
|
||||||
|
3 liter water
|
||||||
|
|
||||||
|
3 uien
|
||||||
|
|
||||||
|
bloem, boter
|
||||||
|
|
||||||
|
2 kopjes melk
|
||||||
|
|
||||||
|
laurier, kruidnagel, kerrie, zout
|
||||||
|
|
||||||
|
De linzgen wassen en in-l liter kokend wa-
|
||||||
|
ter 1 dag laten weken, 2 liter water bij
|
||||||
|
de linzen voegen, zonder het water waarin
|
||||||
|
ze geweekt zijn af te gieten, De helft van
|
||||||
|
de uien bakken met laurier en Kruidnagel.
|
||||||
|
Alle uien, kerrie en zgout bij de linzen
|
||||||
|
voegen, Alles aan de kook brengen, Van de
|
||||||
|
bloem met boter en melk een papje maken en
|
||||||
|
verder afmaken met de soep, Als de linzen
|
||||||
|
gfgaar Zijn is de soep klaar.
|
||||||
BIN
Binary file not shown.
+1
-1
@@ -9,7 +9,7 @@
|
|||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.9_hn68nz/000001_ocr.png"; bbox 0 0 1000 800; ppageno 0; scan_res 200 200'>
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.su6l_waz/000001_ocr.png"; bbox 0 0 1000 800; ppageno 0; scan_res 200 200'>
|
||||||
<div class='ocr_photo' id='block_1_1' title="bbox 296 96 704 504"></div>
|
<div class='ocr_photo' id='block_1_1' title="bbox 296 96 704 504"></div>
|
||||||
<div class='ocr_carea' id='block_1_2' title="bbox 150 592 841 622">
|
<div class='ocr_carea' id='block_1_2' title="bbox 150 592 841 622">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 150 592 841 622">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 150 592 841 622">
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -9,7 +9,7 @@
|
|||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.q0nk4qy2/000001_ocr.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.g296cpyi/000001_ocr.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
||||||
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -9,7 +9,7 @@
|
|||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.q0nk4qy2/000002_ocr.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.91dzclia/000002_ocr.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
||||||
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -9,7 +9,7 @@
|
|||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.q0nk4qy2/000003_ocr.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.91dzclia/000003_ocr.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
||||||
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
||||||
|
|||||||
BIN
Binary file not shown.
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user