Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
250615561d | ||
|
|
a659f83d67 | ||
|
|
08f95c0b13 | ||
|
|
dbd3c93757 | ||
|
|
5d128a91d2 | ||
|
|
a1b8113d56 | ||
|
|
f052e910c9 | ||
|
|
116e2692d0 | ||
|
|
b2669c7d71 | ||
|
|
c8c53d38a3 | ||
|
|
d303b42c86 | ||
|
|
f77f701a50 | ||
|
|
1c3b7d1507 | ||
|
|
bf62562787 | ||
|
|
6c6cbfd4d6 | ||
|
|
ee5acbe94e | ||
|
|
5e478a7774 | ||
|
|
92c5200ad2 | ||
|
|
86a102f8e6 | ||
|
|
2463b91051 | ||
|
|
07f7c6b812 | ||
|
|
8138664287 | ||
|
|
120ca72393 | ||
|
|
f9b3e9a97b | ||
|
|
1e87930bbb | ||
|
|
fe4725658e | ||
|
|
9d042767cc | ||
|
|
23bc247b9c | ||
|
|
e44bf46d77 | ||
|
|
f50620c244 | ||
|
|
6f755321b8 | ||
|
|
706681deb8 | ||
|
|
c283cf0a0d | ||
|
|
0f82d7223e | ||
|
|
9a6150ae53 | ||
|
|
fec0948a13 | ||
|
|
18b59c57b4 | ||
|
|
a67a11e61c | ||
|
|
6ca4940a32 | ||
|
|
0e4cce2642 | ||
|
|
8fca0c71dc | ||
|
|
944d99bdc1 | ||
|
|
5bb6e1c5d7 | ||
|
|
8d7a8f0f98 | ||
|
|
b9dd0a5e3c | ||
|
|
6949ad2c5d | ||
|
|
b3324c3b4e | ||
|
|
b38cac6931 | ||
|
|
bb4c47e707 | ||
|
|
5e1e2497ab | ||
|
|
cd910fbf21 | ||
|
|
1225269a4b | ||
|
|
3a75b20740 | ||
|
|
d35d008806 | ||
|
|
f5662d5eb0 | ||
|
|
39010dd255 | ||
|
|
fbaad570c7 | ||
|
|
f974e3b3c1 | ||
|
|
46b49cc176 | ||
|
|
5256e74d0c | ||
|
|
621d6a0b89 | ||
|
|
08be7c8bbe | ||
|
|
980a5472b6 | ||
|
|
51c618e357 | ||
|
|
4dde3786c2 | ||
|
|
d544342602 | ||
|
|
fac91fca2a | ||
|
|
6edf756849 | ||
|
|
4fb1bb4de6 | ||
|
|
6a8eb7daaa | ||
|
|
0544d06c3d | ||
|
|
34c285c9ac | ||
|
|
2f53b27651 | ||
|
|
772677746b | ||
|
|
f0bad87ea6 | ||
|
|
44e71f8c14 | ||
|
|
964b30ca26 | ||
|
|
214a333e2d | ||
|
|
ec6401ab57 | ||
|
|
cbc5e8ce8d | ||
|
|
a1c4cfe8f1 | ||
|
|
3a721e6578 | ||
|
|
e6b716cdde | ||
|
|
02c39998b8 | ||
|
|
0774bc7f14 | ||
|
|
c6a98b3d0b | ||
|
|
981bbf1105 | ||
|
|
2b0c6cfd40 | ||
|
|
59f6bc8306 | ||
|
|
653c4ffb45 | ||
|
|
d947ca258e | ||
|
|
d5ff7f7db9 | ||
|
|
579cef3649 | ||
|
|
cb2f090c60 | ||
|
|
f3d6387bca | ||
|
|
abf9729c61 | ||
|
|
442e9c9f0d | ||
|
|
397fad249d | ||
|
|
9a3c5a3f7c | ||
|
|
950c700274 | ||
|
|
26432c38a9 | ||
|
|
28be50136c | ||
|
|
0c62f2de5d | ||
|
|
5caf654f22 | ||
|
|
205593445e | ||
|
|
f25fb8c63a | ||
|
|
99c78650b6 | ||
|
|
69355886a8 | ||
|
|
08e89e2dbe |
+22
-27
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM ubuntu:22.04 as base
|
||||
FROM ubuntu:22.04 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -9,19 +9,15 @@ RUN echo 'debconf debconf/frontend select Noninteractive' | debconf-set-selectio
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
python3 \
|
||||
libqpdf-dev \
|
||||
zlib1g \
|
||||
liblept5
|
||||
python-is-python3
|
||||
|
||||
FROM base as builder
|
||||
FROM base AS builder
|
||||
|
||||
# Note we need leptonica here to build jbig2
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential autoconf automake libtool \
|
||||
libleptonica-dev \
|
||||
zlib1g-dev \
|
||||
python3-dev \
|
||||
python3-distutils \
|
||||
libffi-dev \
|
||||
ca-certificates \
|
||||
curl \
|
||||
@@ -29,15 +25,11 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libcairo2-dev \
|
||||
pkg-config
|
||||
|
||||
# Get the latest pip (Ubuntu version doesn't support manylinux2010)
|
||||
RUN \
|
||||
curl https://bootstrap.pypa.io/get-pip.py | python3
|
||||
|
||||
# Compile and install jbig2
|
||||
# Needs libleptonica-dev, zlib1g-dev
|
||||
RUN \
|
||||
mkdir jbig2 \
|
||||
&& curl -L https://github.com/agl/jbig2enc/archive/ea6a40a.tar.gz | \
|
||||
&& curl -L https://github.com/agl/jbig2enc/archive/c0141bf.tar.gz | \
|
||||
tar xz -C jbig2 --strip-components=1 \
|
||||
&& cd jbig2 \
|
||||
&& ./autogen.sh && ./configure && make && make install \
|
||||
@@ -48,23 +40,24 @@ COPY . /app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
RUN pip3 install --no-cache-dir .[test,webservice,watcher]
|
||||
RUN curl -LsSf https://astral.sh/uv/0.4.27/install.sh | sh
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
# Instead of restarting the shell, use uv directly from its installed location.
|
||||
RUN /root/.cargo/bin/uv sync --extra test --extra webservice --extra watcher
|
||||
|
||||
FROM base
|
||||
|
||||
# For Tesseract 5
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
software-properties-common gpg-agent
|
||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr-devel
|
||||
RUN apt-get update && apt-get install -y software-properties-common
|
||||
|
||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr5
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ghostscript \
|
||||
fonts-droid-fallback \
|
||||
jbig2dec \
|
||||
img2pdf \
|
||||
libsm6 libxext6 libxrender-dev \
|
||||
pngquant \
|
||||
python-is-python3 \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-chi-sim \
|
||||
tesseract-ocr-deu \
|
||||
@@ -80,11 +73,13 @@ WORKDIR /app
|
||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
||||
COPY --from=builder /usr/local/bin/ /usr/local/bin/
|
||||
|
||||
COPY --from=builder /app/misc/webservice.py /app/
|
||||
COPY --from=builder /app/misc/watcher.py /app/
|
||||
COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
# Copy minimal project files to get the test suite.
|
||||
COPY --from=builder /app/pyproject.toml /app/README.md /app/
|
||||
COPY --from=builder /app/tests /app/tests
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
|
||||
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
||||
|
||||
+19
-34
@@ -1,7 +1,14 @@
|
||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM alpine:3.19 as base
|
||||
# Note: Alpine 3.20 builds tesseract with --enable-opencl, which is not
|
||||
# supported by anyone. OCRmyPDF is not compatible with Alpine 3.20.0
|
||||
# through 3.20.3. The Alpine issue should be fixed in 3.21.0. It is
|
||||
# not clear if 3.20.4+ will have the fix.
|
||||
# Details
|
||||
# https://gitlab.alpinelinux.org/alpine/aports/-/issues/16143
|
||||
# https://github.com/ocrmypdf/OCRmyPDF/issues/1395
|
||||
FROM alpine:3.19 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -10,40 +17,24 @@ RUN apk add --no-cache \
|
||||
python3 \
|
||||
zlib
|
||||
|
||||
FROM base as builder
|
||||
FROM base AS builder
|
||||
|
||||
RUN apk add --no-cache \
|
||||
ca-certificates \
|
||||
git \
|
||||
python3-dev \
|
||||
py3-pip
|
||||
|
||||
# On arm64, we need to build cffi from source.
|
||||
ARG TARGETPLATFORM
|
||||
|
||||
RUN if [ "${TARGETPLATFORM}" == "linux/arm64" ]; then \
|
||||
apk add --no-cache \
|
||||
build-base \
|
||||
autoconf \
|
||||
automake \
|
||||
libtool \
|
||||
zlib-dev \
|
||||
libffi-dev \
|
||||
cairo-dev \
|
||||
pkgconfig \
|
||||
; \
|
||||
fi
|
||||
curl
|
||||
|
||||
COPY . /app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
RUN python3 -m venv .venv
|
||||
RUN curl -LsSf https://astral.sh/uv/0.4.27/install.sh | sh
|
||||
|
||||
RUN source .venv/bin/activate \
|
||||
&& python3 -m pip install --no-cache-dir --upgrade pip \
|
||||
&& python3 -m pip install --no-cache-dir wheel \
|
||||
&& python3 -m pip install --no-cache-dir .[test,webservice,watcher]
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
# Instead of restarting the shell, use uv directly from its installed location.
|
||||
RUN /root/.cargo/bin/uv sync --extra test --extra webservice --extra watcher
|
||||
|
||||
FROM base
|
||||
|
||||
@@ -66,17 +57,11 @@ RUN apk add --no-cache \
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
||||
COPY --from=builder /usr/local/bin/ /usr/local/bin/
|
||||
COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
COPY --from=builder /app/.venv/ /app/.venv/
|
||||
|
||||
COPY --from=builder /app/misc/webservice.py /app/
|
||||
COPY --from=builder /app/misc/watcher.py /app/
|
||||
|
||||
# Copy minimal project files to get the test suite.
|
||||
COPY --from=builder /app/pyproject.toml /app/README.md /app/
|
||||
COPY --from=builder /app/tests /app/tests
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
name: Installation, packaging, dependencies
|
||||
description: Installation, packages, dependencies, "nothing works", test suite failures...
|
||||
title: "[Bug]: "
|
||||
labels: ["bug", "triage"]
|
||||
labels: ["triage"]
|
||||
assignees:
|
||||
- jbarlow83
|
||||
body:
|
||||
@@ -24,7 +24,7 @@ body:
|
||||
- type: dropdown
|
||||
id: packaging-system
|
||||
attributes:
|
||||
label: Where are you installing from?
|
||||
label: Where are you installing/running from?
|
||||
multiple: true
|
||||
options:
|
||||
- PyPI (pip, poetry, pipx, etc.)
|
||||
@@ -37,6 +37,11 @@ body:
|
||||
- source build
|
||||
validations:
|
||||
required: true
|
||||
- type: input
|
||||
id: version
|
||||
attributes:
|
||||
label: OCRmyPDF version
|
||||
description: Paste "ocrmypdf --version" here
|
||||
- type: dropdown
|
||||
id: operating-system
|
||||
attributes:
|
||||
@@ -47,6 +52,18 @@ body:
|
||||
- Windows
|
||||
- macOS
|
||||
- BSD
|
||||
- type: input
|
||||
id: os_version
|
||||
attributes:
|
||||
label: Operating system details and version
|
||||
- type: checkboxes
|
||||
attributes:
|
||||
label: Simple sanity checks
|
||||
description: Select all that apply
|
||||
options:
|
||||
- label: Operating system is currently supported by its vendor (not end of life)
|
||||
- label: Python version is compatible with OCRmyPDF
|
||||
- label: This issue is not about a specific input file
|
||||
- type: textarea
|
||||
id: logs
|
||||
attributes:
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
name: Problem with specific file
|
||||
description: Something went wrong while trying to OCR a specific file
|
||||
title: "[Bug]: "
|
||||
labels: ["bug", "triage"]
|
||||
labels: ["triage"]
|
||||
assignees:
|
||||
- jbarlow83
|
||||
body:
|
||||
@@ -39,7 +39,7 @@ body:
|
||||
causing the issue. There's really no substitute for a test file.
|
||||
|
||||
We understand files may contain personal or sensitive information. Here are some options:
|
||||
- Try reproducing the issue with a file from the test suite. (See tests/resources)
|
||||
- Try reproducing the issue with a file from the OCRmyPDF test suite. (See tests/resources)
|
||||
- Try to create another file in the same way as your private file.
|
||||
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
||||
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
name: Problem with third party app that uses OCRmyPDF
|
||||
description: |
|
||||
For PDF generation issues with third party software such as Paperless-ngx that
|
||||
uses OCRmyPDF to perform OCR or generate PDFs.
|
||||
title: "[3rdparty]: "
|
||||
labels: ["triage"]
|
||||
assignees:
|
||||
- jbarlow83
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Thanks for taking the time to describe this issue with a particular file
|
||||
and third party app.
|
||||
|
||||
If you are comfortable using OCRmyPDF, please trying to install OCRmyPDF,
|
||||
run it on your file, and see if it works. It's easier for everyone
|
||||
if you can confirm that the issue occurs with OCRmyPDF and not with
|
||||
the third party app.
|
||||
- type: checkboxes
|
||||
attributes:
|
||||
label: Simple sanity checks
|
||||
description: Select all that apply
|
||||
options:
|
||||
- label: This is an issue with an app that uses OCRmyPDF for OCR
|
||||
- label: I am using a recent version of the third party app
|
||||
- label: I will include a file that reproduces the issuse
|
||||
- type: input
|
||||
id: thirdparty-app-name-version
|
||||
attributes:
|
||||
label: Third party app name and version
|
||||
description: e.g. Paperless-ngx 2.9.0
|
||||
- type: textarea
|
||||
id: what-happened
|
||||
attributes:
|
||||
label: Describe the bug
|
||||
description: A clear and concise description of what the bug is.
|
||||
placeholder: Tell us what you see!
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
id: reproduce
|
||||
attributes:
|
||||
label: Steps to reproduce
|
||||
description: Please include steps to reproduce.
|
||||
value: |
|
||||
1. Import attached file into Paperless-ngx
|
||||
2. Trigger OCR
|
||||
3. Check log file
|
||||
4. ...
|
||||
render: plain text
|
||||
- type: textarea
|
||||
id: files
|
||||
attributes:
|
||||
label: Files
|
||||
description: |
|
||||
Please attach the input and output files, or any screenshots that may be helpful.
|
||||
|
||||
If you cannot provide a test file, we probably won't be able to help with the issue.
|
||||
PDF is a complex file format, and there may be technical details in the PDF that are
|
||||
causing the issue. There's really no substitute for a test file.
|
||||
|
||||
We understand files may contain personal or sensitive information. Here are some options:
|
||||
- Try reproducing the issue with a file from the test suite. (See tests/resources)
|
||||
- Try to create another file in the same way as your private file.
|
||||
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
||||
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
||||
omits personal information.
|
||||
placeholder: |
|
||||
Drag and drop files here.
|
||||
- type: input
|
||||
id: version
|
||||
attributes:
|
||||
label: OCRmyPDF version
|
||||
description: Paste "ocrmypdf --version" here
|
||||
placeholder: ocrmypdf --version
|
||||
- type: textarea
|
||||
id: logs
|
||||
attributes:
|
||||
label: Relevant log output
|
||||
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||
placeholder: Run OCRmyPDF with verbosity `-v1` to get more detailed logging output.
|
||||
render: plain text
|
||||
+77
-56
@@ -21,18 +21,13 @@ jobs:
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
os: [ubuntu-22.04, ubuntu-24.04]
|
||||
python: ["3.10", "3.11", "3.12", "3.13"]
|
||||
include:
|
||||
- os: ubuntu-22.04
|
||||
tesseract_ppa: "ppa"
|
||||
python: "3.10"
|
||||
- os: ubuntu-22.04
|
||||
python: "3.11"
|
||||
- os: ubuntu-22.04
|
||||
python: "3.10"
|
||||
tesseract5: true
|
||||
- os: ubuntu-latest
|
||||
python: "3.12"
|
||||
tesseract5: true
|
||||
- os: ubuntu-latest
|
||||
- os: ubuntu-24.04
|
||||
python: "pypy3.10"
|
||||
|
||||
env:
|
||||
@@ -44,16 +39,20 @@ jobs:
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
name: Setup Python
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
with:
|
||||
version: "0.4.27"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
cache: "pip"
|
||||
|
||||
- name: Install Tesseract 5
|
||||
if: matrix.tesseract5
|
||||
- name: Install Tesseract from PPA
|
||||
if: matrix.tesseract_ppa == 'ppa'
|
||||
run: |
|
||||
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr-devel
|
||||
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr5.3
|
||||
|
||||
- name: Install common packages
|
||||
run: |
|
||||
@@ -61,6 +60,7 @@ jobs:
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
ghostscript \
|
||||
jbig2dec \
|
||||
img2pdf \
|
||||
libexempi8 \
|
||||
libffi-dev \
|
||||
@@ -84,8 +84,7 @@ jobs:
|
||||
|
||||
- name: Install Python packages
|
||||
run: |
|
||||
python -m pip install --upgrade pip wheel
|
||||
python -m pip install --prefer-binary .[test]
|
||||
uv sync --extra test
|
||||
|
||||
- name: Report versions
|
||||
run: |
|
||||
@@ -93,14 +92,16 @@ jobs:
|
||||
gs --version
|
||||
pngquant --version
|
||||
unpaper --version
|
||||
img2pdf --version
|
||||
uv run img2pdf --version
|
||||
|
||||
- name: Test
|
||||
run: |
|
||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
uv run pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v4
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
files: ./coverage.xml
|
||||
env_vars: OS,PYTHON
|
||||
@@ -110,8 +111,8 @@ jobs:
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
os: [macos-latest]
|
||||
python: ["3.10", "3.11", "3.12"]
|
||||
os: [macos-latest, macos-13] # macos-latest is arm64, macos-13 is x86_64
|
||||
python: ["3.10", "3.11", "3.12", "3.13"]
|
||||
|
||||
env:
|
||||
OS: ${{ matrix.os }}
|
||||
@@ -131,34 +132,38 @@ jobs:
|
||||
ghostscript \
|
||||
jbig2enc \
|
||||
openjpeg \
|
||||
openssl \
|
||||
pngquant \
|
||||
tesseract
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
name: Setup Python
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
with:
|
||||
version: "0.4.27"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
cache: "pip"
|
||||
|
||||
- name: Install Python packages
|
||||
run: |
|
||||
python -m pip install --upgrade pip wheel
|
||||
python -m pip install --prefer-binary .[test]
|
||||
uv sync --extra test
|
||||
|
||||
- name: Report versions
|
||||
run: |
|
||||
tesseract --version
|
||||
gs --version
|
||||
pngquant --version
|
||||
img2pdf --version
|
||||
uv run img2pdf --version
|
||||
|
||||
- name: Test
|
||||
run: |
|
||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
uv run pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v4
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
files: ./coverage.xml
|
||||
env_vars: OS,PYTHON
|
||||
@@ -169,7 +174,7 @@ jobs:
|
||||
strategy:
|
||||
matrix:
|
||||
os: [windows-latest]
|
||||
python: ["3.10", "3.11", "3.12"]
|
||||
python: ["3.10", "3.11", "3.12", "3.13"]
|
||||
|
||||
env:
|
||||
OS: ${{ matrix.os }}
|
||||
@@ -180,11 +185,15 @@ jobs:
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
name: Setup Python
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
with:
|
||||
version: "0.4.27"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v5
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
cache: "pip"
|
||||
|
||||
- name: Install system packages
|
||||
run: |
|
||||
@@ -193,15 +202,16 @@ jobs:
|
||||
|
||||
- name: Install Python packages
|
||||
run: |
|
||||
python -m pip install --upgrade pip wheel
|
||||
python -m pip install --prefer-binary .[test]
|
||||
uv sync --extra test
|
||||
|
||||
- name: Test
|
||||
run: |
|
||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
uv run pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v4
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
files: ./coverage.xml
|
||||
env_vars: OS,PYTHON
|
||||
@@ -214,16 +224,14 @@ jobs:
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
- uses: actions/setup-python@v5
|
||||
name: Setup Python
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v3
|
||||
with:
|
||||
python-version: "3.10"
|
||||
cache: "pip"
|
||||
version: "0.4.27"
|
||||
|
||||
- name: Make wheels and sdist
|
||||
run: |
|
||||
python -m pip install --upgrade pip wheel build
|
||||
python -m build --sdist --wheel
|
||||
uv build --sdist --wheel
|
||||
|
||||
- uses: actions/upload-artifact@v4
|
||||
with:
|
||||
@@ -251,29 +259,45 @@ jobs:
|
||||
|
||||
create_release:
|
||||
name: Create GitHub release
|
||||
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||
needs: [upload_pypi]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||
permissions:
|
||||
# Required to create a release
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/download-artifact@v4
|
||||
with:
|
||||
name: artifact
|
||||
path: dist
|
||||
|
||||
- name: Create Release
|
||||
id: create-release
|
||||
uses: shogo82148/actions-create-release@v1
|
||||
|
||||
- name: Upload Assets
|
||||
uses: shogo82148/actions-upload-release-asset@v1
|
||||
- name: Sign the dists with Sigstore
|
||||
uses: sigstore/gh-action-sigstore-python@v3.0.0
|
||||
with:
|
||||
upload_url: ${{ steps.create-release.outputs.upload_url }}
|
||||
asset_path: |
|
||||
./dist/*.whl
|
||||
inputs: >-
|
||||
./dist/*.tar.gz
|
||||
./dist/*.whl
|
||||
|
||||
- name: Create GitHub Release
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: >-
|
||||
gh release create
|
||||
'${{ github.ref_name }}'
|
||||
--repo '${{ github.repository }}'
|
||||
--notes ""
|
||||
|
||||
- name: Upload artifact signatures to GitHub Release
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
# Upload to GitHub Release using the `gh` CLI.
|
||||
# `dist/` contains the built packages, and the
|
||||
# sigstore-produced signatures and certificates.
|
||||
run: >-
|
||||
gh release upload
|
||||
'${{ github.ref_name }}' dist/**
|
||||
--repo '${{ github.repository }}'
|
||||
|
||||
docker_ubuntu:
|
||||
name: Build Ubuntu-based Docker image
|
||||
@@ -352,9 +376,6 @@ jobs:
|
||||
username: jbarlow83
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v3
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
@@ -366,6 +387,6 @@ jobs:
|
||||
run: |
|
||||
docker buildx build \
|
||||
--push \
|
||||
--platform linux/amd64 \
|
||||
--platform linux/amd64,linux/arm64 \
|
||||
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \
|
||||
--file .docker/Dockerfile.alpine .
|
||||
|
||||
@@ -44,3 +44,4 @@ docs/_build/
|
||||
docs/_static/
|
||||
docs/_templates/
|
||||
docs/Makefile
|
||||
src/ocrmypdf/_version.py
|
||||
+59
-2
@@ -228,6 +228,59 @@ then run ocrmypdf as follows (along with any other desired arguments):
|
||||
Some combinations of control parameters will break Tesseract or break
|
||||
assumptions that OCRmyPDF makes about Tesseract's output.
|
||||
|
||||
Changing page segmentation mode
|
||||
-------------------------------
|
||||
|
||||
The directive ``--tesseract-pagesegmode Nmode`` forwards the desired page segmentation
|
||||
mode to Tesseract OCR. The default is 3.
|
||||
|
||||
Page segmentation can improve OCR results when you know that a PDF ought to be
|
||||
analyzed a particular way, such as PDFs whose pages contain only a single line of
|
||||
text. For the vast majority of users, changing the page segmentation mode will only
|
||||
make things worse.
|
||||
|
||||
As of June 2024, the Tesseract page segmentation modes are:
|
||||
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| ID | Description |
|
||||
+=====+==================================================================================+
|
||||
| 0 | Orientation and script detection (OSD) only. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 1 | Automatic page segmentation with OSD. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 2 | Automatic page segmentation, but no OSD, or OCR. (not implemented) |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 3 | Fully automatic page segmentation, but no OSD. (Default) |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 4 | Assume a single column of text of variable sizes. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 5 | Assume a single uniform block of vertically aligned text. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 6 | Assume a single uniform block of text. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 7 | Treat the image as a single text line. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 8 | Treat the image as a single word. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 9 | Treat the image as a single word in a circle. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 10 | Treat the image as a single character. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 11 | Sparse text. Find as much text as possible in no particular order. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 12 | Sparse text with OSD. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
| 13 | Raw line. Treat the image as a single text line, bypassing hacks that are |
|
||||
| | Tesseract-specific. |
|
||||
+-----+----------------------------------------------------------------------------------+
|
||||
|
||||
Modes 0, 1, 2, and 12 (all of those that enable orientation and script detection)
|
||||
are not compatible with OCRmyPDF, which performs OSD in a separate step from OCR.
|
||||
Their use may interfere with ``--rotate-pages`` and other features.
|
||||
|
||||
It is currently not possible to use advanced Tesseract OCR features, such as creating
|
||||
OCR information, when using Tesseract through OCRmyPDF.
|
||||
|
||||
Changing the PDF renderer
|
||||
=========================
|
||||
|
||||
@@ -401,6 +454,10 @@ whether it succeeded or failed. An example message is:
|
||||
Temporary working files retained at:
|
||||
/tmp/ocrmypdf.io.u20wpz07
|
||||
|
||||
When OCRmyPDF is launched as a snap, this corresponds to the snap filesystem, for instance:
|
||||
|
||||
/tmp/snap-private-tmp/snap.ocrmypdf/tmp/ocrmypdf.io.u20wpz07
|
||||
|
||||
The organization of this folder is an implementation detail and subject
|
||||
to change between releases. However the general organization is that
|
||||
working files on a per page basis have the page number as a prefix
|
||||
@@ -412,9 +469,9 @@ suffix indicates the file type. Some important files include:
|
||||
on arguments this may differ from the presentation image
|
||||
- ``_pp_deskew.png`` - the image, after deskewing
|
||||
- ``_pp_clean.png`` - the image, after cleaning with unpaper
|
||||
- ``_ocr_tess.pdf`` - the OCR file; appears as a blank page with invisible
|
||||
- ``_ocr_hocr.pdf`` - the OCR file; appears as a blank page with invisible
|
||||
text embedded
|
||||
- ``_ocr_tess.txt`` - the OCR text (not necessarily all text on the page,
|
||||
- ``_ocr_hocr.txt`` - the OCR text (not necessarily all text on the page,
|
||||
if the page is mixed format)
|
||||
- ``fix_docinfo.pdf`` - a temporary file created to fix the PDF DocumentInfo
|
||||
data structure
|
||||
|
||||
+2
-2
@@ -44,7 +44,7 @@ place, and printing each filename in between runs:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
find . -printf '%p\n' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
|
||||
find . -name '*.pdf' -printf '%p\n' -exec ocrmypdf '{}' '{}' \;
|
||||
|
||||
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
||||
``find`` to create a directory list and ``parallel`` to parallelize runs
|
||||
@@ -135,7 +135,7 @@ Users may need to customize the script to meet their requirements.
|
||||
"OCR_ON_SUCCESS_ARCHIVE", "This will move the processed original file to ``OCR_ARCHIVE_DIRECTORY`` if the exit code is 0 (OK). Note that ``OCR_ON_SUCCESS_DELETE`` takes precedence over this option, i.e. if both options are set, the input file will be deleted."
|
||||
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
||||
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
|
||||
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true, ""optimize"": "3"}'``."
|
||||
"OCR_POLL_NEW_FILE_SECONDS", "Polling interval"
|
||||
"OCR_LOGLEVEL", "Level of log messages to report"
|
||||
|
||||
|
||||
+9
-3
@@ -42,7 +42,7 @@ execute the image:
|
||||
- Architecture
|
||||
- Description
|
||||
* - ``jbarlow83/ocrmypdf-alpine``
|
||||
- x86_64 only
|
||||
- x86_64 and arm64
|
||||
- Recommended image, based on Alpine Linux.
|
||||
* - ``jbarlow83/ocrmypdf-ubuntu``
|
||||
- x86_64 and arm64
|
||||
@@ -65,7 +65,13 @@ The ``ocrmypdf`` image is also available, but is deprecated and will be removed
|
||||
in the future.
|
||||
|
||||
OCRmyPDF will use all available CPU cores. See the Docker documentation for
|
||||
`adjusting memory and CPU on other platforms <https://docs.docker.com/config/containers/resource_constraints/>`__.
|
||||
`adjusting memory and CPU on other platforms <https://docs.docker.com/config/containers/resource_constraints/>`__
|
||||
if you are using Docker on macOS or Windows, where you may need to manually assign
|
||||
more resources. On Linux, all resources will be available automatically.
|
||||
|
||||
The underlying operating system and other details in Docker images are subject
|
||||
to change at minor releases. If you are modifying the image, you should pin
|
||||
the version you intend to use.
|
||||
|
||||
Using the Docker image on the command line
|
||||
==========================================
|
||||
@@ -81,7 +87,7 @@ To start a Docker container (instance of the image):
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker tag jbarlow83/ocrmypdf ocrmypdf
|
||||
docker tag jbarlow83/ocrmypdf-alpine ocrmypdf
|
||||
docker run --rm -i ocrmypdf (... all other arguments here...) - -
|
||||
|
||||
For convenience, create a shell alias to hide the Docker command. It is
|
||||
|
||||
@@ -224,7 +224,7 @@ standard tooling needed to build packages, such as a compiler and binary tools.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo pacman -S base-devel
|
||||
sudo pacman -S --needed base-devel
|
||||
|
||||
Now you are ready to install the OCRmyPDF package.
|
||||
|
||||
@@ -341,7 +341,7 @@ OCRmyPDF is includes in MacPorts:
|
||||
sudo port install ocrmypdf
|
||||
|
||||
Note that while this will install tesseract you will need to install
|
||||
the appropriate tesseract `language ports <https://ports.macports.org/search/?selected_facets=categories_exact%3Atextproc&installed_file=&q=tesseract&name=on>`__.
|
||||
the appropriate tesseract `language ports <https://ports.macports.org/search/?selected_facets=categories_exact%3Atextproc&installed_file=&q=tesseract&name=on>`__.
|
||||
|
||||
Manual installation on macOS
|
||||
----------------------------
|
||||
@@ -640,8 +640,7 @@ environment:
|
||||
|
||||
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
|
||||
Or, to install in `development
|
||||
mode <https://packaging.python.org/en/latest/guides/distributing-packages-using-setuptools/#working-in-development-mode>`__,
|
||||
Or, to install in editable mode
|
||||
allowing customization of OCRmyPDF, use the ``-e`` flag:
|
||||
|
||||
.. code-block:: bash
|
||||
@@ -681,7 +680,7 @@ To install all of the development and test requirements:
|
||||
.. code-block:: bash
|
||||
|
||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
python -m .venv
|
||||
python -m venv .venv
|
||||
source .venv/bin/activate
|
||||
cd OCRmyPDF
|
||||
pip install -e .[test]
|
||||
|
||||
+17
-1
@@ -68,6 +68,22 @@ to what languages it should search for. Multiple languages can be
|
||||
requested using either ``-l eng+fra`` (English and French) or
|
||||
``-l eng -l fra``.
|
||||
|
||||
Archlinux
|
||||
------
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# Display a list of all Tesseract language packs
|
||||
pacman -Ss tesseract-data
|
||||
|
||||
# Install German language pack
|
||||
pacman -S tesseract-data-deu
|
||||
|
||||
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||
to what languages it should search for. Multiple languages can be
|
||||
requested using either ``-l eng+fra`` (English and French) or
|
||||
``-l eng -l fra``.
|
||||
|
||||
Gentoo
|
||||
------
|
||||
|
||||
@@ -122,4 +138,4 @@ Custom language packs
|
||||
If you have fine-tuned or trained Tesseract and generated custom trained data, you can
|
||||
copy your ``customlang.traineddata`` file into your Tesseract "tessdata" folder, and
|
||||
then use the ``-l customlang`` argument to tell OCRmyPDF to pass that language on to
|
||||
Tesseract.
|
||||
Tesseract.
|
||||
|
||||
@@ -38,12 +38,13 @@ on ARM and x86_64. Performance may be poor on other processor architectures.
|
||||
Versioning scheme
|
||||
-----------------
|
||||
|
||||
OCRmyPDF uses setuptools-scm for versioning, which derives the version from
|
||||
OCRmyPDF uses hatch-vcs for versioning, which derives the version from
|
||||
Git as a single source of truth. This may be unsuitable for some distributions, e.g.
|
||||
to indicate that your distribution modifies OCRmyPDF in some way.
|
||||
|
||||
You can patch the ``__version__`` variable in ``src/ocrmypdf/_version.py`` if
|
||||
necessary.
|
||||
necessary, or set the environment variable ``SETUPTOOLS_SCM_PRETEND_VERSION``
|
||||
to the required version, if you need to override versioning for some reason.
|
||||
|
||||
jbig2enc
|
||||
--------
|
||||
|
||||
+9
-12
@@ -29,6 +29,9 @@ conventions. Note that: plugins installed with as setuptools entrypoints are
|
||||
not checked currently, because OCRmyPDF assumes you may not want to enable
|
||||
plugins for all files.
|
||||
|
||||
See [OCRmyPDF-EasyOCR](https://github.com/ocrmypdf/OCRmyPDF-EasyOCR) for an
|
||||
example of a straightforward, fully working plugin.
|
||||
|
||||
Script plugins
|
||||
==============
|
||||
|
||||
@@ -70,14 +73,15 @@ similar to ``pytest`` packages such as ``pytest-cov`` (the package) and
|
||||
module), just like pytest plugins. At the same time, please make it clear
|
||||
that your package is not official.
|
||||
|
||||
Setuptools plugins
|
||||
==================
|
||||
Plugins
|
||||
=======
|
||||
|
||||
You can also create a plugin that OCRmyPDF will always automatically load if both are
|
||||
installed in the same virtual environment, using a setuptools entrypoint.
|
||||
installed in the same virtual environment, using a project entrypoint.
|
||||
OCRmyPDF uses the entrypoint namespace "ocrmypdf".
|
||||
|
||||
Your package's ``pyproject.toml`` would need to contain the following, for a plugin
|
||||
named ``ocrmypdf-exampleplugin``:
|
||||
For example, ``pyproject.toml`` would need to contain the following, for a plugin named
|
||||
``ocrmypdf-exampleplugin``:
|
||||
|
||||
.. code-block:: toml
|
||||
|
||||
@@ -87,13 +91,6 @@ named ``ocrmypdf-exampleplugin``:
|
||||
[project.entry-points."ocrmypdf"]
|
||||
exampleplugin = "exampleplugin.pluginmodule"
|
||||
|
||||
.. code-block:: ini
|
||||
|
||||
# equivalent setup.cfg
|
||||
[options.entry_points]
|
||||
ocrmypdf =
|
||||
exampleplugin = exampleplugin.pluginmodule
|
||||
|
||||
Plugin requirements
|
||||
===================
|
||||
|
||||
|
||||
@@ -30,6 +30,112 @@ OCRmyPDF typically supports the three most recent Python versions.
|
||||
|
||||
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||
|
||||
v16.6.2
|
||||
=======
|
||||
|
||||
- Remove invalid hyperlink annotations to satisfy Ghostscript 10.x during PDF/A
|
||||
conversion. :issue:`1425`
|
||||
|
||||
v16.6.1
|
||||
=======
|
||||
|
||||
- Fixed some issues with Docker build, such as removing unnecessary content and using
|
||||
a stable Tesseract version.
|
||||
- Reverted Docker image to Ubuntu 22.04 to access older/more stable Ghostscript
|
||||
for now.
|
||||
- Clarified batch commands in documentation.
|
||||
- Fixed an issue with JSON serialization and pickling of HOCRResult. :issue:`1427`
|
||||
|
||||
v16.6.0
|
||||
=======
|
||||
|
||||
- Fixed an issue where damaged PDFs would fail with ``--redo-ocr``. :issue:`1403`
|
||||
- Fixed an error that prevented JBIG2 optimization on Windows if the image
|
||||
was optimized in an earlier step. :issue:`1396`
|
||||
- Fixed an error detecting the version of unpaper 7.0.0. :issue:`1409`
|
||||
- Fixed a performance regression when scanning pages. :issue:`1378`. Thanks @aliemjay.
|
||||
- Fixed Alpine Docker image by enforcing Alpine 3.19. Alpine 3.20 includes a
|
||||
defective version of Tesseract OCR and so is not usable.
|
||||
- Upgraded Ubuntu Docker image to use Ubuntu 24.04.
|
||||
- Build and test scripts/actions switched to uv.
|
||||
- When running in a container, we now remind the user that temporary folders
|
||||
are inside the container and may not be accessible.
|
||||
- Fixed Linux test coverage matrix, which was missing some key versions.
|
||||
|
||||
v16.5.0
|
||||
=======
|
||||
|
||||
- Fixed issue with interpreting PDFs that have images with array masks.
|
||||
:issue:`1377`
|
||||
- Enabled testing on Python 3.13.
|
||||
- Fixed a test that did not work correctly but still passed. :issue:`1382`
|
||||
- Improved "PDF/A conversion failed" warning message to better describe implications.
|
||||
- Updated documentation to better explain OCR_JSON_SETTINGS in batch processing.
|
||||
- Build backend changed from setuptools to hatchling.
|
||||
|
||||
v16.4.3
|
||||
=======
|
||||
|
||||
- Work around pdfminer.six issue where a token on the buffer boundary is incorrectly
|
||||
parsed as two tokens. :issue:`1361`
|
||||
- New rules are applied to stencil masks and explicit masks when calculating the
|
||||
optimal page DPI for rendering. :issue:`1362`
|
||||
- Fixed attempts to use an incompatible jbig2.EXE provided by TeX Live. :issue:`1363`
|
||||
|
||||
v16.4.2
|
||||
=======
|
||||
|
||||
- Fixed order of filenames passed to Ghostscript for PDF/A generation. :issue:`1359`
|
||||
- Suppressed missing jbig2dec warning message. :issue:`1358`
|
||||
- Fixed calculation of image size when soft mask dimensions don't match image
|
||||
dimension. :issue:`1351`
|
||||
- Several fixes to documentation. Thanks to users Iris and JoKalliauer
|
||||
who contributed these changes.
|
||||
- Fixed error on processing PDFs that are missing certain image metadata. :issue:`1315`
|
||||
|
||||
v16.4.1
|
||||
=======
|
||||
|
||||
- Fixed calculation of image printed area (used in finding weighted DPI for OCR).
|
||||
:issue:`1334`
|
||||
- Fixed "NotImplementedError: not sure how to get colorspace" error
|
||||
messages in logs which simply records a failure to optimize images with
|
||||
print production colorspaces. :issue:`1315`
|
||||
|
||||
v16.4.0
|
||||
=======
|
||||
|
||||
- Selecting the ``osd`` and ``equ`` pseudo-languages with ``-l/--language`` now
|
||||
exits with an error when using Tesseract OCR, because these are not
|
||||
regular Tesseract languages but implementation details implemented.
|
||||
Using them can cause Tesseract to crash.
|
||||
- The hOCR renderer is more tolerant of extra whitespace in input files.
|
||||
- watcher.py now changes the output file extension to .pdf when the input is not
|
||||
.pdf.
|
||||
- Improved handling of PDFs that contain circularly referenced Form XObjects.
|
||||
:issue:`1321`
|
||||
- Fixed Alpine Docker image for ARM64, which was not building correctly.
|
||||
- Docker images now use pikepdf 9.0.0.
|
||||
- Prevent use of Tesseract OCR 5.4.0, a version with known regressions.
|
||||
- Disabled progressbar for "Linearizing" when ``--no-progress-bar`` set.
|
||||
- Fixed some tests that warn about missing JBIG2 decoding via pikepdf, by
|
||||
installing the necessary libraries during tests.
|
||||
|
||||
v16.3.1
|
||||
=======
|
||||
|
||||
- Fixed a test suite failure with Ghostscript 10.03.0+. :issue:`1316`
|
||||
- Fixed an issue with the presentation of the "OCR" progress bar. :issue:`1313`
|
||||
|
||||
v16.3.0
|
||||
=======
|
||||
|
||||
- Fixed progress bar not displaying for Ghostscript PDF/A conversion. :issue:`1313`
|
||||
- Added progress bar for linearization. :issue:`1313`
|
||||
- If `--rotate-pages-threshold` issued without `--rotate-pages` we now exit with
|
||||
an error since the user likely intended to use `--rotate-pages`. :issue:`1309`
|
||||
- If Tesseract hOCR gives an invalid line box, print an error message instead of
|
||||
exiting with an error. :issue:`1312`
|
||||
|
||||
v16.2.0
|
||||
=======
|
||||
|
||||
+4
-4
@@ -14,12 +14,12 @@ You should edit this script to meet your needs.
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import filecmp
|
||||
import logging
|
||||
import sys
|
||||
import os
|
||||
import posixpath
|
||||
import shutil
|
||||
import filecmp
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import ocrmypdf
|
||||
@@ -70,7 +70,7 @@ for filename in start_dir.glob("**/*.pdf"):
|
||||
logging.info(f"Archiving document to {archive_filename}")
|
||||
try:
|
||||
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||
except IOError as io_err:
|
||||
except OSError:
|
||||
os.makedirs(posixpath.dirname(archive_filename))
|
||||
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||
try:
|
||||
@@ -86,6 +86,6 @@ for filename in start_dir.glob("**/*.pdf"):
|
||||
logging.info(
|
||||
"Skipped document because it does not need ocr as it is tagged"
|
||||
)
|
||||
except:
|
||||
except Exception:
|
||||
logging.error("Unhandled error occured")
|
||||
logging.info("OCR complete")
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Helper script for bisecting PDFs to find a page with an issue."""
|
||||
|
||||
import sys
|
||||
|
||||
import pikepdf
|
||||
|
||||
if len(sys.argv) != 2:
|
||||
print(f"Usage: {sys.argv[0]} <input.pdf>")
|
||||
sys.exit(1)
|
||||
|
||||
with pikepdf.open(sys.argv[1]) as pdf:
|
||||
num_pages = len(pdf.pages)
|
||||
low = 0
|
||||
high = num_pages - 1
|
||||
while low <= high:
|
||||
mid = (low + high) // 2
|
||||
with pikepdf.new() as new_pdf:
|
||||
new_pdf.pages.extend(pdf.pages[low : mid + 1])
|
||||
new_pdf.save(f"bisect-issue-{low + 1}-{mid + 1}.pdf")
|
||||
print(f"Is bisect-issue-{low + 1}-{mid + 1}.pdf good or bad?", end=" ")
|
||||
while True:
|
||||
response = input().lower()
|
||||
if response == "good":
|
||||
low = mid + 1
|
||||
break
|
||||
elif response == "bad":
|
||||
high = mid - 1
|
||||
break
|
||||
else:
|
||||
print("Please respond with 'good' or 'bad'.")
|
||||
print(f"The issue is on page {low + 1} of the original PDF.")
|
||||
with pikepdf.new() as new_pdf:
|
||||
new_pdf.pages.extend(pdf.pages[low])
|
||||
new_pdf.save(f"bisect-issue-bad-{low + 1}.pdf")
|
||||
with pikepdf.new() as new_pdf:
|
||||
new_pdf.pages.extend(pdf.pages[:low])
|
||||
new_pdf.pages.extend(pdf.pages[low + 1 :])
|
||||
new_pdf.save(f"bisect-issue-good-{low + 1}.pdf")
|
||||
+7
-4
@@ -46,15 +46,18 @@ class LoggingLevelEnum(str, Enum):
|
||||
CRITICAL = "CRITICAL"
|
||||
|
||||
|
||||
def get_output_dir(root: Path, basename: str, output_dir_year_month: bool) -> Path:
|
||||
def get_output_path(root: Path, basename: str, output_dir_year_month: bool) -> Path:
|
||||
assert '/' not in basename, "basename must not contain '/'"
|
||||
if output_dir_year_month:
|
||||
today = datetime.today()
|
||||
output_directory_year_month = root / str(today.year) / f'{today.month:02d}'
|
||||
if not output_directory_year_month.exists():
|
||||
output_directory_year_month.mkdir(parents=True, exist_ok=True)
|
||||
output_path = Path(output_directory_year_month) / basename
|
||||
output_path = Path(output_directory_year_month) / Path(basename).with_suffix(
|
||||
'.pdf'
|
||||
)
|
||||
else:
|
||||
output_path = root / basename
|
||||
output_path = root / Path(basename).with_suffix('.pdf')
|
||||
return output_path
|
||||
|
||||
|
||||
@@ -98,7 +101,7 @@ def execute_ocrmypdf(
|
||||
retries_loading_file: int,
|
||||
output_dir_year_month: bool,
|
||||
):
|
||||
output_path = get_output_dir(output_dir, file_path.name, output_dir_year_month)
|
||||
output_path = get_output_path(output_dir, file_path.name, output_dir_year_month)
|
||||
|
||||
log.info("-" * 20)
|
||||
log.info(f'New file: {file_path}. Waiting until fully written...')
|
||||
|
||||
+10
-9
@@ -1,8 +1,8 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
[build-system]
|
||||
requires = ["setuptools >= 61", "setuptools_scm[toml] >= 7.0.5", "wheel"]
|
||||
build-backend = "setuptools.build_meta"
|
||||
requires = ["hatchling", "hatch-vcs"]
|
||||
build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "ocrmypdf"
|
||||
@@ -46,6 +46,7 @@ keywords = ["PDF", "OCR", "optical character recognition", "PDF/A", "scanning"]
|
||||
Documentation = "https://ocrmypdf.readthedocs.io/"
|
||||
Source = "https://github.com/ocrmypdf/OCRmyPDF"
|
||||
Tracker = "https://github.com/ocrmypdf/OCRmyPDF/issues"
|
||||
Changelog = "https://github.com/ocrmypdf/OCRmyPDF/docs/release_notes.rst"
|
||||
|
||||
[project.optional-dependencies]
|
||||
docs = ["sphinx", "sphinx-issues", "sphinx-rtd-theme"]
|
||||
@@ -67,14 +68,11 @@ webservice = ["Flask>=2.0.1"]
|
||||
[project.scripts]
|
||||
ocrmypdf = "ocrmypdf.__main__:run"
|
||||
|
||||
[tool.setuptools.package-data]
|
||||
ocrmypdf = ["data/sRGB.icc", "py.typed"]
|
||||
[tool.hatch.version]
|
||||
source = "vcs"
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
where = ["src"]
|
||||
namespaces = false
|
||||
|
||||
[tool.setuptools_scm]
|
||||
[tool.hatch.build.hooks.vcs]
|
||||
version-file = "src/ocrmypdf/_version.py"
|
||||
|
||||
[tool.distutils.bdist_wheel]
|
||||
python-tag = "py310"
|
||||
@@ -157,3 +155,6 @@ convention = "google"
|
||||
|
||||
[tool.ruff.format]
|
||||
quote-style = "preserve"
|
||||
|
||||
[dependency-groups]
|
||||
dev = ["mypy>=1.13.0"]
|
||||
|
||||
+2
-2
@@ -18,8 +18,8 @@ architectures: [amd64]
|
||||
|
||||
environment:
|
||||
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
||||
GS_LIB: $SNAP/usr/share/ghostscript/9.55/Resource/Init
|
||||
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.55/Resource/Font
|
||||
GS_LIB: $SNAP/usr/share/ghostscript/9.55.0/Resource/Init
|
||||
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.55.0/Resource/Font
|
||||
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
||||
|
||||
apps:
|
||||
|
||||
@@ -9,11 +9,12 @@ from pluggy import HookimplMarker as _HookimplMarker
|
||||
|
||||
from ocrmypdf import helpers, hocrtransform, pdfa, pdfinfo
|
||||
from ocrmypdf._concurrent import Executor
|
||||
from ocrmypdf._defaults import PROGRAM_NAME
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._pipelines._common import (
|
||||
configure_debug_logging,
|
||||
)
|
||||
from ocrmypdf._version import PROGRAM_NAME, __version__
|
||||
from ocrmypdf._version import __version__
|
||||
from ocrmypdf.api import (
|
||||
Verbosity,
|
||||
configure_logging,
|
||||
@@ -37,7 +38,6 @@ from ocrmypdf.pluginspec import OcrEngine, OrientationConfidence
|
||||
|
||||
hookimpl = _HookimplMarker('ocrmypdf')
|
||||
|
||||
|
||||
__all__ = [
|
||||
'__version__',
|
||||
'BadArgsError',
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""OCRmyPDF PDF annotation cleanup."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from pikepdf import Dictionary, Name, NameTree, Pdf
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def remove_broken_goto_annotations(pdf: Pdf) -> bool:
|
||||
"""Remove broken goto annotations from a PDF.
|
||||
|
||||
If a PDF contains a GoTo Action that points to a named destination that does not
|
||||
exist, Ghostscript PDF/A conversion will fail. In any event, a named destination
|
||||
that is not defined is not useful.
|
||||
|
||||
Args:
|
||||
pdf: Opened PDF file.
|
||||
|
||||
Returns:
|
||||
bool: True if the file was modified, False if not.
|
||||
"""
|
||||
modified = False
|
||||
|
||||
# Check if there are any named destinations
|
||||
if Name.Names not in pdf.Root:
|
||||
return modified
|
||||
if Name.Dests not in pdf.Root[Name.Names]:
|
||||
return modified
|
||||
|
||||
dests = pdf.Root[Name.Names][Name.Dests]
|
||||
if not isinstance(dests, Dictionary):
|
||||
return modified
|
||||
nametree = NameTree(dests)
|
||||
|
||||
# Create a set of all named destinations
|
||||
names = set(k for k in nametree.keys())
|
||||
|
||||
for n, page in enumerate(pdf.pages):
|
||||
if Name.Annots not in page:
|
||||
continue
|
||||
for annot in page[Name.Annots]:
|
||||
if not isinstance(annot, Dictionary):
|
||||
continue
|
||||
if Name.A not in annot or Name.D not in annot[Name.A]:
|
||||
continue
|
||||
# We found an annotation that points to a named destination
|
||||
named_destination = str(annot[Name.A][Name.D])
|
||||
if named_destination not in names:
|
||||
# If there is no corresponding named destination, remove the
|
||||
# annotation. Having no destination set is still valid and just
|
||||
# makes the link non-functional.
|
||||
log.warning(
|
||||
f"Disabling a hyperlink annotation on page {n + 1} to a "
|
||||
"non-existent named destination "
|
||||
f"{named_destination}."
|
||||
)
|
||||
del annot[Name.A][Name.D]
|
||||
modified = True
|
||||
|
||||
return modified
|
||||
@@ -0,0 +1,10 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
# Enforce English hegemony
|
||||
DEFAULT_LANGUAGE = 'eng'
|
||||
|
||||
# Default rotation threshold
|
||||
DEFAULT_ROTATE_PAGES_THRESHOLD = 14.0
|
||||
|
||||
PROGRAM_NAME = 'OCRmyPDF'
|
||||
@@ -177,6 +177,17 @@ class GhostscriptFollower:
|
||||
self.progressbar_class = progressbar_class
|
||||
self.progressbar = None
|
||||
|
||||
def __enter__(self):
|
||||
# We can't actually set up the progressbar here, because we don't know
|
||||
# how many pages there are until the first __call__() happens. So we
|
||||
# do it in __call__().
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
if self.progressbar:
|
||||
return self.progressbar.__exit__(exc_type, exc_value, traceback)
|
||||
return False
|
||||
|
||||
def __call__(self, line):
|
||||
if not self.progressbar_class:
|
||||
return
|
||||
@@ -187,7 +198,8 @@ class GhostscriptFollower:
|
||||
self.progressbar = self.progressbar_class(
|
||||
total=self.count, desc="PDF/A conversion", unit='page'
|
||||
)
|
||||
return
|
||||
# Now that we know the count, we can set up the progressbar.
|
||||
self.progressbar.__enter__()
|
||||
else:
|
||||
if self.re_page.match(line.strip()):
|
||||
self.progressbar.update()
|
||||
@@ -265,7 +277,10 @@ def generate_pdfa(
|
||||
)
|
||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||
try:
|
||||
with Path(output_file).open('wb') as output:
|
||||
with (
|
||||
Path(output_file).open('wb') as output,
|
||||
GhostscriptFollower(progressbar_class) as pbar,
|
||||
):
|
||||
p = run_polling_stderr(
|
||||
args_gs,
|
||||
stdout=output,
|
||||
@@ -274,7 +289,7 @@ def generate_pdfa(
|
||||
text=True,
|
||||
encoding='utf-8',
|
||||
errors='replace',
|
||||
callback=GhostscriptFollower(progressbar_class),
|
||||
callback=pbar,
|
||||
)
|
||||
except CalledProcessError as e:
|
||||
# Ghostscript does not change return code when it fails to create
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from subprocess import PIPE
|
||||
from subprocess import PIPE, CalledProcessError
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
@@ -14,7 +14,13 @@ from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*'))
|
||||
try:
|
||||
version = get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
except CalledProcessError as e:
|
||||
# TeX Live for Windows provides an incompatible jbig2.EXE which may
|
||||
# be on the PATH.
|
||||
raise MissingDependencyError('jbig2enc') from e
|
||||
return Version(version)
|
||||
|
||||
|
||||
def available():
|
||||
|
||||
@@ -48,7 +48,7 @@ class UnpaperImageTooLargeError(Exception):
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version('unpaper'))
|
||||
return Version(get_version('unpaper', regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)'))
|
||||
|
||||
|
||||
@contextmanager
|
||||
|
||||
@@ -15,8 +15,9 @@ from pikepdf import Dictionary, Name, Pdf
|
||||
from pikepdf import __version__ as PIKEPDF_VERSION
|
||||
from pikepdf.models.metadata import PdfMetadata, encode_pdf_date
|
||||
|
||||
from ocrmypdf._annots import remove_broken_goto_annotations
|
||||
from ocrmypdf._defaults import PROGRAM_NAME
|
||||
from ocrmypdf._jobcontext import PdfContext
|
||||
from ocrmypdf._version import PROGRAM_NAME
|
||||
from ocrmypdf._version import __version__ as OCRMYPF_VERSION
|
||||
from ocrmypdf.languages import iso_639_2_from_3
|
||||
|
||||
@@ -153,18 +154,45 @@ def _set_language(pdf: Pdf, languages: list[str]):
|
||||
pdf.Root.Lang = iso639_2
|
||||
|
||||
|
||||
class MetadataProgress:
|
||||
def __init__(self, progressbar_class, enable: bool = True):
|
||||
self.progressbar_class = progressbar_class
|
||||
self.progressbar = self.progressbar_class(
|
||||
total=100, desc="Linearizing", unit='%', disable=not enable
|
||||
)
|
||||
|
||||
def __enter__(self):
|
||||
self.progressbar.__enter__()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return self.progressbar.__exit__(exc_type, exc_value, traceback)
|
||||
|
||||
def __call__(self, percent: int):
|
||||
if not self.progressbar_class:
|
||||
return
|
||||
self.progressbar.update(completed=percent)
|
||||
|
||||
|
||||
def metadata_fixup(
|
||||
working_file: Path, context: PdfContext, pdf_save_settings: dict[str, Any]
|
||||
) -> Path:
|
||||
"""Fix certain metadata fields after Ghostscript PDF/A conversion.
|
||||
"""Fix certain metadata fields whether PDF or PDF/A.
|
||||
|
||||
Override some of Ghostscript's metadata choices.
|
||||
|
||||
Also report on metadata in the input file that was not retained during
|
||||
PDF/A conversion.
|
||||
conversion.
|
||||
"""
|
||||
output_file = context.get_path('metafix.pdf')
|
||||
options = context.options
|
||||
|
||||
with Pdf.open(context.origin) as original, Pdf.open(working_file) as pdf:
|
||||
pbar_class = context.plugin_manager.hook.get_progressbar_class()
|
||||
with (
|
||||
Pdf.open(context.origin) as original,
|
||||
Pdf.open(working_file) as pdf,
|
||||
MetadataProgress(pbar_class, options.progress_bar) as pbar,
|
||||
):
|
||||
docinfo = get_docinfo(original, context)
|
||||
with (
|
||||
original.open_metadata(
|
||||
@@ -182,6 +210,6 @@ def metadata_fixup(
|
||||
report_on_metadata(options, meta_missing)
|
||||
|
||||
_set_language(pdf, options.languages)
|
||||
pdf.save(output_file, **pdf_save_settings)
|
||||
pdf.save(output_file, progress=pbar, **pdf_save_settings)
|
||||
|
||||
return output_file
|
||||
|
||||
@@ -159,8 +159,13 @@ def triage(
|
||||
"Argument --image-dpi is being ignored because the "
|
||||
"input file is a PDF, not an image."
|
||||
)
|
||||
# Origin file is a pdf create a symlink with pdf extension
|
||||
safe_symlink(input_file, output_file)
|
||||
try:
|
||||
with pikepdf.open(input_file) as pdf:
|
||||
pdf.save(output_file)
|
||||
except pikepdf.PdfError as e:
|
||||
raise InputFileError() from e
|
||||
except pikepdf.PasswordError as e:
|
||||
raise EncryptedPdfError() from e
|
||||
return output_file
|
||||
except OSError as e:
|
||||
log.debug(f"Temporary file was at: {input_file}")
|
||||
@@ -475,7 +480,7 @@ def calculate_raster_dpi(page_context: PageContext):
|
||||
page_dpi = get_page_square_dpi(page_context, image_dpi)
|
||||
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
|
||||
log.warning(
|
||||
"Weight average image DPI is %0.1f, max DPI is %0.1f. "
|
||||
"Weighted average image DPI is %0.1f, max DPI is %0.1f. "
|
||||
"The discrepancy may indicate a high detail region on this page, "
|
||||
"but could also indicate a problem with the input PDF file. "
|
||||
"Page image will be rendered at %0.1f DPI.",
|
||||
@@ -854,7 +859,7 @@ def fix_pagepdf_boxes(
|
||||
page.CropBox = cropbox
|
||||
page.TrimBox = trimbox
|
||||
pdf.save(out_file)
|
||||
return pdf
|
||||
return out_file
|
||||
|
||||
|
||||
def generate_postscript_stub(context: PdfContext) -> Path:
|
||||
|
||||
@@ -20,7 +20,9 @@ from pathlib import Path
|
||||
from typing import NamedTuple, cast
|
||||
|
||||
import PIL
|
||||
from pikepdf import Pdf
|
||||
|
||||
from ocrmypdf._annots import remove_broken_goto_annotations
|
||||
from ocrmypdf._concurrent import Executor, setup_executor
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._logging import PageNumberFilter
|
||||
@@ -33,6 +35,7 @@ from ocrmypdf._pipeline import (
|
||||
generate_postscript_stub,
|
||||
get_orientation_correction,
|
||||
get_pdf_save_settings,
|
||||
get_pdfinfo,
|
||||
optimize_pdf,
|
||||
preprocess_clean,
|
||||
preprocess_deskew,
|
||||
@@ -51,9 +54,12 @@ from ocrmypdf.helpers import (
|
||||
available_cpu_count,
|
||||
check_pdf,
|
||||
pikepdf_enable_mmap,
|
||||
running_in_docker,
|
||||
running_in_snap,
|
||||
samefile,
|
||||
)
|
||||
from ocrmypdf.pdfa import file_claims_pdfa
|
||||
from ocrmypdf.pdfinfo import PdfInfo
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
tls = threading.local()
|
||||
@@ -100,6 +106,23 @@ class PageResult(NamedTuple):
|
||||
"""Orientation correction in degrees."""
|
||||
|
||||
|
||||
class HOCRResultEncoder(json.JSONEncoder):
|
||||
def default(self, obj):
|
||||
if isinstance(obj, Path):
|
||||
return {'Path': str(obj)}
|
||||
return super().default(obj)
|
||||
|
||||
|
||||
class HOCRResultDecoder(json.JSONDecoder):
|
||||
def __init__(self, *args, **kwargs):
|
||||
super().__init__(object_hook=self.dict_to_object, *args, **kwargs)
|
||||
|
||||
def dict_to_object(self, d):
|
||||
if 'Path' in d:
|
||||
return Path(d['Path'])
|
||||
return d
|
||||
|
||||
|
||||
@dataclass
|
||||
class HOCRResult:
|
||||
"""Result when hOCR is finished processing."""
|
||||
@@ -119,38 +142,14 @@ class HOCRResult:
|
||||
orientation_correction: int = 0
|
||||
"""Orientation correction in degrees."""
|
||||
|
||||
def __getstate__(self):
|
||||
"""Return state values to be pickled."""
|
||||
return {
|
||||
k: (
|
||||
('Path://' + str(v))
|
||||
if k in ('pdf_page_from_image', 'hocr', 'textpdf') and v is not None
|
||||
else v
|
||||
)
|
||||
for k, v in self.__dict__.items()
|
||||
}
|
||||
|
||||
def __setstate__(self, state):
|
||||
"""Restore state from the unpickled state values."""
|
||||
self.__dict__.update(
|
||||
{
|
||||
k: (
|
||||
Path(v.removeprefix('Path://'))
|
||||
if k in ('pdf_page_from_image', 'hocr', 'textpdf') and v is not None
|
||||
else v
|
||||
)
|
||||
for k, v in state.items()
|
||||
}
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def from_json(cls, json_str: str) -> HOCRResult:
|
||||
"""Create an instance from a dict."""
|
||||
return cls(**json.loads(json_str))
|
||||
return cls(**json.loads(json_str, cls=HOCRResultDecoder))
|
||||
|
||||
def to_json(self) -> str:
|
||||
"""Serialize to a JSON string."""
|
||||
return json.dumps(self.__getstate__())
|
||||
return json.dumps(self.__dict__, cls=HOCRResultEncoder)
|
||||
|
||||
|
||||
def configure_debug_logging(
|
||||
@@ -183,7 +182,7 @@ def configure_debug_logging(
|
||||
return log_file_handler, remover
|
||||
|
||||
|
||||
def worker_init(max_pixels: int) -> None:
|
||||
def worker_init(max_pixels: int | None) -> None:
|
||||
"""Initialize a worker thread or process."""
|
||||
# In Windows, child process will not inherit our change to this value in
|
||||
# the parent process, so ensure workers get it set. Not needed when running
|
||||
@@ -215,6 +214,22 @@ def manage_debug_log_handler(
|
||||
remover()
|
||||
|
||||
|
||||
def _print_temp_folder_location(work_folder: Path):
|
||||
"""Print the location of the temporary work folder."""
|
||||
msgs = [f"Temporary working files retained at:\n{work_folder}"]
|
||||
if running_in_docker(): # pragma: no cover
|
||||
msgs.append(
|
||||
"OCRmyPDF is running in a Docker container, "
|
||||
"so the files will be inside the container."
|
||||
)
|
||||
elif running_in_snap(): # pragma: no cover
|
||||
msgs.append(
|
||||
"OCRmyPDF is running in a Snap container, "
|
||||
"so the files will be inside the container."
|
||||
)
|
||||
print('\n'.join(msgs), file=sys.stderr)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def manage_work_folder(*, work_folder: Path, retain: bool, print_location: bool):
|
||||
try:
|
||||
@@ -222,10 +237,7 @@ def manage_work_folder(*, work_folder: Path, retain: bool, print_location: bool)
|
||||
finally:
|
||||
if retain:
|
||||
if print_location:
|
||||
print(
|
||||
f"Temporary working files retained at:\n{work_folder}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
_print_temp_folder_location(work_folder)
|
||||
else:
|
||||
shutil.rmtree(work_folder, ignore_errors=True)
|
||||
|
||||
@@ -300,6 +312,20 @@ def setup_pipeline(
|
||||
return executor
|
||||
|
||||
|
||||
def do_get_pdfinfo(
|
||||
pdf_path: Path, executor: Executor, options: argparse.Namespace
|
||||
) -> PdfInfo:
|
||||
return get_pdfinfo(
|
||||
pdf_path,
|
||||
executor=executor,
|
||||
detailed_analysis=options.redo_ocr,
|
||||
progbar=options.progress_bar,
|
||||
max_workers=options.jobs,
|
||||
use_threads=options.use_threads,
|
||||
check_pages=options.pages,
|
||||
)
|
||||
|
||||
|
||||
def preprocess(
|
||||
page_context: PageContext,
|
||||
image: Path,
|
||||
@@ -414,7 +440,14 @@ def postprocess(
|
||||
pdf_file: Path, context: PdfContext, executor: Executor
|
||||
) -> tuple[Path, Sequence[str]]:
|
||||
"""Postprocess the PDF file."""
|
||||
pdf_out = pdf_file
|
||||
# pdf_out = pdf_file
|
||||
with Pdf.open(pdf_file) as pdf:
|
||||
fix_annots = context.get_path('fix_annots.pdf')
|
||||
if remove_broken_goto_annotations(pdf):
|
||||
pdf.save(fix_annots)
|
||||
pdf_out = fix_annots
|
||||
else:
|
||||
pdf_out = pdf_file
|
||||
if context.options.output_type.startswith('pdfa'):
|
||||
ps_stub_out = generate_postscript_stub(context)
|
||||
pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context)
|
||||
@@ -441,7 +474,8 @@ def report_output_pdf(options, start_input_file, optimize_messages) -> ExitCode:
|
||||
log.info("Output file is a %s (as expected)", pdfa_info['conformance'])
|
||||
else:
|
||||
log.warning(
|
||||
"Output file is okay but is not PDF/A (seems to be %s)",
|
||||
"Output file is a valid PDF, but conversion to PDF/A did not "
|
||||
"succeed (issue: %s)",
|
||||
pdfa_info['conformance'],
|
||||
)
|
||||
return ExitCode.pdfa_conversion_failed
|
||||
|
||||
@@ -19,11 +19,11 @@ from ocrmypdf._graft import OcrGrafter
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._pipeline import (
|
||||
copy_final,
|
||||
get_pdfinfo,
|
||||
render_hocr_page,
|
||||
)
|
||||
from ocrmypdf._pipelines._common import (
|
||||
HOCRResult,
|
||||
do_get_pdfinfo,
|
||||
manage_work_folder,
|
||||
postprocess,
|
||||
report_output_pdf,
|
||||
@@ -117,15 +117,7 @@ def run_hocr_to_ocr_pdf_pipeline(
|
||||
origin_pdf = work_folder / 'origin.pdf'
|
||||
|
||||
# Gather pdfinfo and create context
|
||||
pdfinfo = get_pdfinfo(
|
||||
origin_pdf,
|
||||
executor=executor,
|
||||
detailed_analysis=options.redo_ocr,
|
||||
progbar=options.progress_bar,
|
||||
max_workers=options.jobs,
|
||||
use_threads=options.use_threads,
|
||||
check_pages=options.pages,
|
||||
)
|
||||
pdfinfo = do_get_pdfinfo(origin_pdf, executor, options)
|
||||
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
||||
plugin_manager.hook.check_options(options=options)
|
||||
optimize_messages = exec_hocr_to_ocr_pdf(context, executor)
|
||||
|
||||
@@ -21,7 +21,6 @@ from ocrmypdf._graft import OcrGrafter
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._pipeline import (
|
||||
copy_final,
|
||||
get_pdfinfo,
|
||||
is_ocr_required,
|
||||
merge_sidecars,
|
||||
ocr_engine_hocr,
|
||||
@@ -33,6 +32,7 @@ from ocrmypdf._pipeline import (
|
||||
from ocrmypdf._pipelines._common import (
|
||||
PageResult,
|
||||
cli_exception_handler,
|
||||
do_get_pdfinfo,
|
||||
manage_debug_log_handler,
|
||||
manage_work_folder,
|
||||
postprocess,
|
||||
@@ -103,14 +103,14 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
||||
try:
|
||||
set_thread_pageno(result.pageno + 1)
|
||||
sidecars[result.pageno] = result.text
|
||||
pbar.update()
|
||||
pbar.update(0.5)
|
||||
ocrgraft.graft_page(
|
||||
pageno=result.pageno,
|
||||
image=result.pdf_page_from_image,
|
||||
textpdf=result.ocr,
|
||||
autorotate_correction=result.orientation_correction,
|
||||
)
|
||||
pbar.update()
|
||||
pbar.update(0.5)
|
||||
finally:
|
||||
set_thread_pageno(None)
|
||||
|
||||
@@ -118,10 +118,9 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
||||
use_threads=options.use_threads,
|
||||
max_workers=max_workers,
|
||||
progress_kwargs=dict(
|
||||
total=(2 * len(context.pdfinfo)),
|
||||
total=len(context.pdfinfo),
|
||||
desc='OCR' if options.tesseract_timeout > 0 else 'Image processing',
|
||||
unit='page',
|
||||
unit_scale=0.5,
|
||||
disable=not options.progress_bar,
|
||||
),
|
||||
worker_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS),
|
||||
@@ -172,16 +171,7 @@ def _run_pipeline(
|
||||
)
|
||||
|
||||
# Gather pdfinfo and create context
|
||||
pdfinfo = get_pdfinfo(
|
||||
origin_pdf,
|
||||
executor=executor,
|
||||
detailed_analysis=options.redo_ocr,
|
||||
progbar=options.progress_bar,
|
||||
max_workers=options.jobs,
|
||||
use_threads=options.use_threads,
|
||||
check_pages=options.pages,
|
||||
)
|
||||
|
||||
pdfinfo = do_get_pdfinfo(origin_pdf, executor, options)
|
||||
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
||||
|
||||
# Validate options are okay for this pdf
|
||||
|
||||
@@ -17,13 +17,13 @@ import PIL
|
||||
from ocrmypdf._concurrent import Executor
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._pipeline import (
|
||||
get_pdfinfo,
|
||||
is_ocr_required,
|
||||
ocr_engine_hocr,
|
||||
validate_pdfinfo_options,
|
||||
)
|
||||
from ocrmypdf._pipelines._common import (
|
||||
HOCRResult,
|
||||
do_get_pdfinfo,
|
||||
manage_work_folder,
|
||||
process_page,
|
||||
set_thread_pageno,
|
||||
@@ -94,18 +94,11 @@ def run_hocr_pipeline(
|
||||
work_folder=options.output_folder, retain=True, print_location=False
|
||||
) as work_folder:
|
||||
executor = setup_pipeline(options, plugin_manager)
|
||||
shutil.copy2(options.input_file, work_folder / 'origin.pdf')
|
||||
origin_pdf = work_folder / 'origin.pdf'
|
||||
shutil.copy2(options.input_file, origin_pdf)
|
||||
|
||||
# Gather pdfinfo and create context
|
||||
pdfinfo = get_pdfinfo(
|
||||
options.input_file,
|
||||
executor=executor,
|
||||
detailed_analysis=options.redo_ocr,
|
||||
progbar=options.progress_bar,
|
||||
max_workers=options.jobs,
|
||||
use_threads=options.use_threads,
|
||||
check_pages=options.pages,
|
||||
)
|
||||
pdfinfo = do_get_pdfinfo(origin_pdf, executor, options)
|
||||
context = PdfContext(
|
||||
options, work_folder, options.input_file, pdfinfo, plugin_manager
|
||||
)
|
||||
|
||||
@@ -66,7 +66,7 @@ class ProgressBar(Protocol):
|
||||
def __exit__(self, *args):
|
||||
"""Exit a progress bar context."""
|
||||
|
||||
def update(self, n=1):
|
||||
def update(self, n=1, *, completed=None):
|
||||
"""Update the progress bar by an increment.
|
||||
|
||||
For use within a progress bar context.
|
||||
@@ -85,7 +85,7 @@ class NullProgressBar:
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return False
|
||||
|
||||
def update(self, _arg=None):
|
||||
def update(self, _arg=None, *, completed=None):
|
||||
return
|
||||
|
||||
|
||||
@@ -103,6 +103,7 @@ class RichProgressBar:
|
||||
disable: bool = False,
|
||||
**kwargs,
|
||||
):
|
||||
self._entered = False
|
||||
self.progress = Progress(
|
||||
TextColumn(
|
||||
"[progress.description]{task.description}",
|
||||
@@ -130,6 +131,7 @@ class RichProgressBar:
|
||||
|
||||
def __enter__(self):
|
||||
self.progress.start()
|
||||
self._entered = True
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
@@ -137,6 +139,10 @@ class RichProgressBar:
|
||||
self.progress.stop()
|
||||
return False
|
||||
|
||||
def update(self, value=None):
|
||||
advance = self.unit_scale if value is None else value
|
||||
self.progress.update(self.progress_bar, advance=advance)
|
||||
def update(self, n=1, *, completed=None):
|
||||
assert self._entered, "Progress bar must be entered before updating"
|
||||
if completed is None:
|
||||
advance = self.unit_scale if n is None else n
|
||||
self.progress.update(self.progress_bar, advance=advance)
|
||||
else:
|
||||
self.progress.update(self.progress_bar, completed=completed)
|
||||
|
||||
+16
-23
@@ -20,6 +20,7 @@ import pikepdf
|
||||
import PIL
|
||||
from pluggy import PluginManager
|
||||
|
||||
from ocrmypdf._defaults import DEFAULT_LANGUAGE, DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
from ocrmypdf._exec import unpaper
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
@@ -27,20 +28,18 @@ from ocrmypdf.exceptions import (
|
||||
MissingDependencyError,
|
||||
OutputFileAccessError,
|
||||
)
|
||||
from ocrmypdf.helpers import is_file_writable, monotonic, safe_symlink
|
||||
from ocrmypdf.helpers import (
|
||||
is_file_writable,
|
||||
monotonic,
|
||||
running_in_docker,
|
||||
running_in_snap,
|
||||
safe_symlink,
|
||||
)
|
||||
from ocrmypdf.subprocess import check_external_program
|
||||
|
||||
# -------------
|
||||
# External dependencies
|
||||
|
||||
DEFAULT_LANGUAGE = 'eng' # Enforce English hegemony
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# --------
|
||||
|
||||
|
||||
def check_platform() -> None:
|
||||
if sys.maxsize <= 2**32: # pragma: no cover
|
||||
log.warning(
|
||||
@@ -61,6 +60,7 @@ def check_options_languages(
|
||||
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
||||
if not ocr_engine_languages:
|
||||
return
|
||||
|
||||
missing_languages = set(options.languages) - set(ocr_engine_languages)
|
||||
if missing_languages:
|
||||
lang_text = '\n'.join(lang for lang in missing_languages)
|
||||
@@ -130,6 +130,11 @@ def check_options_preprocessing(options: Namespace) -> None:
|
||||
options.clean = True
|
||||
if options.unpaper_args and not options.clean:
|
||||
raise BadArgsError("--clean is required for --unpaper-args")
|
||||
if (
|
||||
options.rotate_pages_threshold != DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
and not options.rotate_pages
|
||||
):
|
||||
raise BadArgsError("--rotate-pages is required for --rotate-pages-threshold")
|
||||
if options.clean:
|
||||
check_external_program(
|
||||
program='unpaper',
|
||||
@@ -238,18 +243,6 @@ def check_options(options: Namespace, plugin_manager: PluginManager) -> None:
|
||||
_check_plugin_options(options, plugin_manager)
|
||||
|
||||
|
||||
def _in_docker():
|
||||
return Path('/.dockerenv').exists()
|
||||
|
||||
|
||||
def _in_snap():
|
||||
try:
|
||||
cgroup_text = Path('/proc/self/cgroup').read_text()
|
||||
return 'snap.ocrmypdf' in cgroup_text
|
||||
except FileNotFoundError:
|
||||
return False
|
||||
|
||||
|
||||
def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]:
|
||||
if options.input_file == '-':
|
||||
# stdin
|
||||
@@ -273,7 +266,7 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
|
||||
return target, os.fspath(options.input_file)
|
||||
except FileNotFoundError as e:
|
||||
msg = f"File not found - {options.input_file}"
|
||||
if _in_docker(): # pragma: no cover
|
||||
if running_in_docker(): # pragma: no cover
|
||||
msg += (
|
||||
"\nDocker cannot access your working directory unless you "
|
||||
"explicitly share it with the Docker container and set up"
|
||||
@@ -283,7 +276,7 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
|
||||
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf"
|
||||
"\n"
|
||||
)
|
||||
elif _in_snap(): # pragma: no cover
|
||||
elif running_in_snap(): # pragma: no cover
|
||||
msg += (
|
||||
"\nSnap applications cannot access files outside of "
|
||||
"your home directory unless you explicitly allow it. "
|
||||
|
||||
@@ -1,16 +0,0 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Get version by introspecting package information.
|
||||
|
||||
OCRmyPDF uses setuptools_scm to derive version from git tags.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from importlib.metadata import version as _package_version
|
||||
|
||||
PROGRAM_NAME = 'ocrmypdf'
|
||||
|
||||
# Official PEP 396
|
||||
__version__ = _package_version('ocrmypdf')
|
||||
+7
-7
@@ -14,7 +14,7 @@ from collections.abc import Iterable, Sequence
|
||||
from enum import IntEnum
|
||||
from io import IOBase
|
||||
from pathlib import Path
|
||||
from typing import AnyStr, BinaryIO
|
||||
from typing import BinaryIO
|
||||
from warnings import warn
|
||||
|
||||
import pluggy
|
||||
@@ -28,7 +28,7 @@ from ocrmypdf._validation import check_options
|
||||
from ocrmypdf.cli import ArgumentParser, get_parser
|
||||
from ocrmypdf.helpers import is_iterable_notstr
|
||||
|
||||
StrPath = Path | AnyStr
|
||||
StrPath = Path | str | bytes
|
||||
PathOrIO = BinaryIO | StrPath
|
||||
|
||||
# Installing plugins affects the global state of the Python interpreter,
|
||||
@@ -140,9 +140,9 @@ def configure_logging(
|
||||
|
||||
def _kwargs_to_cmdline(
|
||||
*, defer_kwargs: set[str], **kwargs
|
||||
) -> tuple[list[str], dict[str, AnyStr]]:
|
||||
) -> tuple[list[str | bytes], dict[str, str | bytes]]:
|
||||
"""Convert kwargs to command line arguments."""
|
||||
cmdline = []
|
||||
cmdline: list[str | bytes] = []
|
||||
deferred = {}
|
||||
for arg, val in kwargs.items():
|
||||
if val is None:
|
||||
@@ -279,7 +279,7 @@ def ocr( # noqa: D417
|
||||
fast_web_view: float | None = None,
|
||||
continue_on_soft_render_error: bool | None = None,
|
||||
invalidate_digital_signatures: bool | None = None,
|
||||
plugins: Iterable[StrPath] | None = None,
|
||||
plugins: Iterable[Path | str] | None = None,
|
||||
plugin_manager=None,
|
||||
keep_temporary_files: bool | None = None,
|
||||
progress_bar: bool | None = None,
|
||||
@@ -420,7 +420,7 @@ def _pdf_to_hocr( # noqa: D417
|
||||
continue_on_soft_render_error: bool | None = None,
|
||||
invalidate_digital_signatures: bool | None = None,
|
||||
plugin_manager=None,
|
||||
plugins: Sequence[StrPath] | None = None,
|
||||
plugins: Sequence[Path | str] | None = None,
|
||||
keep_temporary_files: bool | None = None,
|
||||
**kwargs,
|
||||
):
|
||||
@@ -491,7 +491,7 @@ def _hocr_to_ocr_pdf( # noqa: D417
|
||||
color_conversion_strategy: str | None = None,
|
||||
fast_web_view: float | None = None,
|
||||
plugin_manager=None,
|
||||
plugins: Sequence[StrPath] | None = None,
|
||||
plugins: Sequence[Path | str] | None = None,
|
||||
**kwargs,
|
||||
):
|
||||
"""Run OCRmyPDF on a work folder and produce an output PDF.
|
||||
|
||||
@@ -129,7 +129,7 @@ def generate_pdfa(
|
||||
):
|
||||
"""Generate a PDF/A from the list of PDF pages and PDF/A metadata."""
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[*pdf_pages, pdfmark],
|
||||
pdf_pages=[pdfmark, *pdf_pages],
|
||||
output_file=output_file,
|
||||
compression=context.options.pdfa_image_compression,
|
||||
color_conversion_strategy=context.options.color_conversion_strategy,
|
||||
|
||||
@@ -14,6 +14,7 @@ from ocrmypdf import hookimpl
|
||||
from ocrmypdf._exec import tesseract
|
||||
from ocrmypdf._jobcontext import PageContext
|
||||
from ocrmypdf.cli import numeric, str_to_int
|
||||
from ocrmypdf.exceptions import BadArgsError, MissingDependencyError
|
||||
from ocrmypdf.helpers import clamp
|
||||
from ocrmypdf.imageops import calculate_downsample, downsample_image
|
||||
from ocrmypdf.pluginspec import OcrEngine
|
||||
@@ -144,6 +145,12 @@ def check_options(options):
|
||||
need_version='4.1.1', # Ubuntu 22.04 version (also 20.04)
|
||||
version_parser=tesseract.TesseractVersion,
|
||||
)
|
||||
tess_version = tesseract.version()
|
||||
if tess_version == tesseract.TesseractVersion('5.4.0'):
|
||||
raise MissingDependencyError(
|
||||
"Tesseract 5.4.0 is not supported due to regressions in this version. "
|
||||
"Please upgrade to a newer or supported older version."
|
||||
)
|
||||
|
||||
# Decide on what renderer to use
|
||||
if options.pdf_renderer == 'auto':
|
||||
@@ -164,6 +171,14 @@ def check_options(options):
|
||||
"The --tesseract-pagesegmode argument you select will disable OCR. "
|
||||
"This may cause processing to fail."
|
||||
)
|
||||
DENIED_LANGUAGES = {'equ', 'osd'}
|
||||
if DENIED_LANGUAGES & set(options.languages):
|
||||
raise BadArgsError(
|
||||
"The following languages for Tesseract's internal use and should not "
|
||||
"be issued explicitly: "
|
||||
f"{', '.join(DENIED_LANGUAGES & set(options.languages))}\n"
|
||||
"Remove them from the -l/--language argument."
|
||||
)
|
||||
|
||||
|
||||
@hookimpl
|
||||
|
||||
+3
-2
@@ -9,7 +9,8 @@ import argparse
|
||||
from collections.abc import Callable, Mapping
|
||||
from typing import Any, TypeVar
|
||||
|
||||
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
||||
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
from ocrmypdf._defaults import PROGRAM_NAME as _PROGRAM_NAME
|
||||
from ocrmypdf._version import __version__ as _VERSION
|
||||
|
||||
T = TypeVar('T', int, float)
|
||||
@@ -403,7 +404,7 @@ Online documentation is located at:
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--rotate-pages-threshold',
|
||||
default=14.0,
|
||||
default=DEFAULT_ROTATE_PAGES_THRESHOLD,
|
||||
type=numeric(float, 0, 1000),
|
||||
metavar='CONFIDENCE',
|
||||
help="Only rotate pages when confidence is above this value (arbitrary "
|
||||
|
||||
+17
-1
@@ -268,7 +268,9 @@ def check_pdf(input_file: Path) -> bool:
|
||||
return False
|
||||
else:
|
||||
with pdf:
|
||||
messages = pdf.check()
|
||||
with warnings.catch_warnings():
|
||||
warnings.filterwarnings('ignore', message=r'pikepdf.*JBIG2.*')
|
||||
messages = pdf.check()
|
||||
success = True
|
||||
for msg in messages:
|
||||
if 'error' in msg.lower():
|
||||
@@ -333,3 +335,17 @@ def pikepdf_enable_mmap() -> None:
|
||||
)
|
||||
except AttributeError:
|
||||
log.debug("pikepdf mmap not available")
|
||||
|
||||
|
||||
def running_in_docker() -> bool:
|
||||
"""Returns True if we seem to be running in a Docker container."""
|
||||
return Path('/.dockerenv').exists()
|
||||
|
||||
|
||||
def running_in_snap() -> bool:
|
||||
"""Returns True if we seem to be running in a Snap container."""
|
||||
try:
|
||||
cgroup_text = Path('/proc/self/cgroup').read_text()
|
||||
return 'snap.ocrmypdf' in cgroup_text
|
||||
except FileNotFoundError:
|
||||
return False
|
||||
|
||||
@@ -61,15 +61,25 @@ class HocrTransform:
|
||||
"""A class for converting documents from the hOCR format.
|
||||
|
||||
For details of the hOCR format, see:
|
||||
http://kba.cloud/hocr-spec/.
|
||||
http://kba.github.io/hocr-spec/1.2/.
|
||||
"""
|
||||
|
||||
box_pattern = re.compile(r'bbox (\d+) (\d+) (\d+) (\d+)')
|
||||
box_pattern = re.compile(
|
||||
r'''
|
||||
bbox \s+
|
||||
(\d+) \s+ # left: uint
|
||||
(\d+) \s+ # top: uint
|
||||
(\d+) \s+ # right: uint
|
||||
(\d+) # bottom: uint
|
||||
''',
|
||||
re.VERBOSE,
|
||||
)
|
||||
baseline_pattern = re.compile(
|
||||
r'''
|
||||
baseline \s+
|
||||
([\-\+]?\d*\.?\d*) \s+ # +/- decimal float
|
||||
([\-\+]?\d+) # +/- int''',
|
||||
([\-\+]?\d+) # +/- int
|
||||
''',
|
||||
re.VERBOSE,
|
||||
)
|
||||
|
||||
@@ -117,15 +127,12 @@ class HocrTransform:
|
||||
# Stop after first div that has page coordinates
|
||||
break
|
||||
|
||||
def _get_element_text(self, element: Element):
|
||||
def _get_element_text(self, element: Element) -> str:
|
||||
"""Return the textual content of the element and its children."""
|
||||
text = ''
|
||||
if element.text is not None:
|
||||
text += element.text
|
||||
text = element.text if element.text is not None else ''
|
||||
for child in element:
|
||||
text += self._get_element_text(child)
|
||||
if element.tail is not None:
|
||||
text += element.tail
|
||||
text += element.tail if element.tail is not None else ''
|
||||
return text
|
||||
|
||||
@classmethod
|
||||
@@ -286,7 +293,13 @@ class HocrTransform:
|
||||
line_box = self.element_coordinates(line)
|
||||
if not line_box:
|
||||
return
|
||||
assert line_box.ury > line_box.lly # lly is top, ury is bottom
|
||||
if line_box.ury <= line_box.lly:
|
||||
log.error(
|
||||
"line box is invalid so we cannot render it: box=%s text=%s",
|
||||
line_box,
|
||||
self._get_element_text(line),
|
||||
)
|
||||
return
|
||||
|
||||
self._debug_draw_line_bbox(canvas, line_box)
|
||||
|
||||
@@ -344,7 +357,7 @@ class HocrTransform:
|
||||
line_matrix: Matrix,
|
||||
text: Text,
|
||||
fontsize: float,
|
||||
elem: Element,
|
||||
elem: Element | None,
|
||||
next_elem: Element | None,
|
||||
text_direction: TextDirection,
|
||||
inject_word_breaks: bool,
|
||||
@@ -421,7 +434,7 @@ class HocrTransform:
|
||||
if ocr_par is None:
|
||||
continue
|
||||
canvas.do.rect(
|
||||
ocr_par.llx, ocr_par.lly, ocr_par.width, ocr_par.height, fill=0
|
||||
ocr_par.llx, ocr_par.lly, ocr_par.width, ocr_par.height, fill=False
|
||||
)
|
||||
|
||||
def _debug_draw_line_bbox(self, canvas: Canvas, line_box: Rectangle, color=BLUE):
|
||||
@@ -430,7 +443,7 @@ class HocrTransform:
|
||||
return
|
||||
with canvas.do.save_state():
|
||||
canvas.do.stroke_color(color).line_width(0.15).rect(
|
||||
line_box.llx, line_box.lly, line_box.width, line_box.height, fill=0
|
||||
line_box.llx, line_box.lly, line_box.width, line_box.height, fill=False
|
||||
)
|
||||
|
||||
def _debug_draw_word_triangle(
|
||||
@@ -454,7 +467,7 @@ class HocrTransform:
|
||||
return
|
||||
with canvas.do.save_state():
|
||||
canvas.do.stroke_color(color).line_width(line_width).rect(
|
||||
box.llx, box.lly, box.width, box.height, fill=0
|
||||
box.llx, box.lly, box.width, box.height, fill=False
|
||||
)
|
||||
|
||||
def _debug_draw_space_bbox(
|
||||
@@ -465,7 +478,7 @@ class HocrTransform:
|
||||
return
|
||||
with canvas.do.save_state():
|
||||
canvas.do.fill_color(color).line_width(line_width).rect(
|
||||
box.llx, box.lly, box.width, box.height, fill=1
|
||||
box.llx, box.lly, box.width, box.height, fill=True
|
||||
)
|
||||
|
||||
def _debug_draw_baseline(
|
||||
|
||||
@@ -28,6 +28,7 @@ from pikepdf import (
|
||||
Stream,
|
||||
UnsupportedImageTypeError,
|
||||
)
|
||||
from pikepdf.models.image import HifiPrintImageNotTranscodableError
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||
@@ -74,10 +75,15 @@ def extract_image_filter(
|
||||
"""Determine if an image is extractable."""
|
||||
if image.Subtype != Name.Image:
|
||||
return None
|
||||
if image.Length < 100:
|
||||
if not isinstance(image.Length, int) or image.Length < 100:
|
||||
log.debug(f"xref {xref}: skipping image with small stream size")
|
||||
return None
|
||||
if image.Width < 8 or image.Height < 8: # Issue 732
|
||||
if (
|
||||
not isinstance(image.Width, int)
|
||||
or not isinstance(image.Height, int)
|
||||
or image.Width < 8
|
||||
or image.Height < 8
|
||||
): # Issue 732
|
||||
log.debug(f"xref {xref}: skipping image with unusually small dimensions")
|
||||
return None
|
||||
|
||||
@@ -153,7 +159,10 @@ def extract_image_jbig2(
|
||||
imgname = root / f'{xref:08d}'
|
||||
with imgname.open('wb') as f:
|
||||
ext = pim.extract_to(stream=f)
|
||||
imgname.rename(imgname.with_suffix(ext))
|
||||
# Rename the file so it has .prejbig2.ext extension
|
||||
# Making it unique avoids problems with Windows if the
|
||||
# same image is extracted multiple times
|
||||
imgname.rename(imgname.with_suffix(".prejbig2" + ext))
|
||||
except NotImplementedError as e:
|
||||
if '/Decode' in str(e):
|
||||
log.debug(
|
||||
@@ -169,7 +178,7 @@ def extract_image_jbig2(
|
||||
pim.obj.ColorSpace = colorspace
|
||||
else:
|
||||
del pim.obj.ColorSpace
|
||||
return XrefExt(xref, ext)
|
||||
return XrefExt(xref, ".prejbig2" + ext)
|
||||
return None
|
||||
|
||||
|
||||
@@ -200,7 +209,7 @@ def extract_image_generic(
|
||||
with imgname.open('wb') as f:
|
||||
ext = pim.extract_to(stream=f)
|
||||
imgname.rename(imgname.with_suffix(ext))
|
||||
except UnsupportedImageTypeError:
|
||||
except (UnsupportedImageTypeError, HifiPrintImageNotTranscodableError):
|
||||
return None
|
||||
return XrefExt(xref, ext)
|
||||
elif (
|
||||
@@ -256,6 +265,9 @@ def _find_image_xrefs_container(
|
||||
for _imname, image in dict(xobjs).items():
|
||||
if image.objgen[1] != 0:
|
||||
continue # Ignore images in an incremental PDF
|
||||
xref = Xref(image.objgen[0])
|
||||
if xref in include_xrefs or xref in exclude_xrefs:
|
||||
continue # Already processed
|
||||
if Name.Subtype in image and image.Subtype == Name.Form:
|
||||
# Recurse into Form XObjects
|
||||
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
||||
@@ -269,7 +281,6 @@ def _find_image_xrefs_container(
|
||||
depth + 1,
|
||||
)
|
||||
continue
|
||||
xref = Xref(image.objgen[0])
|
||||
if Name.SMask in image:
|
||||
# Ignore soft masks
|
||||
smask_xref = Xref(image.SMask.objgen[0])
|
||||
|
||||
@@ -12,7 +12,7 @@ import re
|
||||
import statistics
|
||||
from collections import defaultdict
|
||||
from collections.abc import Callable, Container, Iterable, Iterator, Mapping, Sequence
|
||||
from contextlib import contextmanager
|
||||
from contextlib import contextmanager, nullcontext
|
||||
from decimal import Decimal
|
||||
from enum import Enum, auto
|
||||
from functools import partial
|
||||
@@ -24,6 +24,7 @@ from warnings import warn
|
||||
|
||||
from pdfminer.layout import LTPage, LTTextBox
|
||||
from pikepdf import (
|
||||
Dictionary,
|
||||
Matrix,
|
||||
Name,
|
||||
Object,
|
||||
@@ -40,7 +41,12 @@ from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||
from ocrmypdf._progressbar import ProgressBar
|
||||
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
||||
from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap
|
||||
from ocrmypdf.pdfinfo.layout import LTStateAwareChar, get_page_analysis, get_text_boxes
|
||||
from ocrmypdf.pdfinfo.layout import (
|
||||
LTStateAwareChar,
|
||||
PdfMinerState,
|
||||
get_page_analysis,
|
||||
get_text_boxes,
|
||||
)
|
||||
|
||||
logger = logging.getLogger()
|
||||
|
||||
@@ -239,7 +245,13 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
# to do. Just pretend nothing happened, keep calm and carry on.
|
||||
warn("PDF graphics stack underflowed - PDF may be malformed")
|
||||
elif operator == 'cm':
|
||||
ctm = Matrix(operands) @ ctm
|
||||
try:
|
||||
ctm = Matrix(operands) @ ctm
|
||||
except ValueError:
|
||||
raise InputFileError(
|
||||
"PDF content stream is corrupt - this PDF is malformed. "
|
||||
"Use a PDF editor that is capable of visually inspecting the PDF."
|
||||
)
|
||||
elif operator == 'Do':
|
||||
image_name = operands[0]
|
||||
settings = XobjectSettings(
|
||||
@@ -363,8 +375,26 @@ class ImageInfo:
|
||||
pim = PdfImage(pdfimage)
|
||||
else:
|
||||
raise ValueError("Either pdfimage or inline must be set")
|
||||
|
||||
self._width = pim.width
|
||||
self._height = pim.height
|
||||
if (smask := pim.obj.get(Name.SMask, None)) is not None:
|
||||
# SMask is pretty much an alpha channel, but in PDF it's possible
|
||||
# for channel to have different dimensions than the image
|
||||
# itself. Some PDF writers use this to create a grayscale stencil
|
||||
# mask. For our purposes, the effective size is the size of the
|
||||
# larger component (image or smask).
|
||||
if isinstance(smask, Stream | Dictionary):
|
||||
self._width = max(smask.get(Name.Width, 0), self._width)
|
||||
self._height = max(smask.get(Name.Height, 0), self._height)
|
||||
if (mask := pim.obj.get(Name.Mask, None)) is not None:
|
||||
# If the image has a /Mask entry, it has an explicit mask.
|
||||
# /Mask can be a Stream or an Array. If it's a Stream,
|
||||
# use its /Width and /Height if they are larger than the main
|
||||
# image's.
|
||||
if isinstance(mask, Stream | Dictionary):
|
||||
self._width = max(mask.get(Name.Width, 0), self._width)
|
||||
self._height = max(mask.get(Name.Height, 0), self._height)
|
||||
|
||||
# If /ImageMask is true, then this image is a stencil mask
|
||||
# (Images that draw with this stencil mask will have a reference to
|
||||
@@ -468,9 +498,18 @@ class ImageInfo:
|
||||
def renderable(self) -> bool:
|
||||
"""Whether the image is renderable.
|
||||
|
||||
Some PDFs in the wild have invalid images that are not renderable.
|
||||
Some PDFs in the wild have invalid images that are not renderable,
|
||||
due to unusual dimensions.
|
||||
|
||||
Stencil masks are not also not renderable, since they are not
|
||||
drawn, but rather they control how rendering happens.
|
||||
"""
|
||||
return self.dpi.is_finite and self.width >= 0 and self.height >= 0
|
||||
return (
|
||||
self.dpi.is_finite
|
||||
and self.width >= 0
|
||||
and self.height >= 0
|
||||
and self.type_ != 'stencil'
|
||||
)
|
||||
|
||||
@property
|
||||
def dpi(self) -> Resolution:
|
||||
@@ -485,7 +524,7 @@ class ImageInfo:
|
||||
"""Physical area of the image in square inches."""
|
||||
if not self.renderable:
|
||||
return 0.0
|
||||
return float(self.width * self.dpi.x * self.height * self.dpi.y)
|
||||
return float((self.width / self.dpi.x) * (self.height / self.dpi.y))
|
||||
|
||||
def __repr__(self):
|
||||
"""Return a string representation of the image."""
|
||||
@@ -567,7 +606,7 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
||||
xobjs = resources[Name.XObject].as_dict()
|
||||
for xobj in xobjs:
|
||||
candidate = xobjs[xobj]
|
||||
if candidate is None or candidate[Name.Subtype] != Name.Form:
|
||||
if candidate is None or candidate.get(Name.Subtype) != Name.Form:
|
||||
continue
|
||||
|
||||
form_xobject = candidate
|
||||
@@ -668,13 +707,13 @@ def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) ->
|
||||
|
||||
|
||||
def simplify_textboxes(
|
||||
miner: LTPage, textbox_getter: Callable[[LTPage], Iterator[LTTextBox]]
|
||||
miner_page: LTPage, textbox_getter: Callable[[LTPage], Iterator[LTTextBox]]
|
||||
) -> Iterator[TextboxInfo]:
|
||||
"""Extract only limited content from text boxes.
|
||||
|
||||
We do this to save memory and ensure that our objects are pickleable.
|
||||
"""
|
||||
for box in textbox_getter(miner):
|
||||
for box in textbox_getter(miner_page):
|
||||
first_line = box._objs[0] # pylint: disable=protected-access
|
||||
first_char = first_line._objs[0] # pylint: disable=protected-access
|
||||
if not isinstance(first_char, LTStateAwareChar):
|
||||
@@ -721,9 +760,12 @@ def _pdf_pageinfo_sync(
|
||||
infile: Path,
|
||||
check_pages: Container[int],
|
||||
detailed_analysis: bool,
|
||||
miner_state: PdfMinerState | None,
|
||||
) -> PageInfo:
|
||||
with _pdf_pageinfo_sync_pdf(thread_pdf, infile) as pdf:
|
||||
return PageInfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
||||
return PageInfo(
|
||||
pdf, pageno, infile, check_pages, detailed_analysis, miner_state
|
||||
)
|
||||
|
||||
|
||||
def _pdf_pageinfo_concurrent(
|
||||
@@ -735,6 +777,7 @@ def _pdf_pageinfo_concurrent(
|
||||
progbar,
|
||||
check_pages,
|
||||
detailed_analysis: bool = False,
|
||||
miner_state: PdfMinerState | None = None,
|
||||
) -> Sequence[PageInfo | None]:
|
||||
pages: list[PageInfo | None] = [None] * len(pdf.pages)
|
||||
|
||||
@@ -766,7 +809,8 @@ def _pdf_pageinfo_concurrent(
|
||||
initial_pdf = pdf if use_threads else None
|
||||
|
||||
contexts = (
|
||||
(n, initial_pdf, infile, check_pages, detailed_analysis) for n in range(total)
|
||||
(n, initial_pdf, infile, check_pages, detailed_analysis, miner_state)
|
||||
for n in range(total)
|
||||
)
|
||||
assert n_workers == 1 if use_threads else n_workers >= 1, "Not multithreadable"
|
||||
logger.debug(
|
||||
@@ -833,12 +877,15 @@ class PageInfo:
|
||||
infile: PathLike,
|
||||
check_pages: Container[int],
|
||||
detailed_analysis: bool = False,
|
||||
miner_state: PdfMinerState | None = None,
|
||||
):
|
||||
"""Initialize a PageInfo object."""
|
||||
self._pageno = pageno
|
||||
self._infile = infile
|
||||
self._detailed_analysis = detailed_analysis
|
||||
self._gather_pageinfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
||||
self._gather_pageinfo(
|
||||
pdf, pageno, infile, check_pages, detailed_analysis, miner_state
|
||||
)
|
||||
|
||||
def _gather_pageinfo(
|
||||
self,
|
||||
@@ -847,6 +894,7 @@ class PageInfo:
|
||||
infile: PathLike,
|
||||
check_pages: Container[int],
|
||||
detailed_analysis: bool,
|
||||
miner_state: PdfMinerState | None,
|
||||
):
|
||||
page: Page = pdf.pages[pageno]
|
||||
mediabox = [Decimal(d) for d in page.mediabox.as_list()]
|
||||
@@ -862,10 +910,11 @@ class PageInfo:
|
||||
check_this_page = pageno in check_pages
|
||||
|
||||
if check_this_page and detailed_analysis:
|
||||
pscript5_mode = str(pdf.docinfo.get(Name.Creator)).startswith('PScript5')
|
||||
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
||||
if miner is not None:
|
||||
self._textboxes = list(simplify_textboxes(miner, get_text_boxes))
|
||||
page_analysis = miner_state.get_page_analysis(pageno)
|
||||
if page_analysis is not None:
|
||||
self._textboxes = list(
|
||||
simplify_textboxes(page_analysis, get_text_boxes)
|
||||
)
|
||||
else:
|
||||
self._textboxes = []
|
||||
bboxes = (box.bbox for box in self._textboxes)
|
||||
@@ -1049,10 +1098,14 @@ class PageInfo:
|
||||
|
||||
Returns None if there is no meaningful DPI for the page.
|
||||
"""
|
||||
image_dpis = [
|
||||
image.dpi.to_scalar() for image in self._images if image.renderable
|
||||
]
|
||||
image_areas = [image.printed_area for image in self._images if image.renderable]
|
||||
image_dpis = []
|
||||
image_areas = []
|
||||
for image in self._images:
|
||||
if not image.renderable:
|
||||
continue
|
||||
image_dpis.append(image.dpi.to_scalar())
|
||||
image_areas.append(image.printed_area)
|
||||
|
||||
total_drawn_area = sum(image_areas)
|
||||
if total_drawn_area == 0:
|
||||
return None
|
||||
@@ -1065,7 +1118,6 @@ class PageInfo:
|
||||
|
||||
arg_max_dpi = image_dpis.index(max_dpi)
|
||||
max_area_ratio = image_areas[arg_max_dpi] / total_drawn_area
|
||||
|
||||
return PageResolutionProfile(
|
||||
weighted_dpi,
|
||||
max_dpi,
|
||||
@@ -1115,16 +1167,26 @@ class PdfInfo:
|
||||
with Pdf.open(infile) as pdf:
|
||||
if pdf.is_encrypted:
|
||||
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
||||
self._pages = _pdf_pageinfo_concurrent(
|
||||
pdf,
|
||||
executor,
|
||||
max_workers,
|
||||
use_threads,
|
||||
infile,
|
||||
progbar,
|
||||
check_pages=check_pages,
|
||||
detailed_analysis=detailed_analysis,
|
||||
pscript5_mode = str(pdf.docinfo.get(Name.Creator, "")).startswith(
|
||||
'PScript5'
|
||||
)
|
||||
self._miner_state = (
|
||||
PdfMinerState(infile, pscript5_mode)
|
||||
if detailed_analysis
|
||||
else nullcontext()
|
||||
)
|
||||
with self._miner_state as miner_state:
|
||||
self._pages = _pdf_pageinfo_concurrent(
|
||||
pdf,
|
||||
executor,
|
||||
max_workers,
|
||||
use_threads,
|
||||
infile,
|
||||
progbar,
|
||||
check_pages=check_pages,
|
||||
detailed_analysis=detailed_analysis,
|
||||
miner_state=miner_state,
|
||||
)
|
||||
self._needs_rendering = pdf.Root.get(Name.NeedsRendering, False)
|
||||
if Name.AcroForm in pdf.Root:
|
||||
if len(pdf.Root.AcroForm.get(Name.Fields, [])) > 0:
|
||||
|
||||
@@ -17,6 +17,8 @@ import pdfminer
|
||||
import pdfminer.encodingdb
|
||||
import pdfminer.pdfdevice
|
||||
import pdfminer.pdfinterp
|
||||
import pdfminer.psparser
|
||||
from deprecation import deprecated
|
||||
from pdfminer.converter import PDFLayoutAnalyzer
|
||||
from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox
|
||||
from pdfminer.pdfcolor import PDFColorSpace
|
||||
@@ -58,9 +60,10 @@ def pdfsimplefont__init__(
|
||||
|
||||
setattr(PDFSimpleFont, '__init__', pdfsimplefont__init__)
|
||||
|
||||
#
|
||||
# pdfminer patches when creator is PScript5.dll
|
||||
#
|
||||
# Patch pdfminer.six buffer size
|
||||
# The parser doesn't properly handle keyword tokens are split across the end of the
|
||||
# buffer, so increase the buffer size something far larger than will ever be seen.
|
||||
pdfminer.psparser.PSBaseParser.BUFSIZ = 256 * 1024 * 1024
|
||||
|
||||
|
||||
def pdftype3font__pscript5_get_height(self):
|
||||
@@ -287,6 +290,7 @@ def patch_pdfminer(pscript5_mode: bool):
|
||||
yield
|
||||
|
||||
|
||||
@deprecated(deprecated_in='16.6.0', details='Use PdfMinerState instead.')
|
||||
def get_page_analysis(
|
||||
infile: PathLike, pageno: int, pscript5_mode: bool
|
||||
) -> LTPage | None:
|
||||
@@ -317,6 +321,73 @@ def get_page_analysis(
|
||||
return dev.get_result()
|
||||
|
||||
|
||||
class PdfMinerState:
|
||||
"""Provide a context manager for using pdfminer.six.
|
||||
|
||||
This ensures that the file is closed. It also provides a cache of pages
|
||||
from the PDF so that they can be reused if needed, to improve performance.
|
||||
"""
|
||||
|
||||
def __init__(self, infile: Path, pscript5_mode: bool) -> None:
|
||||
"""Initialize the context manager.
|
||||
|
||||
Args:
|
||||
infile: The path to the PDF file to be analyzed.
|
||||
pscript5_mode: Whether the PDF was generated by PScript5.dll.
|
||||
"""
|
||||
self.infile = infile
|
||||
self.rman = pdfminer.pdfinterp.PDFResourceManager(caching=True)
|
||||
self.disable_boxes_flow = None
|
||||
self.page_cache: list[PDFPage] = []
|
||||
self.pscript5_mode = pscript5_mode
|
||||
self.file = None
|
||||
|
||||
def __enter__(self):
|
||||
"""Enter the context manager."""
|
||||
self.file = Path(self.infile).open('rb')
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
"""Exit the context manager."""
|
||||
if self.file:
|
||||
self.file.close()
|
||||
return True
|
||||
|
||||
def _load_page_cache(self):
|
||||
"""Load the page cache."""
|
||||
try:
|
||||
self.page_cache = list(PDFPage.get_pages(self.file))
|
||||
if not self.page_cache:
|
||||
raise InputFileError(
|
||||
"pdfminer did not find any pages in the input file."
|
||||
)
|
||||
for n, page in enumerate(self.page_cache):
|
||||
if page is None:
|
||||
raise InputFileError(
|
||||
f"pdfminer could not process page {n} (counting from 0)."
|
||||
)
|
||||
except PDFTextExtractionNotAllowed as e:
|
||||
raise EncryptedPdfError() from e
|
||||
|
||||
def get_page_analysis(self, pageno: int):
|
||||
"""Get the page analysis for a given page."""
|
||||
if not self.page_cache:
|
||||
self._load_page_cache()
|
||||
page = self.page_cache[pageno]
|
||||
dev = TextPositionTracker(
|
||||
self.rman,
|
||||
laparams=LAParams(
|
||||
all_texts=True, detect_vertical=True, boxes_flow=self.disable_boxes_flow
|
||||
),
|
||||
)
|
||||
interp = pdfminer.pdfinterp.PDFPageInterpreter(self.rman, dev)
|
||||
|
||||
with patch_pdfminer(self.pscript5_mode):
|
||||
interp.process_page(page)
|
||||
|
||||
return dev.get_result()
|
||||
|
||||
|
||||
def get_text_boxes(obj) -> Iterator[LTTextBox]:
|
||||
"""Get the text boxes attached to the current node."""
|
||||
for child in obj:
|
||||
|
||||
@@ -215,8 +215,10 @@ to have {found_version}. Please update this program.
|
||||
|
||||
OLD_VERSION_REQUIRED_FOR = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
||||
{required_for} arguments. If you omit these arguments, OCRmyPDF may be able to
|
||||
proceed. For best results, install the program.
|
||||
{required_for} arguments. {program} {found_version} is installed.
|
||||
|
||||
If you omit these arguments, OCRmyPDF may be able to
|
||||
proceed. For best results, update the program.
|
||||
'''
|
||||
|
||||
OSX_INSTALL_ADVICE = '''
|
||||
|
||||
@@ -24,11 +24,6 @@ def is_macos():
|
||||
return platform.system() == 'Darwin'
|
||||
|
||||
|
||||
def running_in_docker():
|
||||
# Docker creates a file named /.dockerenv in all supported versions
|
||||
return Path('/.dockerenv').exists()
|
||||
|
||||
|
||||
def have_unpaper():
|
||||
try:
|
||||
unpaper.version()
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from pikepdf import Array, Dictionary, Name, NameTree, Pdf
|
||||
|
||||
from ocrmypdf._annots import remove_broken_goto_annotations
|
||||
|
||||
|
||||
def test_remove_broken_goto_annotations(resources):
|
||||
with Pdf.open(resources / 'link.pdf') as pdf:
|
||||
assert not remove_broken_goto_annotations(pdf), "File should not be modified"
|
||||
|
||||
# Construct Dests nametree
|
||||
nt = NameTree.new(pdf)
|
||||
names = pdf.Root[Name.Names] = pdf.make_indirect(Dictionary())
|
||||
names[Name.Dests] = nt.obj
|
||||
# Create a broken named destination
|
||||
nt['Invalid'] = pdf.make_indirect(Dictionary())
|
||||
# Create a valid named destination
|
||||
nt['Valid'] = Array([pdf.pages[0].obj, Name.XYZ, 0, 0, 0])
|
||||
|
||||
pdf.pages[0].Annots[0].A.D = 'Missing'
|
||||
pdf.pages[1].Annots[0].A.D = 'Valid'
|
||||
|
||||
assert remove_broken_goto_annotations(pdf), "File should be modified"
|
||||
|
||||
assert Name.D not in pdf.pages[0].Annots[0].A
|
||||
assert Name.D in pdf.pages[1].Annots[0].A
|
||||
+29
-1
@@ -3,6 +3,7 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pickle
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
@@ -10,6 +11,7 @@ import pytest
|
||||
from pdfminer.high_level import extract_text
|
||||
|
||||
import ocrmypdf
|
||||
import ocrmypdf._pipelines
|
||||
import ocrmypdf.api
|
||||
|
||||
|
||||
@@ -35,7 +37,7 @@ def test_sidecar_stringio(resources: Path, outdir: Path, outpdf: Path):
|
||||
resources / 'ccitt.pdf',
|
||||
outpdf,
|
||||
plugins=['tests/plugins/tesseract_cache.py'],
|
||||
sidecar=s
|
||||
sidecar=s,
|
||||
)
|
||||
s.seek(0)
|
||||
assert b'the' in s.getvalue()
|
||||
@@ -75,3 +77,29 @@ def test_hocr_to_pdf_api(resources: Path, outdir: Path, outpdf: Path):
|
||||
text = extract_text(outpdf)
|
||||
assert 'hocr' in text and 'the' not in text
|
||||
|
||||
|
||||
def test_hocr_result_json():
|
||||
result = ocrmypdf._pipelines._common.HOCRResult(
|
||||
pageno=1,
|
||||
pdf_page_from_image=Path('a'),
|
||||
hocr=Path('b'),
|
||||
textpdf=Path('c'),
|
||||
orientation_correction=180,
|
||||
)
|
||||
assert (
|
||||
result.to_json()
|
||||
== '{"pageno": 1, "pdf_page_from_image": {"Path": "a"}, "hocr": {"Path": "b"}, '
|
||||
'"textpdf": {"Path": "c"}, "orientation_correction": 180}'
|
||||
)
|
||||
assert ocrmypdf._pipelines._common.HOCRResult.from_json(result.to_json()) == result
|
||||
|
||||
|
||||
def test_hocr_result_pickle():
|
||||
result = ocrmypdf._pipelines._common.HOCRResult(
|
||||
pageno=1,
|
||||
pdf_page_from_image=Path('a'),
|
||||
hocr=Path('b'),
|
||||
textpdf=Path('c'),
|
||||
orientation_correction=180,
|
||||
)
|
||||
assert result == pickle.loads(pickle.dumps(result))
|
||||
|
||||
@@ -8,7 +8,7 @@ from subprocess import run
|
||||
|
||||
import pytest
|
||||
|
||||
from .conftest import running_in_docker
|
||||
from ocrmypdf.helpers import running_in_docker
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
running_in_docker(),
|
||||
|
||||
@@ -13,8 +13,7 @@ import pytest
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf import helpers
|
||||
|
||||
from .conftest import running_in_docker
|
||||
from ocrmypdf.helpers import running_in_docker
|
||||
|
||||
needs_symlink = pytest.mark.skipif(os.name == 'nt', reason='needs posix symlink')
|
||||
windows_only = pytest.mark.skipif(os.name != 'nt', reason="Windows test")
|
||||
|
||||
+1
-1
@@ -18,6 +18,7 @@ from PIL import Image
|
||||
import ocrmypdf
|
||||
from ocrmypdf._exec import tesseract
|
||||
from ocrmypdf.exceptions import ExitCode, MissingDependencyError
|
||||
from ocrmypdf.helpers import running_in_docker
|
||||
from ocrmypdf.pdfa import file_claims_pdfa
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
||||
from ocrmypdf.subprocess import get_version
|
||||
@@ -29,7 +30,6 @@ from .conftest import (
|
||||
is_macos,
|
||||
run_ocrmypdf,
|
||||
run_ocrmypdf_api,
|
||||
running_in_docker,
|
||||
)
|
||||
|
||||
# pylint: disable=redefined-outer-name
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
import hypothesis
|
||||
import pytest
|
||||
@@ -208,12 +208,13 @@ def test_pages_issue700(monkeypatch, resources):
|
||||
monkeypatch.setattr(PDFPage, 'get_pages', get_no_pages)
|
||||
|
||||
with pytest.raises(InputFileError, match="pdfminer"):
|
||||
pdfinfo.PdfInfo(
|
||||
pi = pdfinfo.PdfInfo(
|
||||
resources / 'cardinal.pdf',
|
||||
detailed_analysis=True,
|
||||
progbar=False,
|
||||
max_workers=1,
|
||||
)
|
||||
pi._miner_state.get_page_analysis(0)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
|
||||
@@ -239,7 +239,9 @@ def make_rotate_test(imagefile, outdir, prefix, image_angle, page_angle):
|
||||
@pytest.mark.parametrize('image_angle', (0, 90, 180, 270))
|
||||
def test_rotate_page_level(image_angle, page_angle, resources, outdir, caplog):
|
||||
reference = make_rotate_test(resources / 'typewriter.png', outdir, 'ref', 0, 0)
|
||||
test = make_rotate_test(resources, outdir, 'test', image_angle, page_angle)
|
||||
test = make_rotate_test(
|
||||
resources / 'typewriter.png', outdir, 'test', image_angle, page_angle
|
||||
)
|
||||
out = test.with_suffix('.out.pdf')
|
||||
|
||||
exitcode = run_ocrmypdf_api(
|
||||
|
||||
@@ -23,4 +23,4 @@ def test_semfree(resources, outpdf):
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
assert exitcode == ExitCode.ok
|
||||
assert exitcode in (ExitCode.ok, ExitCode.pdfa_conversion_failed)
|
||||
|
||||
@@ -13,9 +13,9 @@ import pytest
|
||||
|
||||
from ocrmypdf import pdfinfo
|
||||
from ocrmypdf._exec import tesseract
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.exceptions import BadArgsError, ExitCode, MissingDependencyError
|
||||
|
||||
from .conftest import check_ocrmypdf
|
||||
from .conftest import check_ocrmypdf, run_ocrmypdf_api
|
||||
|
||||
# pylint: disable=redefined-outer-name
|
||||
|
||||
@@ -144,3 +144,10 @@ def test_tesseract_log_output_raises(caplog):
|
||||
with pytest.raises(tesseract.TesseractConfigError):
|
||||
tesseract.tesseract_log_output(b'parameter not found: moo')
|
||||
assert 'not found' in caplog.text
|
||||
|
||||
|
||||
def test_blocked_language(resources, no_outpdf):
|
||||
infile = resources / 'masks.pdf'
|
||||
for bad_lang in ['osd', 'equ']:
|
||||
with pytest.raises(BadArgsError):
|
||||
run_ocrmypdf_api(infile, no_outpdf, '-l', bad_lang)
|
||||
|
||||
+14
-1
@@ -5,7 +5,7 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from os import fspath
|
||||
from unittest.mock import patch
|
||||
from unittest.mock import Mock, patch
|
||||
|
||||
import pytest
|
||||
from packaging.version import Version
|
||||
@@ -48,6 +48,19 @@ def test_old_unpaper(resources, no_outpdf):
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
def test_unpaper_version_chatter(resources, no_outpdf):
|
||||
input_ = fspath(resources / "c02-22.pdf")
|
||||
output = fspath(no_outpdf)
|
||||
|
||||
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
|
||||
with patch("ocrmypdf.subprocess.run") as mock:
|
||||
mock.return_value = Mock(stdout='Warning: using insecure memory!\n7.0.0\n')
|
||||
|
||||
with pytest.raises(MissingDependencyError):
|
||||
check_options(options, pm)
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
@needs_unpaper
|
||||
def test_clean(resources, outpdf):
|
||||
check_ocrmypdf(
|
||||
|
||||
Reference in New Issue
Block a user