Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fbaad570c7 | ||
|
|
f974e3b3c1 | ||
|
|
46b49cc176 | ||
|
|
5256e74d0c | ||
|
|
621d6a0b89 | ||
|
|
08be7c8bbe | ||
|
|
980a5472b6 | ||
|
|
51c618e357 | ||
|
|
4dde3786c2 | ||
|
|
d544342602 | ||
|
|
fac91fca2a | ||
|
|
6edf756849 | ||
|
|
4fb1bb4de6 | ||
|
|
6a8eb7daaa | ||
|
|
0544d06c3d | ||
|
|
34c285c9ac | ||
|
|
2f53b27651 | ||
|
|
772677746b | ||
|
|
f0bad87ea6 | ||
|
|
44e71f8c14 | ||
|
|
964b30ca26 | ||
|
|
214a333e2d | ||
|
|
ec6401ab57 | ||
|
|
cbc5e8ce8d | ||
|
|
a1c4cfe8f1 | ||
|
|
3a721e6578 | ||
|
|
e6b716cdde | ||
|
|
02c39998b8 | ||
|
|
0774bc7f14 | ||
|
|
c6a98b3d0b | ||
|
|
981bbf1105 | ||
|
|
2b0c6cfd40 | ||
|
|
59f6bc8306 | ||
|
|
653c4ffb45 | ||
|
|
d947ca258e | ||
|
|
d5ff7f7db9 | ||
|
|
579cef3649 | ||
|
|
cb2f090c60 | ||
|
|
f3d6387bca | ||
|
|
abf9729c61 | ||
|
|
442e9c9f0d | ||
|
|
397fad249d | ||
|
|
9a3c5a3f7c | ||
|
|
950c700274 | ||
|
|
26432c38a9 | ||
|
|
28be50136c | ||
|
|
0c62f2de5d | ||
|
|
5caf654f22 | ||
|
|
205593445e | ||
|
|
f25fb8c63a | ||
|
|
99c78650b6 | ||
|
|
69355886a8 | ||
|
|
08e89e2dbe | ||
|
|
0e013df161 | ||
|
|
9ba4e3ab46 | ||
|
|
5fdcb7602b | ||
|
|
b4db1b741f | ||
|
|
7a8cc21e31 | ||
|
|
0674829d8f | ||
|
|
315aa0474b | ||
|
|
df3451e779 | ||
|
|
3ba42802d1 | ||
|
|
d6342cb8c2 | ||
|
|
065bddbc6c | ||
|
|
067f429dde | ||
|
|
6895c2d70f | ||
|
|
686481982a | ||
|
|
a9e1d19b78 | ||
|
|
f95aa63718 | ||
|
|
855de287b2 | ||
|
|
feeb9f213f | ||
|
|
e7eb8fa805 | ||
|
|
8a747f005a | ||
|
|
16ab4a8b4e | ||
|
|
8d30cff4ef | ||
|
|
59d5b0d1bd | ||
|
|
9ec0745ab8 | ||
|
|
3a3635f7f9 | ||
|
|
6a746a1cbb | ||
|
|
906c130f96 |
@@ -64,6 +64,7 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
|||||||
img2pdf \
|
img2pdf \
|
||||||
libsm6 libxext6 libxrender-dev \
|
libsm6 libxext6 libxrender-dev \
|
||||||
pngquant \
|
pngquant \
|
||||||
|
python-is-python3 \
|
||||||
tesseract-ocr \
|
tesseract-ocr \
|
||||||
tesseract-ocr-chi-sim \
|
tesseract-ocr-chi-sim \
|
||||||
tesseract-ocr-deu \
|
tesseract-ocr-deu \
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
FROM alpine:3.18 as base
|
FROM alpine:3.19.1 as base
|
||||||
|
|
||||||
ENV LANG=C.UTF-8
|
ENV LANG=C.UTF-8
|
||||||
ENV TZ=UTC
|
ENV TZ=UTC
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
name: Installation, packaging, dependencies
|
name: Installation, packaging, dependencies
|
||||||
description: Installation, packages, dependencies, "nothing works", test suite failures...
|
description: Installation, packages, dependencies, "nothing works", test suite failures...
|
||||||
title: "[Bug]: "
|
title: "[Bug]: "
|
||||||
labels: ["bug", "triage"]
|
labels: ["triage"]
|
||||||
assignees:
|
assignees:
|
||||||
- jbarlow83
|
- jbarlow83
|
||||||
body:
|
body:
|
||||||
@@ -24,7 +24,7 @@ body:
|
|||||||
- type: dropdown
|
- type: dropdown
|
||||||
id: packaging-system
|
id: packaging-system
|
||||||
attributes:
|
attributes:
|
||||||
label: Where are you installing from?
|
label: Where are you installing/running from?
|
||||||
multiple: true
|
multiple: true
|
||||||
options:
|
options:
|
||||||
- PyPI (pip, poetry, pipx, etc.)
|
- PyPI (pip, poetry, pipx, etc.)
|
||||||
@@ -37,6 +37,11 @@ body:
|
|||||||
- source build
|
- source build
|
||||||
validations:
|
validations:
|
||||||
required: true
|
required: true
|
||||||
|
- type: input
|
||||||
|
id: version
|
||||||
|
attributes:
|
||||||
|
label: OCRmyPDF version
|
||||||
|
description: Paste "ocrmypdf --version" here
|
||||||
- type: dropdown
|
- type: dropdown
|
||||||
id: operating-system
|
id: operating-system
|
||||||
attributes:
|
attributes:
|
||||||
@@ -47,6 +52,18 @@ body:
|
|||||||
- Windows
|
- Windows
|
||||||
- macOS
|
- macOS
|
||||||
- BSD
|
- BSD
|
||||||
|
- type: input
|
||||||
|
id: os_version
|
||||||
|
attributes:
|
||||||
|
label: Operating system details and version
|
||||||
|
- type: checkboxes
|
||||||
|
attributes:
|
||||||
|
label: Simple sanity checks
|
||||||
|
description: Select all that apply
|
||||||
|
options:
|
||||||
|
- label: Operating system is currently supported by its vendor (not end of life)
|
||||||
|
- label: Python version is compatible with OCRmyPDF
|
||||||
|
- label: This issue is not about a specific input file
|
||||||
- type: textarea
|
- type: textarea
|
||||||
id: logs
|
id: logs
|
||||||
attributes:
|
attributes:
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
name: Problem with specific file
|
name: Problem with specific file
|
||||||
description: Something went wrong while trying to OCR a specific file
|
description: Something went wrong while trying to OCR a specific file
|
||||||
title: "[Bug]: "
|
title: "[Bug]: "
|
||||||
labels: ["bug", "triage"]
|
labels: ["triage"]
|
||||||
assignees:
|
assignees:
|
||||||
- jbarlow83
|
- jbarlow83
|
||||||
body:
|
body:
|
||||||
@@ -39,7 +39,7 @@ body:
|
|||||||
causing the issue. There's really no substitute for a test file.
|
causing the issue. There's really no substitute for a test file.
|
||||||
|
|
||||||
We understand files may contain personal or sensitive information. Here are some options:
|
We understand files may contain personal or sensitive information. Here are some options:
|
||||||
- Try reproducing the issue with a file from the test suite. (See tests/resources)
|
- Try reproducing the issue with a file from the OCRmyPDF test suite. (See tests/resources)
|
||||||
- Try to create another file in the same way as your private file.
|
- Try to create another file in the same way as your private file.
|
||||||
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
||||||
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
||||||
|
|||||||
@@ -0,0 +1,83 @@
|
|||||||
|
name: Problem with third party app that uses OCRmyPDF
|
||||||
|
description: |
|
||||||
|
For PDF generation issues with third party software such as Paperless-ngx that
|
||||||
|
uses OCRmyPDF to perform OCR or generate PDFs.
|
||||||
|
title: "[3rdparty]: "
|
||||||
|
labels: ["triage"]
|
||||||
|
assignees:
|
||||||
|
- jbarlow83
|
||||||
|
body:
|
||||||
|
- type: markdown
|
||||||
|
attributes:
|
||||||
|
value: |
|
||||||
|
Thanks for taking the time to describe this issue with a particular file
|
||||||
|
and third party app.
|
||||||
|
|
||||||
|
If you are comfortable using OCRmyPDF, please trying to install OCRmyPDF,
|
||||||
|
run it on your file, and see if it works. It's easier for everyone
|
||||||
|
if you can confirm that the issue occurs with OCRmyPDF and not with
|
||||||
|
the third party app.
|
||||||
|
- type: checkboxes
|
||||||
|
attributes:
|
||||||
|
label: Simple sanity checks
|
||||||
|
description: Select all that apply
|
||||||
|
options:
|
||||||
|
- label: This is an issue with an app that uses OCRmyPDF for OCR
|
||||||
|
- label: I am using a recent version of the third party app
|
||||||
|
- label: I will include a file that reproduces the issuse
|
||||||
|
- type: input
|
||||||
|
id: thirdparty-app-name-version
|
||||||
|
attributes:
|
||||||
|
label: Third party app name and version
|
||||||
|
description: e.g. Paperless-ngx 2.9.0
|
||||||
|
- type: textarea
|
||||||
|
id: what-happened
|
||||||
|
attributes:
|
||||||
|
label: Describe the bug
|
||||||
|
description: A clear and concise description of what the bug is.
|
||||||
|
placeholder: Tell us what you see!
|
||||||
|
validations:
|
||||||
|
required: true
|
||||||
|
- type: textarea
|
||||||
|
id: reproduce
|
||||||
|
attributes:
|
||||||
|
label: Steps to reproduce
|
||||||
|
description: Please include steps to reproduce.
|
||||||
|
value: |
|
||||||
|
1. Import attached file into Paperless-ngx
|
||||||
|
2. Trigger OCR
|
||||||
|
3. Check log file
|
||||||
|
4. ...
|
||||||
|
render: plain text
|
||||||
|
- type: textarea
|
||||||
|
id: files
|
||||||
|
attributes:
|
||||||
|
label: Files
|
||||||
|
description: |
|
||||||
|
Please attach the input and output files, or any screenshots that may be helpful.
|
||||||
|
|
||||||
|
If you cannot provide a test file, we probably won't be able to help with the issue.
|
||||||
|
PDF is a complex file format, and there may be technical details in the PDF that are
|
||||||
|
causing the issue. There's really no substitute for a test file.
|
||||||
|
|
||||||
|
We understand files may contain personal or sensitive information. Here are some options:
|
||||||
|
- Try reproducing the issue with a file from the test suite. (See tests/resources)
|
||||||
|
- Try to create another file in the same way as your private file.
|
||||||
|
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
||||||
|
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
||||||
|
omits personal information.
|
||||||
|
placeholder: |
|
||||||
|
Drag and drop files here.
|
||||||
|
- type: input
|
||||||
|
id: version
|
||||||
|
attributes:
|
||||||
|
label: OCRmyPDF version
|
||||||
|
description: Paste "ocrmypdf --version" here
|
||||||
|
placeholder: ocrmypdf --version
|
||||||
|
- type: textarea
|
||||||
|
id: logs
|
||||||
|
attributes:
|
||||||
|
label: Relevant log output
|
||||||
|
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||||
|
placeholder: Run OCRmyPDF with verbosity `-v1` to get more detailed logging output.
|
||||||
|
render: plain text
|
||||||
@@ -32,8 +32,8 @@ jobs:
|
|||||||
- os: ubuntu-latest
|
- os: ubuntu-latest
|
||||||
python: "3.12"
|
python: "3.12"
|
||||||
tesseract5: true
|
tesseract5: true
|
||||||
# - os: ubuntu-latest
|
- os: ubuntu-latest
|
||||||
# python: "pypy3.10"
|
python: "pypy3.10"
|
||||||
|
|
||||||
env:
|
env:
|
||||||
OS: ${{ matrix.os }}
|
OS: ${{ matrix.os }}
|
||||||
@@ -61,6 +61,7 @@ jobs:
|
|||||||
sudo apt-get install -y --no-install-recommends \
|
sudo apt-get install -y --no-install-recommends \
|
||||||
curl \
|
curl \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
|
jbig2dec \
|
||||||
img2pdf \
|
img2pdf \
|
||||||
libexempi8 \
|
libexempi8 \
|
||||||
libffi-dev \
|
libffi-dev \
|
||||||
@@ -101,16 +102,19 @@ jobs:
|
|||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v4
|
uses: codecov/codecov-action@v4
|
||||||
|
env:
|
||||||
|
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
|
|
||||||
|
|
||||||
test_macos:
|
test_macos:
|
||||||
name: Test macOS
|
name: Test macOS
|
||||||
runs-on: ${{ matrix.os }}
|
runs-on: ${{ matrix.os }}
|
||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
os: [macos-latest]
|
os: [macos-latest, macos-13] # macos-latest is arm64, macos-13 is x86_64
|
||||||
python: ["3.10", "3.11", "3.12"]
|
python: ["3.10", "3.11", "3.12"]
|
||||||
|
|
||||||
env:
|
env:
|
||||||
@@ -131,7 +135,6 @@ jobs:
|
|||||||
ghostscript \
|
ghostscript \
|
||||||
jbig2enc \
|
jbig2enc \
|
||||||
openjpeg \
|
openjpeg \
|
||||||
openssl \
|
|
||||||
pngquant \
|
pngquant \
|
||||||
tesseract
|
tesseract
|
||||||
|
|
||||||
@@ -159,6 +162,8 @@ jobs:
|
|||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v4
|
uses: codecov/codecov-action@v4
|
||||||
|
env:
|
||||||
|
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
@@ -202,6 +207,8 @@ jobs:
|
|||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v4
|
uses: codecov/codecov-action@v4
|
||||||
|
env:
|
||||||
|
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
@@ -352,9 +359,6 @@ jobs:
|
|||||||
username: jbarlow83
|
username: jbarlow83
|
||||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||||
|
|
||||||
- name: Set up QEMU
|
|
||||||
uses: docker/setup-qemu-action@v3
|
|
||||||
|
|
||||||
- name: Set up Docker Buildx
|
- name: Set up Docker Buildx
|
||||||
id: buildx
|
id: buildx
|
||||||
uses: docker/setup-buildx-action@v3
|
uses: docker/setup-buildx-action@v3
|
||||||
@@ -366,6 +370,6 @@ jobs:
|
|||||||
run: |
|
run: |
|
||||||
docker buildx build \
|
docker buildx build \
|
||||||
--push \
|
--push \
|
||||||
--platform linux/amd64 \
|
--platform linux/amd64,linux/arm64 \
|
||||||
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \
|
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \
|
||||||
--file .docker/Dockerfile.alpine .
|
--file .docker/Dockerfile.alpine .
|
||||||
|
|||||||
@@ -70,6 +70,7 @@ Linux, Windows, macOS and FreeBSD are supported. Docker images are also availabl
|
|||||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||||
| Fedora | ``dnf install ocrmypdf`` |
|
| Fedora | ``dnf install ocrmypdf`` |
|
||||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||||
|
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||||
|
|||||||
+60
-6
@@ -118,15 +118,16 @@ OCR for huge images
|
|||||||
-------------------
|
-------------------
|
||||||
|
|
||||||
Tesseract has internal limits on the size
|
Tesseract has internal limits on the size
|
||||||
of images it will process. If you issue
|
of images it will process. By default,
|
||||||
``--tesseract-downsample-large-images``, OCRmyPDF will downsample images
|
``--tesseract-downsample-large-images`` is enabled, and OCRmyPDF will
|
||||||
to fit Tesseract limits. (The limits are usually entered only for scanned
|
downsample images to fit Tesseract limits. (The limits are usually encountered
|
||||||
images of oversized media, such as large maps or blueprints exceeding
|
only for scanned images of oversized media, such as large maps or blueprints exceeding
|
||||||
110 cm or 43 inches in either dimension, and at high DPI.)
|
110 cm or 43 inches in either dimension, and at high DPI.) This feature can disabled
|
||||||
|
using ``--no-tesseract-downsample-large-images``.
|
||||||
|
|
||||||
``--tesseract-downsample-above Npixels`` adjusts the threshold at which images
|
``--tesseract-downsample-above Npixels`` adjusts the threshold at which images
|
||||||
will be downsampled. By default, only images that exceed any of Tesseract's
|
will be downsampled. By default, only images that exceed any of Tesseract's
|
||||||
internal limits are downsampled.
|
internal limits are downsampled (32767 pixels on either dimension).
|
||||||
|
|
||||||
You will also need to set ``--tesseract-timeout`` high enough to allow
|
You will also need to set ``--tesseract-timeout`` high enough to allow
|
||||||
for processing.
|
for processing.
|
||||||
@@ -227,6 +228,59 @@ then run ocrmypdf as follows (along with any other desired arguments):
|
|||||||
Some combinations of control parameters will break Tesseract or break
|
Some combinations of control parameters will break Tesseract or break
|
||||||
assumptions that OCRmyPDF makes about Tesseract's output.
|
assumptions that OCRmyPDF makes about Tesseract's output.
|
||||||
|
|
||||||
|
Changing page segmentation mode
|
||||||
|
-------------------------------
|
||||||
|
|
||||||
|
The directive ``--tesseract-pagesegmode Nmode`` forwards the desired page segmentation
|
||||||
|
mode to Tesseract OCR. The default is 3.
|
||||||
|
|
||||||
|
Page segmentation can improve OCR results when you know that a PDF ought to be
|
||||||
|
analyzed a particular way, such as PDFs whose pages contain only a single line of
|
||||||
|
text. For the vast majority of users, changing the page segmentation mode will only
|
||||||
|
make things worse.
|
||||||
|
|
||||||
|
As of June 2024, the Tesseract page segmentation modes are:
|
||||||
|
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| ID | Description |
|
||||||
|
+=====+==================================================================================+
|
||||||
|
| 0 | Orientation and script detection (OSD) only. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 1 | Automatic page segmentation with OSD. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 2 | Automatic page segmentation, but no OSD, or OCR. (not implemented) |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 3 | Fully automatic page segmentation, but no OSD. (Default) |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 4 | Assume a single column of text of variable sizes. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 5 | Assume a single uniform block of vertically aligned text. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 6 | Assume a single uniform block of text. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 7 | Treat the image as a single text line. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 8 | Treat the image as a single word. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 9 | Treat the image as a single word in a circle. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 10 | Treat the image as a single character. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 11 | Sparse text. Find as much text as possible in no particular order. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 12 | Sparse text with OSD. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
| 13 | Raw line. Treat the image as a single text line, bypassing hacks that are |
|
||||||
|
| | Tesseract-specific. |
|
||||||
|
+-----+----------------------------------------------------------------------------------+
|
||||||
|
|
||||||
|
Modes 0, 1, 2, and 12 (all of those that enable orientation and script detection)
|
||||||
|
are not compatible with OCRmyPDF, which performs OSD in a separate step from OCR.
|
||||||
|
Their use may interfere with ``--rotate-pages`` and other features.
|
||||||
|
|
||||||
|
It is currently not possible to use advanced Tesseract OCR features, such as creating
|
||||||
|
OCR information, when using Tesseract through OCRmyPDF.
|
||||||
|
|
||||||
Changing the PDF renderer
|
Changing the PDF renderer
|
||||||
=========================
|
=========================
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -42,7 +42,7 @@ execute the image:
|
|||||||
- Architecture
|
- Architecture
|
||||||
- Description
|
- Description
|
||||||
* - ``jbarlow83/ocrmypdf-alpine``
|
* - ``jbarlow83/ocrmypdf-alpine``
|
||||||
- x86_64 only
|
- x86_64 and arm64
|
||||||
- Recommended image, based on Alpine Linux.
|
- Recommended image, based on Alpine Linux.
|
||||||
* - ``jbarlow83/ocrmypdf-ubuntu``
|
* - ``jbarlow83/ocrmypdf-ubuntu``
|
||||||
- x86_64 and arm64
|
- x86_64 and arm64
|
||||||
@@ -81,7 +81,7 @@ To start a Docker container (instance of the image):
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
docker tag jbarlow83/ocrmypdf ocrmypdf
|
docker tag jbarlow83/ocrmypdf-alpine ocrmypdf
|
||||||
docker run --rm -i ocrmypdf (... all other arguments here...) - -
|
docker run --rm -i ocrmypdf (... all other arguments here...) - -
|
||||||
|
|
||||||
For convenience, create a shell alias to hide the Docker command. It is
|
For convenience, create a shell alias to hide the Docker command. It is
|
||||||
|
|||||||
+31
-13
@@ -23,7 +23,9 @@ These platforms have one-liner installs:
|
|||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| Fedora | ``dnf install ocrmypdf tesseract-osd`` |
|
| Fedora | ``dnf install ocrmypdf tesseract-osd`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| macOS | ``brew install ocrmypdf`` |
|
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||||
|
+-------------------------------+-----------------------------------------+
|
||||||
|
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
@@ -99,12 +101,12 @@ For full details on version availability for your platform, check the
|
|||||||
Fedora
|
Fedora
|
||||||
------
|
------
|
||||||
|
|
||||||
.. |fedora-37| image:: https://repology.org/badge/version-for-repo/fedora_37/ocrmypdf.svg
|
|
||||||
:alt: Fedora 37
|
|
||||||
|
|
||||||
.. |fedora-38| image:: https://repology.org/badge/version-for-repo/fedora_38/ocrmypdf.svg
|
.. |fedora-38| image:: https://repology.org/badge/version-for-repo/fedora_38/ocrmypdf.svg
|
||||||
:alt: Fedora 38
|
:alt: Fedora 38
|
||||||
|
|
||||||
|
.. |fedora-39| image:: https://repology.org/badge/version-for-repo/fedora_39/ocrmypdf.svg
|
||||||
|
:alt: Fedora 39
|
||||||
|
|
||||||
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
||||||
:alt: Fedore Rawhide
|
:alt: Fedore Rawhide
|
||||||
|
|
||||||
@@ -113,7 +115,7 @@ Fedora
|
|||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |latest| |
|
| |latest| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |fedora-37| |fedora-38| |fedora-rawhide| |
|
| |fedora-38| |fedora-39| |fedora-rawhide| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
|
|
||||||
Users of Fedora may simply
|
Users of Fedora may simply
|
||||||
@@ -123,7 +125,7 @@ Users of Fedora may simply
|
|||||||
dnf install ocrmypdf tesseract-osd
|
dnf install ocrmypdf tesseract-osd
|
||||||
|
|
||||||
For full details on version availability, check the `Fedora Package
|
For full details on version availability, check the `Fedora Package
|
||||||
Tracker <https://apps.fedoraproject.org/packages/ocrmypdf>`__.
|
Tracker <https://packages.fedoraproject.org/pkgs/ocrmypdf/ocrmypdf/>`__.
|
||||||
|
|
||||||
If the version available for your platform is out of date, you could opt
|
If the version available for your platform is out of date, you could opt
|
||||||
to install the latest version from source. See `Installing HEAD revision
|
to install the latest version from source. See `Installing HEAD revision
|
||||||
@@ -135,7 +137,7 @@ from sources <#installing-head-revision-from-sources>`__.
|
|||||||
issues. OCRmyPDF works fine without it but will produce larger output
|
issues. OCRmyPDF works fine without it but will produce larger output
|
||||||
files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later
|
files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later
|
||||||
will automatically detect it on the ``PATH``. To add JBIG2 encoding,
|
will automatically detect it on the ``PATH``. To add JBIG2 encoding,
|
||||||
see `Installing the JBIG2 encoder <jbig2>`__.
|
see :ref:`Installing the JBIG2 encoder <jbig2>`.
|
||||||
|
|
||||||
.. _ubuntu-lts-latest:
|
.. _ubuntu-lts-latest:
|
||||||
|
|
||||||
@@ -160,7 +162,7 @@ and build ocrmypdf in virtual environment:
|
|||||||
|
|
||||||
python3.11 -m venv .venv
|
python3.11 -m venv .venv
|
||||||
|
|
||||||
To add JBIG2 encoding, see `Installing the JBIG2 encoder <jbig2>`__.
|
To add JBIG2 encoding, see :ref:`Installing the JBIG2 encoder <jbig2>`.
|
||||||
|
|
||||||
Note Fedora packages for language data haven't been branched for RHEL/EPEL, but you can get traineddata files directly from `tesseract
|
Note Fedora packages for language data haven't been branched for RHEL/EPEL, but you can get traineddata files directly from `tesseract
|
||||||
<https://github.com/tesseract-ocr/tessdata/>`__ and place them in ``/usr/share/tesseract/tessdata``.
|
<https://github.com/tesseract-ocr/tessdata/>`__ and place them in ``/usr/share/tesseract/tessdata``.
|
||||||
@@ -217,12 +219,12 @@ you are using a VM image, such as `the official Vagrant image
|
|||||||
be completed for you.
|
be completed for you.
|
||||||
|
|
||||||
Next you should install the `base-devel package group
|
Next you should install the `base-devel package group
|
||||||
<https://www.archlinux.org/groups/x86_64/base-devel/>`__. This includes the
|
<https://archlinux.org/packages/core/any/base-devel/>`__. This includes the
|
||||||
standard tooling needed to build packages, such as a compiler and binary tools.
|
standard tooling needed to build packages, such as a compiler and binary tools.
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
sudo pacman -S base-devel
|
sudo pacman -S --needed base-devel
|
||||||
|
|
||||||
Now you are ready to install the OCRmyPDF package.
|
Now you are ready to install the OCRmyPDF package.
|
||||||
|
|
||||||
@@ -260,7 +262,7 @@ page.
|
|||||||
<https://aur.archlinux.org/packages/jbig2enc-git/>`__ and may be installed
|
<https://aur.archlinux.org/packages/jbig2enc-git/>`__ and may be installed
|
||||||
using the same series of steps as for the installation OCRmyPDF AUR
|
using the same series of steps as for the installation OCRmyPDF AUR
|
||||||
package. Alternatively, it may be built manually from source following the
|
package. Alternatively, it may be built manually from source following the
|
||||||
instructions in `Installing the JBIG2 encoder <jbig2>`__. If JBIG2 is
|
instructions in :ref:`Installing the JBIG2 encoder <jbig2>`. If JBIG2 is
|
||||||
installed, OCRmyPDF 7.0.0 and later will automatically detect it.
|
installed, OCRmyPDF 7.0.0 and later will automatically detect it.
|
||||||
|
|
||||||
Alpine Linux
|
Alpine Linux
|
||||||
@@ -325,6 +327,22 @@ languages you can optionally install them all:
|
|||||||
|
|
||||||
brew install tesseract-lang # Optional: Install all language packs
|
brew install tesseract-lang # Optional: Install all language packs
|
||||||
|
|
||||||
|
MacPorts
|
||||||
|
--------
|
||||||
|
|
||||||
|
.. image:: https://img.shields.io/badge/dynamic/json?url=https%3A%2F%2Fports.macports.org%2Fapi%2Fv1%2Fports%2Focrmypdf%2F%3Fformat%3Djson&query=version&label=MacPorts
|
||||||
|
:alt: Macports Version Information
|
||||||
|
:target: https://ports.macports.org/port/ocrmypdf
|
||||||
|
|
||||||
|
OCRmyPDF is includes in MacPorts:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
sudo port install ocrmypdf
|
||||||
|
|
||||||
|
Note that while this will install tesseract you will need to install
|
||||||
|
the appropriate tesseract `language ports <https://ports.macports.org/search/?selected_facets=categories_exact%3Atextproc&installed_file=&q=tesseract&name=on>`__.
|
||||||
|
|
||||||
Manual installation on macOS
|
Manual installation on macOS
|
||||||
----------------------------
|
----------------------------
|
||||||
|
|
||||||
@@ -623,7 +641,7 @@ environment:
|
|||||||
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
|
|
||||||
Or, to install in `development
|
Or, to install in `development
|
||||||
mode <https://pythonhosted.org/setuptools/setuptools.html#development-mode>`__,
|
mode <https://packaging.python.org/en/latest/guides/distributing-packages-using-setuptools/#working-in-development-mode>`__,
|
||||||
allowing customization of OCRmyPDF, use the ``-e`` flag:
|
allowing customization of OCRmyPDF, use the ``-e`` flag:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
@@ -663,7 +681,7 @@ To install all of the development and test requirements:
|
|||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
python -m .venv
|
python -m venv .venv
|
||||||
source .venv/bin/activate
|
source .venv/bin/activate
|
||||||
cd OCRmyPDF
|
cd OCRmyPDF
|
||||||
pip install -e .[test]
|
pip install -e .[test]
|
||||||
|
|||||||
+17
-1
@@ -68,6 +68,22 @@ to what languages it should search for. Multiple languages can be
|
|||||||
requested using either ``-l eng+fra`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fra``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
|
Archlinux
|
||||||
|
------
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
# Display a list of all Tesseract language packs
|
||||||
|
pacman -Ss tesseract-data
|
||||||
|
|
||||||
|
# Install German language pack
|
||||||
|
pacman -S tesseract-data-deu
|
||||||
|
|
||||||
|
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||||
|
to what languages it should search for. Multiple languages can be
|
||||||
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
|
``-l eng -l fra``.
|
||||||
|
|
||||||
Gentoo
|
Gentoo
|
||||||
------
|
------
|
||||||
|
|
||||||
@@ -122,4 +138,4 @@ Custom language packs
|
|||||||
If you have fine-tuned or trained Tesseract and generated custom trained data, you can
|
If you have fine-tuned or trained Tesseract and generated custom trained data, you can
|
||||||
copy your ``customlang.traineddata`` file into your Tesseract "tessdata" folder, and
|
copy your ``customlang.traineddata`` file into your Tesseract "tessdata" folder, and
|
||||||
then use the ``-l customlang`` argument to tell OCRmyPDF to pass that language on to
|
then use the ``-l customlang`` argument to tell OCRmyPDF to pass that language on to
|
||||||
Tesseract.
|
Tesseract.
|
||||||
|
|||||||
+10
-1
@@ -64,4 +64,13 @@ installation documentation.
|
|||||||
|
|
||||||
If you maintain a Linux distribution that supports 32-bit x86 or ARM, OCRmyPDF
|
If you maintain a Linux distribution that supports 32-bit x86 or ARM, OCRmyPDF
|
||||||
should continue to work as long as all of its dependencies continue to be
|
should continue to work as long as all of its dependencies continue to be
|
||||||
available in 32-bit form. Please note we do not test on 32-bit platforms.
|
available in 32-bit form. Please note we do not test on 32-bit platforms.
|
||||||
|
|
||||||
|
HEIF/HEIC
|
||||||
|
---------
|
||||||
|
|
||||||
|
OCRmyPDF defaults to installing the pi-heif PyPI package, which supports converting
|
||||||
|
HEIF (High Efficiency Image File Format) images to PDF from the command line.
|
||||||
|
If your distribution does not have this library available, you can exclude it and
|
||||||
|
OCRmyPDF will gracefully degrade automatically, losing only support for this
|
||||||
|
feature.
|
||||||
@@ -30,6 +30,90 @@ OCRmyPDF typically supports the three most recent Python versions.
|
|||||||
|
|
||||||
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||||
|
|
||||||
|
v16.4.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed order of filenames passed to Ghostscript for PDF/A generation. :issue:`1359`
|
||||||
|
- Suppressed missing jbig2dec warning message. :issue:`1358`
|
||||||
|
- Fixed calculation of image size when soft mask dimensions don't match image
|
||||||
|
dimension. :issue:`1351`
|
||||||
|
- Several fixes to documentation. Thanks to users Iris and JoKalliauer
|
||||||
|
who contributed these changes.
|
||||||
|
- Fixed error on processing PDFs that are missing certain image metadata. :issue:`1315`
|
||||||
|
|
||||||
|
v16.4.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed calculation of image printed area (used in finding weighted DPI for OCR).
|
||||||
|
:issue:`1334`
|
||||||
|
- Fixed "NotImplementedError: not sure how to get colorspace" error
|
||||||
|
messages in logs which simply records a failure to optimize images with
|
||||||
|
print production colorspaces. :issue:`1315`
|
||||||
|
|
||||||
|
v16.4.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Selecting the ``osd`` and ``equ`` pseudo-languages with ``-l/--language`` now
|
||||||
|
exits with an error when using Tesseract OCR, because these are not
|
||||||
|
regular Tesseract languages but implementation details implemented.
|
||||||
|
Using them can cause Tesseract to crash.
|
||||||
|
- The hOCR renderer is more tolerant of extra whitespace in input files.
|
||||||
|
- watcher.py now changes the output file extension to .pdf when the input is not
|
||||||
|
.pdf.
|
||||||
|
- Improved handling of PDFs that contain circularly referenced Form XObjects.
|
||||||
|
:issue:`1321`
|
||||||
|
- Fixed Alpine Docker image for ARM64, which was not building correctly.
|
||||||
|
- Docker images now use pikepdf 9.0.0.
|
||||||
|
- Prevent use of Tesseract OCR 5.4.0, a version with known regressions.
|
||||||
|
- Disabled progressbar for "Linearizing" when ``--no-progress-bar`` set.
|
||||||
|
- Fixed some tests that warn about missing JBIG2 decoding via pikepdf, by
|
||||||
|
installing the necessary libraries during tests.
|
||||||
|
|
||||||
|
v16.3.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed a test suite failure with Ghostscript 10.03.0+. :issue:`1316`
|
||||||
|
- Fixed an issue with the presentation of the "OCR" progress bar. :issue:`1313`
|
||||||
|
|
||||||
|
v16.3.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed progress bar not displaying for Ghostscript PDF/A conversion. :issue:`1313`
|
||||||
|
- Added progress bar for linearization. :issue:`1313`
|
||||||
|
- If `--rotate-pages-threshold` issued without `--rotate-pages` we now exit with
|
||||||
|
an error since the user likely intended to use `--rotate-pages`. :issue:`1309`
|
||||||
|
- If Tesseract hOCR gives an invalid line box, print an error message instead of
|
||||||
|
exiting with an error. :issue:`1312`
|
||||||
|
|
||||||
|
v16.2.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed issue 'NoneType' object has no attribute 'get' when optimizing certain PDFs.
|
||||||
|
:issue:`1293,1271`
|
||||||
|
- Switched formatting from black to ruff.
|
||||||
|
- Added support for sending sidecar output to io.BytesIO.
|
||||||
|
- Added support for converting HEIF/HEIC images (the native image of iPhones and
|
||||||
|
some other devices) to PDFs, when the appropriate pi-hief library is installed.
|
||||||
|
This library is marked as a dependency, but maintainers may opt out if needed.
|
||||||
|
- We now default to downsampling large images that would exceed Tesseract's internal
|
||||||
|
limits, but only if it cause processing to fail. Previously, this behavior only
|
||||||
|
occurred if specifically requested on command line. It can still be configured
|
||||||
|
and disabled. See the --tesseract command line options.
|
||||||
|
- Added Macports install instructions. Thanks @akierig.
|
||||||
|
- Improved logging output when an unexpected error occurs while trying to obtain
|
||||||
|
the version of a third party program.
|
||||||
|
|
||||||
|
v16.1.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed test suite failure when using Ghostscript 10.3.
|
||||||
|
- Other minor corrections.
|
||||||
|
|
||||||
|
v16.1.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed PyPy 3.10 support.
|
||||||
|
|
||||||
v16.1.0
|
v16.1.0
|
||||||
=======
|
=======
|
||||||
|
|
||||||
|
|||||||
+48
-10
@@ -1,5 +1,6 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# SPDX-FileCopyrightText: 2016 findingorder <https://github.com/findingorder>
|
# SPDX-FileCopyrightText: 2016 findingorder <https://github.com/findingorder>
|
||||||
|
# SPDX-FileCopyrightText: 2024 nilsro <https://github.com/nilsro>
|
||||||
# SPDX-License-Identifier: MIT
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
"""Example of using ocrmypdf as a library in a script.
|
"""Example of using ocrmypdf as a library in a script.
|
||||||
@@ -13,7 +14,11 @@ You should edit this script to meet your needs.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import filecmp
|
||||||
import logging
|
import logging
|
||||||
|
import os
|
||||||
|
import posixpath
|
||||||
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -22,32 +27,65 @@ import ocrmypdf
|
|||||||
# pylint: disable=logging-format-interpolation
|
# pylint: disable=logging-format-interpolation
|
||||||
# pylint: disable=logging-not-lazy
|
# pylint: disable=logging-not-lazy
|
||||||
|
|
||||||
|
|
||||||
|
def filecompare(a, b):
|
||||||
|
try:
|
||||||
|
return filecmp.cmp(a, b, shallow=True)
|
||||||
|
except FileNotFoundError:
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
script_dir = Path(__file__).parent
|
script_dir = Path(__file__).parent
|
||||||
|
# set archive_dir to a path for backup original documents. Leave empty if not required.
|
||||||
|
archive_dir = "/pdfbak"
|
||||||
|
|
||||||
if len(sys.argv) > 1:
|
if len(sys.argv) > 1:
|
||||||
start_dir = Path(sys.argv[1])
|
start_dir = Path(sys.argv[1])
|
||||||
else:
|
else:
|
||||||
start_dir = Path('.')
|
start_dir = Path(".")
|
||||||
|
|
||||||
if len(sys.argv) > 2:
|
if len(sys.argv) > 2:
|
||||||
log_file = Path(sys.argv[2])
|
log_file = Path(sys.argv[2])
|
||||||
else:
|
else:
|
||||||
log_file = script_dir.with_name('ocr-tree.log')
|
log_file = script_dir.with_name("ocr-tree.log")
|
||||||
|
|
||||||
logging.basicConfig(
|
logging.basicConfig(
|
||||||
level=logging.INFO,
|
level=logging.INFO,
|
||||||
format='%(asctime)s %(message)s',
|
format="%(asctime)s %(message)s",
|
||||||
filename=log_file,
|
filename=log_file,
|
||||||
filemode='a',
|
filemode="a",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
logging.info(f"Start directory {start_dir}")
|
||||||
|
|
||||||
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
||||||
|
|
||||||
for filename in start_dir.glob("**/*.py"):
|
for filename in start_dir.glob("**/*.pdf"):
|
||||||
logging.info(f"Processing {filename}")
|
logging.info(f"Processing {filename}")
|
||||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
if ocrmypdf.pdfa.file_claims_pdfa(filename)["pass"]:
|
||||||
if result == ocrmypdf.ExitCode.already_done_ocr:
|
logging.info("Skipped document because it already contained text")
|
||||||
logging.error("Skipped document because it already contained text")
|
else:
|
||||||
elif result == ocrmypdf.ExitCode.ok:
|
archive_filename = archive_dir + str(filename)
|
||||||
|
if len(archive_dir) > 0 and not filecompare(filename, archive_filename):
|
||||||
|
logging.info(f"Archiving document to {archive_filename}")
|
||||||
|
try:
|
||||||
|
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||||
|
except OSError:
|
||||||
|
os.makedirs(posixpath.dirname(archive_filename))
|
||||||
|
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||||
|
try:
|
||||||
|
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||||
|
logging.info(result)
|
||||||
|
except ocrmypdf.exceptions.EncryptedPdfError:
|
||||||
|
logging.info("Skipped document because it is encrypted")
|
||||||
|
except ocrmypdf.exceptions.PriorOcrFoundError:
|
||||||
|
logging.info("Skipped document because it already contained text")
|
||||||
|
except ocrmypdf.exceptions.DigitalSignatureError:
|
||||||
|
logging.info("Skipped document because it has a digital signature")
|
||||||
|
except ocrmypdf.exceptions.TaggedPDFError:
|
||||||
|
logging.info(
|
||||||
|
"Skipped document because it does not need ocr as it is tagged"
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
logging.error("Unhandled error occured")
|
||||||
logging.info("OCR complete")
|
logging.info("OCR complete")
|
||||||
logging.info(result)
|
|
||||||
|
|||||||
+4
-3
@@ -53,9 +53,10 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
|||||||
]
|
]
|
||||||
logging.info(cmd)
|
logging.info(cmd)
|
||||||
full_path_ocr = os.path.join(dir_name, filename_ocr)
|
full_path_ocr = os.path.join(dir_name, filename_ocr)
|
||||||
with open(filename, 'rb') as input_file, open(
|
with (
|
||||||
full_path_ocr, 'wb'
|
open(filename, 'rb') as input_file,
|
||||||
) as output_file:
|
open(full_path_ocr, 'wb') as output_file,
|
||||||
|
):
|
||||||
proc = subprocess.run(
|
proc = subprocess.run(
|
||||||
cmd,
|
cmd,
|
||||||
stdin=input_file,
|
stdin=input_file,
|
||||||
|
|||||||
+9
-7
@@ -7,7 +7,6 @@
|
|||||||
|
|
||||||
# Do not enable annotations!
|
# Do not enable annotations!
|
||||||
# https://github.com/tiangolo/typer/discussions/598
|
# https://github.com/tiangolo/typer/discussions/598
|
||||||
# from __future__ import annotations
|
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
@@ -47,15 +46,18 @@ class LoggingLevelEnum(str, Enum):
|
|||||||
CRITICAL = "CRITICAL"
|
CRITICAL = "CRITICAL"
|
||||||
|
|
||||||
|
|
||||||
def get_output_dir(root: Path, basename: str, output_dir_year_month: bool) -> Path:
|
def get_output_path(root: Path, basename: str, output_dir_year_month: bool) -> Path:
|
||||||
|
assert '/' not in basename, "basename must not contain '/'"
|
||||||
if output_dir_year_month:
|
if output_dir_year_month:
|
||||||
today = datetime.today()
|
today = datetime.today()
|
||||||
output_directory_year_month = root / str(today.year) / f'{today.month:02d}'
|
output_directory_year_month = root / str(today.year) / f'{today.month:02d}'
|
||||||
if not output_directory_year_month.exists():
|
if not output_directory_year_month.exists():
|
||||||
output_directory_year_month.mkdir(parents=True, exist_ok=True)
|
output_directory_year_month.mkdir(parents=True, exist_ok=True)
|
||||||
output_path = Path(output_directory_year_month) / basename
|
output_path = Path(output_directory_year_month) / Path(basename).with_suffix(
|
||||||
|
'.pdf'
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
output_path = root / basename
|
output_path = root / Path(basename).with_suffix('.pdf')
|
||||||
return output_path
|
return output_path
|
||||||
|
|
||||||
|
|
||||||
@@ -99,7 +101,7 @@ def execute_ocrmypdf(
|
|||||||
retries_loading_file: int,
|
retries_loading_file: int,
|
||||||
output_dir_year_month: bool,
|
output_dir_year_month: bool,
|
||||||
):
|
):
|
||||||
output_path = get_output_dir(output_dir, file_path.name, output_dir_year_month)
|
output_path = get_output_path(output_dir, file_path.name, output_dir_year_month)
|
||||||
|
|
||||||
log.info("-" * 20)
|
log.info("-" * 20)
|
||||||
log.info(f'New file: {file_path}. Waiting until fully written...')
|
log.info(f'New file: {file_path}. Waiting until fully written...')
|
||||||
@@ -131,7 +133,7 @@ def execute_ocrmypdf(
|
|||||||
|
|
||||||
|
|
||||||
class HandleObserverEvent(PatternMatchingEventHandler):
|
class HandleObserverEvent(PatternMatchingEventHandler):
|
||||||
def __init__(
|
def __init__( # noqa: D107
|
||||||
self,
|
self,
|
||||||
patterns=None,
|
patterns=None,
|
||||||
ignore_patterns=None,
|
ignore_patterns=None,
|
||||||
@@ -191,7 +193,7 @@ def main(
|
|||||||
bool,
|
bool,
|
||||||
typer.Option(
|
typer.Option(
|
||||||
envvar='OCR_OUTPUT_DIRECTORY_YEAR_MONTH',
|
envvar='OCR_OUTPUT_DIRECTORY_YEAR_MONTH',
|
||||||
help='Create a subdirectory in the output directory for each year and month',
|
help='Create a subdirectory in the output directory for each year/month',
|
||||||
),
|
),
|
||||||
] = False,
|
] = False,
|
||||||
on_success_delete: Annotated[
|
on_success_delete: Annotated[
|
||||||
|
|||||||
+13
-31
@@ -12,12 +12,13 @@ readme = "README.md"
|
|||||||
license = { text = "MPL-2.0" }
|
license = { text = "MPL-2.0" }
|
||||||
requires-python = ">=3.10"
|
requires-python = ">=3.10"
|
||||||
dependencies = [
|
dependencies = [
|
||||||
"Pillow>=10.0.1",
|
|
||||||
"deprecation>=2.1.0",
|
"deprecation>=2.1.0",
|
||||||
"img2pdf>=0.5",
|
"img2pdf>=0.5",
|
||||||
"packaging>=20",
|
"packaging>=20",
|
||||||
"pdfminer.six>=20220319",
|
"pdfminer.six>=20220319",
|
||||||
|
"pi-heif", # Heif image format - maintainers: if this is removed, it will NOT break
|
||||||
"pikepdf>=8.10.1",
|
"pikepdf>=8.10.1",
|
||||||
|
"Pillow>=10.0.1",
|
||||||
"pluggy>=1",
|
"pluggy>=1",
|
||||||
"rich>=13",
|
"rich>=13",
|
||||||
]
|
]
|
||||||
@@ -60,7 +61,7 @@ test = [
|
|||||||
"types-Pillow",
|
"types-Pillow",
|
||||||
"types-humanfriendly",
|
"types-humanfriendly",
|
||||||
]
|
]
|
||||||
watcher = ["watchdog>=1.0.2", "typer[all]", "python-dotenv"]
|
watcher = ["watchdog>=1.0.2", "typer-slim[standard]", "python-dotenv"]
|
||||||
webservice = ["Flask>=2.0.1"]
|
webservice = ["Flask>=2.0.1"]
|
||||||
|
|
||||||
[project.scripts]
|
[project.scripts]
|
||||||
@@ -78,29 +79,6 @@ namespaces = false
|
|||||||
[tool.distutils.bdist_wheel]
|
[tool.distutils.bdist_wheel]
|
||||||
python-tag = "py310"
|
python-tag = "py310"
|
||||||
|
|
||||||
[tool.black]
|
|
||||||
line-length = 88
|
|
||||||
target-version = ["py310", "py311", "py312"]
|
|
||||||
skip-string-normalization = true
|
|
||||||
include = '\.pyi?$'
|
|
||||||
exclude = '''
|
|
||||||
/(
|
|
||||||
\.eggs
|
|
||||||
| \.git
|
|
||||||
| \.hg
|
|
||||||
| \.mypy_cache
|
|
||||||
| \.tox
|
|
||||||
| \.venv
|
|
||||||
| _build
|
|
||||||
| buck-out
|
|
||||||
| build
|
|
||||||
| dist
|
|
||||||
| docs
|
|
||||||
| misc
|
|
||||||
| \.egg-info
|
|
||||||
)/
|
|
||||||
'''
|
|
||||||
|
|
||||||
[tool.coverage.run]
|
[tool.coverage.run]
|
||||||
branch = true
|
branch = true
|
||||||
parallel = true
|
parallel = true
|
||||||
@@ -153,7 +131,10 @@ module = [
|
|||||||
ignore_missing_imports = true
|
ignore_missing_imports = true
|
||||||
|
|
||||||
[tool.ruff]
|
[tool.ruff]
|
||||||
select = [
|
target-version = "py310"
|
||||||
|
|
||||||
|
[tool.ruff.lint]
|
||||||
|
"select" = [
|
||||||
"D", # pydocstyle
|
"D", # pydocstyle
|
||||||
"E", # pycodestyle
|
"E", # pycodestyle
|
||||||
"W", # pycodestyle
|
"W", # pycodestyle
|
||||||
@@ -161,17 +142,18 @@ select = [
|
|||||||
"I001", # isort
|
"I001", # isort
|
||||||
"UP", # pyupgrade
|
"UP", # pyupgrade
|
||||||
]
|
]
|
||||||
target-version = "py310"
|
|
||||||
|
|
||||||
[tool.ruff.isort]
|
[tool.ruff.lint.isort]
|
||||||
known-first-party = ["ocrmypdf"]
|
known-first-party = ["ocrmypdf"]
|
||||||
required-imports = ["from __future__ import annotations"]
|
|
||||||
|
|
||||||
[tool.ruff.pydocstyle]
|
[tool.ruff.lint.pydocstyle]
|
||||||
convention = "google"
|
convention = "google"
|
||||||
|
|
||||||
[tool.ruff.per-file-ignores]
|
[tool.ruff.lint.per-file-ignores]
|
||||||
"docs/conf.py" = ["D100", "D101", "D105"]
|
"docs/conf.py" = ["D100", "D101", "D105"]
|
||||||
"tests/*.py" = ["D100", "D101", "D102", "D103", "D105"]
|
"tests/*.py" = ["D100", "D101", "D102", "D103", "D105"]
|
||||||
"misc/*.py" = ["D103", "D101", "D102"]
|
"misc/*.py" = ["D103", "D101", "D102"]
|
||||||
"src/ocrmypdf/builtin_plugins/*.py" = ["D103", "D102", "D105"]
|
"src/ocrmypdf/builtin_plugins/*.py" = ["D103", "D102", "D105"]
|
||||||
|
|
||||||
|
[tool.ruff.format]
|
||||||
|
quote-style = "preserve"
|
||||||
|
|||||||
+2
-2
@@ -18,8 +18,8 @@ architectures: [amd64]
|
|||||||
|
|
||||||
environment:
|
environment:
|
||||||
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
||||||
GS_LIB: $SNAP/usr/share/ghostscript/9.55/Resource/Init
|
GS_LIB: $SNAP/usr/share/ghostscript/9.55.0/Resource/Init
|
||||||
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.55/Resource/Font
|
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.55.0/Resource/Font
|
||||||
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
||||||
|
|
||||||
apps:
|
apps:
|
||||||
|
|||||||
@@ -7,8 +7,8 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import threading
|
import threading
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
from collections.abc import Iterable
|
from collections.abc import Callable, Iterable
|
||||||
from typing import Any, Callable, TypeVar
|
from typing import Any, TypeVar
|
||||||
|
|
||||||
from ocrmypdf._progressbar import NullProgressBar, ProgressBar
|
from ocrmypdf._progressbar import NullProgressBar, ProgressBar
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,8 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
# Enforce English hegemony
|
||||||
|
DEFAULT_LANGUAGE = 'eng'
|
||||||
|
|
||||||
|
# Default rotation threshold
|
||||||
|
DEFAULT_ROTATE_PAGES_THRESHOLD = 14.0
|
||||||
@@ -177,6 +177,17 @@ class GhostscriptFollower:
|
|||||||
self.progressbar_class = progressbar_class
|
self.progressbar_class = progressbar_class
|
||||||
self.progressbar = None
|
self.progressbar = None
|
||||||
|
|
||||||
|
def __enter__(self):
|
||||||
|
# We can't actually set up the progressbar here, because we don't know
|
||||||
|
# how many pages there are until the first __call__() happens. So we
|
||||||
|
# do it in __call__().
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
|
if self.progressbar:
|
||||||
|
return self.progressbar.__exit__(exc_type, exc_value, traceback)
|
||||||
|
return False
|
||||||
|
|
||||||
def __call__(self, line):
|
def __call__(self, line):
|
||||||
if not self.progressbar_class:
|
if not self.progressbar_class:
|
||||||
return
|
return
|
||||||
@@ -187,7 +198,8 @@ class GhostscriptFollower:
|
|||||||
self.progressbar = self.progressbar_class(
|
self.progressbar = self.progressbar_class(
|
||||||
total=self.count, desc="PDF/A conversion", unit='page'
|
total=self.count, desc="PDF/A conversion", unit='page'
|
||||||
)
|
)
|
||||||
return
|
# Now that we know the count, we can set up the progressbar.
|
||||||
|
self.progressbar.__enter__()
|
||||||
else:
|
else:
|
||||||
if self.re_page.match(line.strip()):
|
if self.re_page.match(line.strip()):
|
||||||
self.progressbar.update()
|
self.progressbar.update()
|
||||||
@@ -265,7 +277,10 @@ def generate_pdfa(
|
|||||||
)
|
)
|
||||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||||
try:
|
try:
|
||||||
with Path(output_file).open('wb') as output:
|
with (
|
||||||
|
Path(output_file).open('wb') as output,
|
||||||
|
GhostscriptFollower(progressbar_class) as pbar,
|
||||||
|
):
|
||||||
p = run_polling_stderr(
|
p = run_polling_stderr(
|
||||||
args_gs,
|
args_gs,
|
||||||
stdout=output,
|
stdout=output,
|
||||||
@@ -274,7 +289,7 @@ def generate_pdfa(
|
|||||||
text=True,
|
text=True,
|
||||||
encoding='utf-8',
|
encoding='utf-8',
|
||||||
errors='replace',
|
errors='replace',
|
||||||
callback=GhostscriptFollower(progressbar_class),
|
callback=pbar,
|
||||||
)
|
)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
# Ghostscript does not change return code when it fails to create
|
# Ghostscript does not change return code when it fails to create
|
||||||
|
|||||||
@@ -220,7 +220,8 @@ def get_deskew(
|
|||||||
|
|
||||||
def tesseract_log_output(stream: bytes) -> None:
|
def tesseract_log_output(stream: bytes) -> None:
|
||||||
tlog = TesseractLoggerAdapter(
|
tlog = TesseractLoggerAdapter(
|
||||||
log, extra=log.extra if hasattr(log, 'extra') else None # type: ignore
|
log,
|
||||||
|
extra=log.extra if hasattr(log, 'extra') else None, # type: ignore
|
||||||
)
|
)
|
||||||
|
|
||||||
if not stream:
|
if not stream:
|
||||||
|
|||||||
@@ -8,19 +8,17 @@ from __future__ import annotations
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import shlex
|
import shlex
|
||||||
import sys
|
|
||||||
from collections.abc import Iterator
|
from collections.abc import Iterator
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT
|
from subprocess import PIPE, STDOUT
|
||||||
from tempfile import TemporaryDirectory
|
from tempfile import TemporaryDirectory
|
||||||
from typing import Union
|
|
||||||
|
|
||||||
from packaging.version import Version
|
from packaging.version import Version
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
from ocrmypdf.exceptions import SubprocessOutputError
|
||||||
from ocrmypdf.subprocess import get_version, run
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
# unpaper documentation:
|
# unpaper documentation:
|
||||||
@@ -29,7 +27,7 @@ from ocrmypdf.subprocess import get_version, run
|
|||||||
|
|
||||||
UNPAPER_IMAGE_PIXEL_LIMIT = 256 * 1024 * 1024
|
UNPAPER_IMAGE_PIXEL_LIMIT = 256 * 1024 * 1024
|
||||||
|
|
||||||
DecFloat = Union[Decimal, float]
|
DecFloat = Decimal | float
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|||||||
@@ -93,8 +93,8 @@ class PageContext:
|
|||||||
state = self.__dict__.copy()
|
state = self.__dict__.copy()
|
||||||
|
|
||||||
state['options'] = copy(self.options)
|
state['options'] = copy(self.options)
|
||||||
if not isinstance(state['options'].input_file, (str, bytes, os.PathLike)):
|
if not isinstance(state['options'].input_file, str | bytes | os.PathLike):
|
||||||
state['options'].input_file = 'stream'
|
state['options'].input_file = 'stream'
|
||||||
if not isinstance(state['options'].output_file, (str, bytes, os.PathLike)):
|
if not isinstance(state['options'].output_file, str | bytes | os.PathLike):
|
||||||
state['options'].output_file = 'stream'
|
state['options'].output_file = 'stream'
|
||||||
return state
|
return state
|
||||||
|
|||||||
@@ -153,22 +153,52 @@ def _set_language(pdf: Pdf, languages: list[str]):
|
|||||||
pdf.Root.Lang = iso639_2
|
pdf.Root.Lang = iso639_2
|
||||||
|
|
||||||
|
|
||||||
|
class MetadataProgress:
|
||||||
|
def __init__(self, progressbar_class, enable: bool = True):
|
||||||
|
self.progressbar_class = progressbar_class
|
||||||
|
self.progressbar = self.progressbar_class(
|
||||||
|
total=100, desc="Linearizing", unit='%', disable=not enable
|
||||||
|
)
|
||||||
|
|
||||||
|
def __enter__(self):
|
||||||
|
self.progressbar.__enter__()
|
||||||
|
return self
|
||||||
|
|
||||||
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
|
return self.progressbar.__exit__(exc_type, exc_value, traceback)
|
||||||
|
|
||||||
|
def __call__(self, percent: int):
|
||||||
|
if not self.progressbar_class:
|
||||||
|
return
|
||||||
|
self.progressbar.update(completed=percent)
|
||||||
|
|
||||||
|
|
||||||
def metadata_fixup(
|
def metadata_fixup(
|
||||||
working_file: Path, context: PdfContext, pdf_save_settings: dict[str, Any]
|
working_file: Path, context: PdfContext, pdf_save_settings: dict[str, Any]
|
||||||
) -> Path:
|
) -> Path:
|
||||||
"""Fix certain metadata fields after Ghostscript PDF/A conversion.
|
"""Fix certain metadata fields whether PDF or PDF/A.
|
||||||
|
|
||||||
|
Override some of Ghostscript's metadata choices.
|
||||||
|
|
||||||
Also report on metadata in the input file that was not retained during
|
Also report on metadata in the input file that was not retained during
|
||||||
PDF/A conversion.
|
conversion.
|
||||||
"""
|
"""
|
||||||
output_file = context.get_path('metafix.pdf')
|
output_file = context.get_path('metafix.pdf')
|
||||||
options = context.options
|
options = context.options
|
||||||
|
|
||||||
with Pdf.open(context.origin) as original, Pdf.open(working_file) as pdf:
|
pbar_class = context.plugin_manager.hook.get_progressbar_class()
|
||||||
|
with (
|
||||||
|
Pdf.open(context.origin) as original,
|
||||||
|
Pdf.open(working_file) as pdf,
|
||||||
|
MetadataProgress(pbar_class, options.progress_bar) as pbar,
|
||||||
|
):
|
||||||
docinfo = get_docinfo(original, context)
|
docinfo = get_docinfo(original, context)
|
||||||
with original.open_metadata(
|
with (
|
||||||
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
original.open_metadata(
|
||||||
) as meta_original, pdf.open_metadata() as meta_pdf:
|
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
||||||
|
) as meta_original,
|
||||||
|
pdf.open_metadata() as meta_pdf,
|
||||||
|
):
|
||||||
meta_pdf.load_from_docinfo(
|
meta_pdf.load_from_docinfo(
|
||||||
docinfo, delete_missing=False, raise_failure=False
|
docinfo, delete_missing=False, raise_failure=False
|
||||||
)
|
)
|
||||||
@@ -179,6 +209,6 @@ def metadata_fixup(
|
|||||||
report_on_metadata(options, meta_missing)
|
report_on_metadata(options, meta_missing)
|
||||||
|
|
||||||
_set_language(pdf, options.languages)
|
_set_language(pdf, options.languages)
|
||||||
pdf.save(output_file, **pdf_save_settings)
|
pdf.save(output_file, progress=pbar, **pdf_save_settings)
|
||||||
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ from collections.abc import Iterable, Iterator, Sequence
|
|||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj, copystat
|
from shutil import copyfileobj
|
||||||
from typing import Any, BinaryIO, TypeVar, cast
|
from typing import Any, BinaryIO, TypeVar, cast
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
@@ -41,12 +41,23 @@ from ocrmypdf.pdfa import generate_pdfa_ps
|
|||||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, PageInfo, PdfInfo
|
from ocrmypdf.pdfinfo import Colorspace, Encoding, PageInfo, PdfInfo
|
||||||
from ocrmypdf.pluginspec import OrientationConfidence
|
from ocrmypdf.pluginspec import OrientationConfidence
|
||||||
|
|
||||||
|
try:
|
||||||
|
from pi_heif import register_heif_opener
|
||||||
|
except ImportError:
|
||||||
|
|
||||||
|
def register_heif_opener():
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
T = TypeVar("T")
|
T = TypeVar("T")
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
VECTOR_PAGE_DPI = 400
|
VECTOR_PAGE_DPI = 400
|
||||||
|
|
||||||
|
|
||||||
|
register_heif_opener()
|
||||||
|
|
||||||
|
|
||||||
def triage_image_file(input_file: Path, output_file: Path, options) -> None:
|
def triage_image_file(input_file: Path, output_file: Path, options) -> None:
|
||||||
"""Triage the input image file.
|
"""Triage the input image file.
|
||||||
|
|
||||||
@@ -131,7 +142,7 @@ def _pdf_guess_version(input_file: Path, search_window=1024) -> str:
|
|||||||
"""
|
"""
|
||||||
with open(input_file, 'rb') as f:
|
with open(input_file, 'rb') as f:
|
||||||
signature = f.read(search_window)
|
signature = f.read(search_window)
|
||||||
m = re.search(br'%PDF-(\d\.\d)', signature)
|
m = re.search(rb'%PDF-(\d\.\d)', signature)
|
||||||
if m:
|
if m:
|
||||||
return m.group(1).decode('ascii')
|
return m.group(1).decode('ascii')
|
||||||
return ''
|
return ''
|
||||||
@@ -464,7 +475,7 @@ def calculate_raster_dpi(page_context: PageContext):
|
|||||||
page_dpi = get_page_square_dpi(page_context, image_dpi)
|
page_dpi = get_page_square_dpi(page_context, image_dpi)
|
||||||
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
|
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
|
||||||
log.warning(
|
log.warning(
|
||||||
"Weight average image DPI is %0.1f, max DPI is %0.1f. "
|
"Weighted average image DPI is %0.1f, max DPI is %0.1f. "
|
||||||
"The discrepancy may indicate a high detail region on this page, "
|
"The discrepancy may indicate a high detail region on this page, "
|
||||||
"but could also indicate a problem with the input PDF file. "
|
"but could also indicate a problem with the input PDF file. "
|
||||||
"Page image will be rendered at %0.1f DPI.",
|
"Page image will be rendered at %0.1f DPI.",
|
||||||
@@ -767,7 +778,9 @@ def render_hocr_page(hocr: Path, page_context: PageContext) -> Path:
|
|||||||
font=Courier(),
|
font=Courier(),
|
||||||
)
|
)
|
||||||
HocrTransform(
|
HocrTransform(
|
||||||
hocr_filename=hocr, dpi=dpi.to_scalar(), **debug_kwargs # square
|
hocr_filename=hocr,
|
||||||
|
dpi=dpi.to_scalar(),
|
||||||
|
**debug_kwargs, # square
|
||||||
).to_pdf(
|
).to_pdf(
|
||||||
out_filename=output_file,
|
out_filename=output_file,
|
||||||
image_filename=None,
|
image_filename=None,
|
||||||
|
|||||||
@@ -11,13 +11,13 @@ import os
|
|||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
from collections.abc import Sequence
|
from collections.abc import Callable, Sequence
|
||||||
from concurrent.futures.process import BrokenProcessPool
|
from concurrent.futures.process import BrokenProcessPool
|
||||||
from concurrent.futures.thread import BrokenThreadPool
|
from concurrent.futures.thread import BrokenThreadPool
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Callable, NamedTuple, cast
|
from typing import NamedTuple, cast
|
||||||
|
|
||||||
import PIL
|
import PIL
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
|
|
||||||
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
|
|
||||||
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
@@ -104,14 +103,14 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
|||||||
try:
|
try:
|
||||||
set_thread_pageno(result.pageno + 1)
|
set_thread_pageno(result.pageno + 1)
|
||||||
sidecars[result.pageno] = result.text
|
sidecars[result.pageno] = result.text
|
||||||
pbar.update()
|
pbar.update(0.5)
|
||||||
ocrgraft.graft_page(
|
ocrgraft.graft_page(
|
||||||
pageno=result.pageno,
|
pageno=result.pageno,
|
||||||
image=result.pdf_page_from_image,
|
image=result.pdf_page_from_image,
|
||||||
textpdf=result.ocr,
|
textpdf=result.ocr,
|
||||||
autorotate_correction=result.orientation_correction,
|
autorotate_correction=result.orientation_correction,
|
||||||
)
|
)
|
||||||
pbar.update()
|
pbar.update(0.5)
|
||||||
finally:
|
finally:
|
||||||
set_thread_pageno(None)
|
set_thread_pageno(None)
|
||||||
|
|
||||||
@@ -119,10 +118,9 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
|||||||
use_threads=options.use_threads,
|
use_threads=options.use_threads,
|
||||||
max_workers=max_workers,
|
max_workers=max_workers,
|
||||||
progress_kwargs=dict(
|
progress_kwargs=dict(
|
||||||
total=(2 * len(context.pdfinfo)),
|
total=len(context.pdfinfo),
|
||||||
desc='OCR' if options.tesseract_timeout > 0 else 'Image processing',
|
desc='OCR' if options.tesseract_timeout > 0 else 'Image processing',
|
||||||
unit='page',
|
unit='page',
|
||||||
unit_scale=0.5,
|
|
||||||
disable=not options.progress_bar,
|
disable=not options.progress_bar,
|
||||||
),
|
),
|
||||||
worker_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS),
|
worker_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS),
|
||||||
@@ -155,12 +153,13 @@ def _run_pipeline(
|
|||||||
options: argparse.Namespace,
|
options: argparse.Namespace,
|
||||||
plugin_manager: OcrmypdfPluginManager,
|
plugin_manager: OcrmypdfPluginManager,
|
||||||
) -> ExitCode:
|
) -> ExitCode:
|
||||||
with manage_work_folder(
|
with (
|
||||||
work_folder=Path(mkdtemp(prefix="ocrmypdf.io.")),
|
manage_work_folder(
|
||||||
retain=options.keep_temporary_files,
|
work_folder=Path(mkdtemp(prefix="ocrmypdf.io.")),
|
||||||
print_location=options.keep_temporary_files,
|
retain=options.keep_temporary_files,
|
||||||
) as work_folder, manage_debug_log_handler(
|
print_location=options.keep_temporary_files,
|
||||||
options=options, work_folder=work_folder
|
) as work_folder,
|
||||||
|
manage_debug_log_handler(options=options, work_folder=work_folder),
|
||||||
):
|
):
|
||||||
executor = setup_pipeline(options, plugin_manager)
|
executor = setup_pipeline(options, plugin_manager)
|
||||||
check_requested_output_file(options)
|
check_requested_output_file(options)
|
||||||
|
|||||||
@@ -4,7 +4,6 @@
|
|||||||
|
|
||||||
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
|
|||||||
@@ -66,7 +66,7 @@ class ProgressBar(Protocol):
|
|||||||
def __exit__(self, *args):
|
def __exit__(self, *args):
|
||||||
"""Exit a progress bar context."""
|
"""Exit a progress bar context."""
|
||||||
|
|
||||||
def update(self, n=1):
|
def update(self, n=1, *, completed=None):
|
||||||
"""Update the progress bar by an increment.
|
"""Update the progress bar by an increment.
|
||||||
|
|
||||||
For use within a progress bar context.
|
For use within a progress bar context.
|
||||||
@@ -85,7 +85,7 @@ class NullProgressBar:
|
|||||||
def __exit__(self, exc_type, exc_value, traceback):
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def update(self, _arg=None):
|
def update(self, _arg=None, *, completed=None):
|
||||||
return
|
return
|
||||||
|
|
||||||
|
|
||||||
@@ -103,6 +103,7 @@ class RichProgressBar:
|
|||||||
disable: bool = False,
|
disable: bool = False,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
|
self._entered = False
|
||||||
self.progress = Progress(
|
self.progress = Progress(
|
||||||
TextColumn(
|
TextColumn(
|
||||||
"[progress.description]{task.description}",
|
"[progress.description]{task.description}",
|
||||||
@@ -130,6 +131,7 @@ class RichProgressBar:
|
|||||||
|
|
||||||
def __enter__(self):
|
def __enter__(self):
|
||||||
self.progress.start()
|
self.progress.start()
|
||||||
|
self._entered = True
|
||||||
return self
|
return self
|
||||||
|
|
||||||
def __exit__(self, exc_type, exc_value, traceback):
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
@@ -137,6 +139,10 @@ class RichProgressBar:
|
|||||||
self.progress.stop()
|
self.progress.stop()
|
||||||
return False
|
return False
|
||||||
|
|
||||||
def update(self, value=None):
|
def update(self, n=1, *, completed=None):
|
||||||
advance = self.unit_scale if value is None else value
|
assert self._entered, "Progress bar must be entered before updating"
|
||||||
self.progress.update(self.progress_bar, advance=advance)
|
if completed is None:
|
||||||
|
advance = self.unit_scale if n is None else n
|
||||||
|
self.progress.update(self.progress_bar, advance=advance)
|
||||||
|
else:
|
||||||
|
self.progress.update(self.progress_bar, completed=completed)
|
||||||
|
|||||||
@@ -20,6 +20,7 @@ import pikepdf
|
|||||||
import PIL
|
import PIL
|
||||||
from pluggy import PluginManager
|
from pluggy import PluginManager
|
||||||
|
|
||||||
|
from ocrmypdf._defaults import DEFAULT_LANGUAGE, DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||||
from ocrmypdf._exec import unpaper
|
from ocrmypdf._exec import unpaper
|
||||||
from ocrmypdf.exceptions import (
|
from ocrmypdf.exceptions import (
|
||||||
BadArgsError,
|
BadArgsError,
|
||||||
@@ -30,17 +31,9 @@ from ocrmypdf.exceptions import (
|
|||||||
from ocrmypdf.helpers import is_file_writable, monotonic, safe_symlink
|
from ocrmypdf.helpers import is_file_writable, monotonic, safe_symlink
|
||||||
from ocrmypdf.subprocess import check_external_program
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
# -------------
|
|
||||||
# External dependencies
|
|
||||||
|
|
||||||
DEFAULT_LANGUAGE = 'eng' # Enforce English hegemony
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
# --------
|
|
||||||
|
|
||||||
|
|
||||||
def check_platform() -> None:
|
def check_platform() -> None:
|
||||||
if sys.maxsize <= 2**32: # pragma: no cover
|
if sys.maxsize <= 2**32: # pragma: no cover
|
||||||
log.warning(
|
log.warning(
|
||||||
@@ -61,6 +54,7 @@ def check_options_languages(
|
|||||||
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
||||||
if not ocr_engine_languages:
|
if not ocr_engine_languages:
|
||||||
return
|
return
|
||||||
|
|
||||||
missing_languages = set(options.languages) - set(ocr_engine_languages)
|
missing_languages = set(options.languages) - set(ocr_engine_languages)
|
||||||
if missing_languages:
|
if missing_languages:
|
||||||
lang_text = '\n'.join(lang for lang in missing_languages)
|
lang_text = '\n'.join(lang for lang in missing_languages)
|
||||||
@@ -130,6 +124,11 @@ def check_options_preprocessing(options: Namespace) -> None:
|
|||||||
options.clean = True
|
options.clean = True
|
||||||
if options.unpaper_args and not options.clean:
|
if options.unpaper_args and not options.clean:
|
||||||
raise BadArgsError("--clean is required for --unpaper-args")
|
raise BadArgsError("--clean is required for --unpaper-args")
|
||||||
|
if (
|
||||||
|
options.rotate_pages_threshold != DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||||
|
and not options.rotate_pages
|
||||||
|
):
|
||||||
|
raise BadArgsError("--rotate-pages is required for --rotate-pages-threshold")
|
||||||
if options.clean:
|
if options.clean:
|
||||||
check_external_program(
|
check_external_program(
|
||||||
program='unpaper',
|
program='unpaper',
|
||||||
|
|||||||
+13
-8
@@ -14,7 +14,7 @@ from collections.abc import Iterable, Sequence
|
|||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
from io import IOBase
|
from io import IOBase
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import AnyStr, BinaryIO, Union
|
from typing import AnyStr, BinaryIO
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
import pluggy
|
import pluggy
|
||||||
@@ -28,8 +28,8 @@ from ocrmypdf._validation import check_options
|
|||||||
from ocrmypdf.cli import ArgumentParser, get_parser
|
from ocrmypdf.cli import ArgumentParser, get_parser
|
||||||
from ocrmypdf.helpers import is_iterable_notstr
|
from ocrmypdf.helpers import is_iterable_notstr
|
||||||
|
|
||||||
StrPath = Union[Path, AnyStr]
|
StrPath = Path | AnyStr
|
||||||
PathOrIO = Union[BinaryIO, StrPath]
|
PathOrIO = BinaryIO | StrPath
|
||||||
|
|
||||||
# Installing plugins affects the global state of the Python interpreter,
|
# Installing plugins affects the global state of the Python interpreter,
|
||||||
# so we need to use a lock to prevent multiple threads from installing
|
# so we need to use a lock to prevent multiple threads from installing
|
||||||
@@ -169,7 +169,7 @@ def _kwargs_to_cmdline(
|
|||||||
|
|
||||||
# We have a parameter
|
# We have a parameter
|
||||||
cmdline.append(f"--{cmd_style_arg}")
|
cmdline.append(f"--{cmd_style_arg}")
|
||||||
if isinstance(val, (int, float)):
|
if isinstance(val, int | float):
|
||||||
cmdline.append(str(val))
|
cmdline.append(str(val))
|
||||||
elif isinstance(val, str):
|
elif isinstance(val, str):
|
||||||
cmdline.append(val)
|
cmdline.append(val)
|
||||||
@@ -201,14 +201,17 @@ def create_options(
|
|||||||
defer_kwargs={'progress_bar', 'plugins', 'parser', 'input_file', 'output_file'},
|
defer_kwargs={'progress_bar', 'plugins', 'parser', 'input_file', 'output_file'},
|
||||||
**kwargs,
|
**kwargs,
|
||||||
)
|
)
|
||||||
if isinstance(input_file, (BinaryIO, IOBase)):
|
if isinstance(input_file, BinaryIO | IOBase):
|
||||||
cmdline.append('stream://input_file')
|
cmdline.append('stream://input_file')
|
||||||
else:
|
else:
|
||||||
cmdline.append(os.fspath(input_file))
|
cmdline.append(os.fspath(input_file))
|
||||||
if isinstance(output_file, (BinaryIO, IOBase)):
|
if isinstance(output_file, BinaryIO | IOBase):
|
||||||
cmdline.append('stream://output_file')
|
cmdline.append('stream://output_file')
|
||||||
else:
|
else:
|
||||||
cmdline.append(os.fspath(output_file))
|
cmdline.append(os.fspath(output_file))
|
||||||
|
if 'sidecar' in kwargs and isinstance(kwargs['sidecar'], BinaryIO | IOBase):
|
||||||
|
cmdline.append('--sidecar')
|
||||||
|
cmdline.append('stream://sidecar')
|
||||||
|
|
||||||
parser.enable_api_mode()
|
parser.enable_api_mode()
|
||||||
options = parser.parse_args(cmdline)
|
options = parser.parse_args(cmdline)
|
||||||
@@ -219,6 +222,8 @@ def create_options(
|
|||||||
options.input_file = input_file
|
options.input_file = input_file
|
||||||
if options.output_file == 'stream://output_file':
|
if options.output_file == 'stream://output_file':
|
||||||
options.output_file = output_file
|
options.output_file = output_file
|
||||||
|
if options.sidecar == 'stream://sidecar':
|
||||||
|
options.sidecar = kwargs['sidecar']
|
||||||
|
|
||||||
return options
|
return options
|
||||||
|
|
||||||
@@ -230,7 +235,7 @@ def ocr( # noqa: D417
|
|||||||
language: Iterable[str] | None = None,
|
language: Iterable[str] | None = None,
|
||||||
image_dpi: int | None = None,
|
image_dpi: int | None = None,
|
||||||
output_type: str | None = None,
|
output_type: str | None = None,
|
||||||
sidecar: StrPath | None = None,
|
sidecar: PathOrIO | None = None,
|
||||||
jobs: int | None = None,
|
jobs: int | None = None,
|
||||||
use_threads: bool | None = None,
|
use_threads: bool | None = None,
|
||||||
title: str | None = None,
|
title: str | None = None,
|
||||||
@@ -343,7 +348,7 @@ def ocr( # noqa: D417
|
|||||||
|
|
||||||
if not plugins:
|
if not plugins:
|
||||||
plugins = []
|
plugins = []
|
||||||
elif isinstance(plugins, (str, Path)):
|
elif isinstance(plugins, str | Path):
|
||||||
plugins = [plugins]
|
plugins = [plugins]
|
||||||
else:
|
else:
|
||||||
plugins = list(plugins)
|
plugins = list(plugins)
|
||||||
|
|||||||
@@ -12,10 +12,10 @@ import queue
|
|||||||
import signal
|
import signal
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
from collections.abc import Iterable
|
from collections.abc import Callable, Iterable
|
||||||
from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed
|
from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from typing import Callable, Union
|
from typing import Union
|
||||||
|
|
||||||
from rich.console import Console as RichConsole
|
from rich.console import Console as RichConsole
|
||||||
|
|
||||||
@@ -25,8 +25,10 @@ from ocrmypdf._progressbar import RichProgressBar
|
|||||||
from ocrmypdf.exceptions import InputFileError
|
from ocrmypdf.exceptions import InputFileError
|
||||||
from ocrmypdf.helpers import remove_all_log_handlers
|
from ocrmypdf.helpers import remove_all_log_handlers
|
||||||
|
|
||||||
FuturesExecutorClass = Union[type[ThreadPoolExecutor], type[ProcessPoolExecutor]]
|
FuturesExecutorClass = Union[ # noqa: UP007
|
||||||
Queue = Union[multiprocessing.Queue, queue.Queue]
|
type[ThreadPoolExecutor], type[ProcessPoolExecutor]
|
||||||
|
]
|
||||||
|
Queue = Union[multiprocessing.Queue, queue.Queue] # noqa: UP007
|
||||||
UserInit = Callable[[], None]
|
UserInit = Callable[[], None]
|
||||||
WorkerInit = Callable[[Queue, UserInit, int], None]
|
WorkerInit = Callable[[Queue, UserInit, int], None]
|
||||||
|
|
||||||
@@ -128,11 +130,14 @@ class StandardExecutor(Executor):
|
|||||||
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
||||||
listener.start()
|
listener.start()
|
||||||
|
|
||||||
with self.pbar_class(**progress_kwargs) as pbar, executor_class(
|
with (
|
||||||
max_workers=max_workers,
|
self.pbar_class(**progress_kwargs) as pbar,
|
||||||
initializer=initializer,
|
executor_class(
|
||||||
initargs=(log_queue, worker_initializer, logging.getLogger("").level),
|
max_workers=max_workers,
|
||||||
) as executor:
|
initializer=initializer,
|
||||||
|
initargs=(log_queue, worker_initializer, logging.getLogger("").level),
|
||||||
|
) as executor,
|
||||||
|
):
|
||||||
futures = [executor.submit(task, *args) for args in task_arguments]
|
futures = [executor.submit(task, *args) for args in task_arguments]
|
||||||
try:
|
try:
|
||||||
for future in as_completed(futures):
|
for future in as_completed(futures):
|
||||||
|
|||||||
@@ -8,7 +8,5 @@ from ocrmypdf import hookimpl
|
|||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def filter_pdf_page(
|
def filter_pdf_page(page, image_filename, output_pdf): # pylint: disable=unused-argument
|
||||||
page, image_filename, output_pdf
|
|
||||||
): # pylint: disable=unused-argument
|
|
||||||
return output_pdf
|
return output_pdf
|
||||||
|
|||||||
@@ -129,7 +129,7 @@ def generate_pdfa(
|
|||||||
):
|
):
|
||||||
"""Generate a PDF/A from the list of PDF pages and PDF/A metadata."""
|
"""Generate a PDF/A from the list of PDF pages and PDF/A metadata."""
|
||||||
ghostscript.generate_pdfa(
|
ghostscript.generate_pdfa(
|
||||||
pdf_pages=[*pdf_pages, pdfmark],
|
pdf_pages=[pdfmark, *pdf_pages],
|
||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
compression=context.options.pdfa_image_compression,
|
compression=context.options.pdfa_image_compression,
|
||||||
color_conversion_strategy=context.options.color_conversion_strategy,
|
color_conversion_strategy=context.options.color_conversion_strategy,
|
||||||
|
|||||||
@@ -2,9 +2,9 @@
|
|||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
"""Built-in plugin to implement OCR using Tesseract."""
|
"""Built-in plugin to implement OCR using Tesseract."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
|
||||||
@@ -14,6 +14,7 @@ from ocrmypdf import hookimpl
|
|||||||
from ocrmypdf._exec import tesseract
|
from ocrmypdf._exec import tesseract
|
||||||
from ocrmypdf._jobcontext import PageContext
|
from ocrmypdf._jobcontext import PageContext
|
||||||
from ocrmypdf.cli import numeric, str_to_int
|
from ocrmypdf.cli import numeric, str_to_int
|
||||||
|
from ocrmypdf.exceptions import BadArgsError, MissingDependencyError
|
||||||
from ocrmypdf.helpers import clamp
|
from ocrmypdf.helpers import clamp
|
||||||
from ocrmypdf.imageops import calculate_downsample, downsample_image
|
from ocrmypdf.imageops import calculate_downsample, downsample_image
|
||||||
from ocrmypdf.pluginspec import OcrEngine
|
from ocrmypdf.pluginspec import OcrEngine
|
||||||
@@ -94,7 +95,8 @@ def add_options(parser):
|
|||||||
)
|
)
|
||||||
tess.add_argument(
|
tess.add_argument(
|
||||||
'--tesseract-downsample-large-images',
|
'--tesseract-downsample-large-images',
|
||||||
action='store_true',
|
action=argparse.BooleanOptionalAction,
|
||||||
|
default=True,
|
||||||
help=(
|
help=(
|
||||||
"Downsample large images before OCR. Tesseract has an upper limit on the "
|
"Downsample large images before OCR. Tesseract has an upper limit on the "
|
||||||
"size images it will support. If this argument is given, OCRmyPDF will "
|
"size images it will support. If this argument is given, OCRmyPDF will "
|
||||||
@@ -143,6 +145,12 @@ def check_options(options):
|
|||||||
need_version='4.1.1', # Ubuntu 22.04 version (also 20.04)
|
need_version='4.1.1', # Ubuntu 22.04 version (also 20.04)
|
||||||
version_parser=tesseract.TesseractVersion,
|
version_parser=tesseract.TesseractVersion,
|
||||||
)
|
)
|
||||||
|
tess_version = tesseract.version()
|
||||||
|
if tess_version == tesseract.TesseractVersion('5.4.0'):
|
||||||
|
raise MissingDependencyError(
|
||||||
|
"Tesseract 5.4.0 is not supported due to regressions in this version. "
|
||||||
|
"Please upgrade to a newer or supported older version."
|
||||||
|
)
|
||||||
|
|
||||||
# Decide on what renderer to use
|
# Decide on what renderer to use
|
||||||
if options.pdf_renderer == 'auto':
|
if options.pdf_renderer == 'auto':
|
||||||
@@ -163,6 +171,14 @@ def check_options(options):
|
|||||||
"The --tesseract-pagesegmode argument you select will disable OCR. "
|
"The --tesseract-pagesegmode argument you select will disable OCR. "
|
||||||
"This may cause processing to fail."
|
"This may cause processing to fail."
|
||||||
)
|
)
|
||||||
|
DENIED_LANGUAGES = {'equ', 'osd'}
|
||||||
|
if DENIED_LANGUAGES & set(options.languages):
|
||||||
|
raise BadArgsError(
|
||||||
|
"The following languages for Tesseract's internal use and should not "
|
||||||
|
"be issued explicitly: "
|
||||||
|
f"{', '.join(DENIED_LANGUAGES & set(options.languages))}\n"
|
||||||
|
"Remove them from the -l/--language argument."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
@@ -220,7 +236,7 @@ class TesseractOcrEngine(OcrEngine):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def creator_tag(options):
|
def creator_tag(options):
|
||||||
tag = '-PDF' if options.pdf_renderer == 'sandwich' else 'hOCR'
|
tag = '-PDF' if options.pdf_renderer == 'sandwich' else '-hOCR'
|
||||||
return f"Tesseract OCR{tag} {TesseractOcrEngine.version()}"
|
return f"Tesseract OCR{tag} {TesseractOcrEngine.version()}"
|
||||||
|
|
||||||
def __str__(self):
|
def __str__(self):
|
||||||
|
|||||||
+5
-4
@@ -6,9 +6,10 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
from collections.abc import Mapping
|
from collections.abc import Callable, Mapping
|
||||||
from typing import Any, Callable, TypeVar
|
from typing import Any, TypeVar
|
||||||
|
|
||||||
|
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||||
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
||||||
from ocrmypdf._version import __version__ as _VERSION
|
from ocrmypdf._version import __version__ as _VERSION
|
||||||
|
|
||||||
@@ -390,7 +391,7 @@ Online documentation is located at:
|
|||||||
action='store',
|
action='store',
|
||||||
type=numeric(float, 0),
|
type=numeric(float, 0),
|
||||||
metavar='MPixels',
|
metavar='MPixels',
|
||||||
help="Set maximum number of pixels to unpack before treating an image as a "
|
help="Set maximum number of megapixels to unpack before treating an image as a "
|
||||||
"decompression bomb",
|
"decompression bomb",
|
||||||
default=250.0,
|
default=250.0,
|
||||||
)
|
)
|
||||||
@@ -403,7 +404,7 @@ Online documentation is located at:
|
|||||||
)
|
)
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--rotate-pages-threshold',
|
'--rotate-pages-threshold',
|
||||||
default=14.0,
|
default=DEFAULT_ROTATE_PAGES_THRESHOLD,
|
||||||
type=numeric(float, 0, 1000),
|
type=numeric(float, 0, 1000),
|
||||||
metavar='CONFIDENCE',
|
metavar='CONFIDENCE',
|
||||||
help="Only rotate pages when confidence is above this value (arbitrary "
|
help="Only rotate pages when confidence is above this value (arbitrary "
|
||||||
|
|||||||
@@ -20,13 +20,12 @@ from __future__ import annotations
|
|||||||
import logging
|
import logging
|
||||||
import logging.handlers
|
import logging.handlers
|
||||||
import signal
|
import signal
|
||||||
from collections.abc import Iterable, Iterator
|
from collections.abc import Callable, Iterable, Iterator
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from enum import Enum, auto
|
from enum import Enum, auto
|
||||||
from itertools import islice, repeat, takewhile, zip_longest
|
from itertools import islice, repeat, takewhile, zip_longest
|
||||||
from multiprocessing import Pipe, Process
|
from multiprocessing import Pipe, Process
|
||||||
from multiprocessing.connection import Connection, wait
|
from multiprocessing.connection import Connection, wait
|
||||||
from typing import Callable
|
|
||||||
|
|
||||||
from ocrmypdf import Executor, hookimpl
|
from ocrmypdf import Executor, hookimpl
|
||||||
from ocrmypdf._concurrent import NullProgressBar
|
from ocrmypdf._concurrent import NullProgressBar
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ import multiprocessing
|
|||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import warnings
|
import warnings
|
||||||
from collections.abc import Iterable, Sequence
|
from collections.abc import Callable, Iterable, Sequence
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from io import StringIO
|
from io import StringIO
|
||||||
@@ -19,7 +19,6 @@ from pathlib import Path
|
|||||||
from statistics import harmonic_mean
|
from statistics import harmonic_mean
|
||||||
from typing import (
|
from typing import (
|
||||||
Any,
|
Any,
|
||||||
Callable,
|
|
||||||
Generic,
|
Generic,
|
||||||
TypeVar,
|
TypeVar,
|
||||||
)
|
)
|
||||||
@@ -269,7 +268,9 @@ def check_pdf(input_file: Path) -> bool:
|
|||||||
return False
|
return False
|
||||||
else:
|
else:
|
||||||
with pdf:
|
with pdf:
|
||||||
messages = pdf.check()
|
with warnings.catch_warnings():
|
||||||
|
warnings.filterwarnings('ignore', message=r'pikepdf.*JBIG2.*')
|
||||||
|
messages = pdf.check()
|
||||||
success = True
|
success = True
|
||||||
for msg in messages:
|
for msg in messages:
|
||||||
if 'error' in msg.lower():
|
if 'error' in msg.lower():
|
||||||
|
|||||||
@@ -61,15 +61,25 @@ class HocrTransform:
|
|||||||
"""A class for converting documents from the hOCR format.
|
"""A class for converting documents from the hOCR format.
|
||||||
|
|
||||||
For details of the hOCR format, see:
|
For details of the hOCR format, see:
|
||||||
http://kba.cloud/hocr-spec/.
|
http://kba.github.io/hocr-spec/1.2/.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
box_pattern = re.compile(r'bbox (\d+) (\d+) (\d+) (\d+)')
|
box_pattern = re.compile(
|
||||||
|
r'''
|
||||||
|
bbox \s+
|
||||||
|
(\d+) \s+ # left: uint
|
||||||
|
(\d+) \s+ # top: uint
|
||||||
|
(\d+) \s+ # right: uint
|
||||||
|
(\d+) # bottom: uint
|
||||||
|
''',
|
||||||
|
re.VERBOSE,
|
||||||
|
)
|
||||||
baseline_pattern = re.compile(
|
baseline_pattern = re.compile(
|
||||||
r'''
|
r'''
|
||||||
baseline \s+
|
baseline \s+
|
||||||
([\-\+]?\d*\.?\d*) \s+ # +/- decimal float
|
([\-\+]?\d*\.?\d*) \s+ # +/- decimal float
|
||||||
([\-\+]?\d+) # +/- int''',
|
([\-\+]?\d+) # +/- int
|
||||||
|
''',
|
||||||
re.VERBOSE,
|
re.VERBOSE,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -84,7 +94,6 @@ class HocrTransform:
|
|||||||
debug_render_options: DebugRenderOptions | None = None,
|
debug_render_options: DebugRenderOptions | None = None,
|
||||||
):
|
):
|
||||||
"""Initialize the HocrTransform object."""
|
"""Initialize the HocrTransform object."""
|
||||||
|
|
||||||
if debug:
|
if debug:
|
||||||
log.warning("Use debug_render_options instead", DeprecationWarning)
|
log.warning("Use debug_render_options instead", DeprecationWarning)
|
||||||
self.render_options = DebugRenderOptions(
|
self.render_options = DebugRenderOptions(
|
||||||
@@ -118,15 +127,12 @@ class HocrTransform:
|
|||||||
# Stop after first div that has page coordinates
|
# Stop after first div that has page coordinates
|
||||||
break
|
break
|
||||||
|
|
||||||
def _get_element_text(self, element: Element):
|
def _get_element_text(self, element: Element) -> str:
|
||||||
"""Return the textual content of the element and its children."""
|
"""Return the textual content of the element and its children."""
|
||||||
text = ''
|
text = element.text if element.text is not None else ''
|
||||||
if element.text is not None:
|
|
||||||
text += element.text
|
|
||||||
for child in element:
|
for child in element:
|
||||||
text += self._get_element_text(child)
|
text += self._get_element_text(child)
|
||||||
if element.tail is not None:
|
text += element.tail if element.tail is not None else ''
|
||||||
text += element.tail
|
|
||||||
return text
|
return text
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -287,7 +293,13 @@ class HocrTransform:
|
|||||||
line_box = self.element_coordinates(line)
|
line_box = self.element_coordinates(line)
|
||||||
if not line_box:
|
if not line_box:
|
||||||
return
|
return
|
||||||
assert line_box.ury > line_box.lly # lly is top, ury is bottom
|
if line_box.ury <= line_box.lly:
|
||||||
|
log.error(
|
||||||
|
"line box is invalid so we cannot render it: box=%s text=%s",
|
||||||
|
line_box,
|
||||||
|
self._get_element_text(line),
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
self._debug_draw_line_bbox(canvas, line_box)
|
self._debug_draw_line_bbox(canvas, line_box)
|
||||||
|
|
||||||
|
|||||||
@@ -7,12 +7,12 @@ Derived from
|
|||||||
https://www.loc.gov/standards/iso639-2/ascii_8bits.html
|
https://www.loc.gov/standards/iso639-2/ascii_8bits.html
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
from typing import NamedTuple
|
from typing import NamedTuple
|
||||||
|
|
||||||
|
|
||||||
class ISOCodeData(NamedTuple):
|
class ISOCodeData(NamedTuple):
|
||||||
"""Data for a single ISO 639 code."""
|
"""Data for a single ISO 639 code."""
|
||||||
|
|
||||||
alt: str
|
alt: str
|
||||||
alpha_2: str
|
alpha_2: str
|
||||||
english: str
|
english: str
|
||||||
@@ -168,8 +168,10 @@ ISO_639_3 = {
|
|||||||
'chu': ISOCodeData(
|
'chu': ISOCodeData(
|
||||||
'',
|
'',
|
||||||
'cu',
|
'cu',
|
||||||
('Church Slavic; Old Slavonic; Church Slavonic;'
|
(
|
||||||
' Old Bulgarian; Old Church Slavonic'),
|
'Church Slavic; Old Slavonic; Church Slavonic;'
|
||||||
|
' Old Bulgarian; Old Church Slavonic'
|
||||||
|
),
|
||||||
"slavon d'église; vieux slave; slavon liturgique; vieux bulgare",
|
"slavon d'église; vieux slave; slavon liturgique; vieux bulgare",
|
||||||
),
|
),
|
||||||
'chv': ISOCodeData('', 'cv', 'Chuvash', 'tchouvache'),
|
'chv': ISOCodeData('', 'cv', 'Chuvash', 'tchouvache'),
|
||||||
|
|||||||
@@ -3,7 +3,6 @@
|
|||||||
|
|
||||||
"""Post-processing image optimization of OCR PDFs."""
|
"""Post-processing image optimization of OCR PDFs."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
@@ -11,11 +10,10 @@ import sys
|
|||||||
import tempfile
|
import tempfile
|
||||||
import threading
|
import threading
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from collections.abc import Iterator, MutableSet, Sequence
|
from collections.abc import Callable, Iterator, MutableSet, Sequence
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, NamedTuple, NewType
|
from typing import Any, NamedTuple, NewType
|
||||||
from warnings import warn
|
|
||||||
from zlib import compress
|
from zlib import compress
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
@@ -30,6 +28,7 @@ from pikepdf import (
|
|||||||
Stream,
|
Stream,
|
||||||
UnsupportedImageTypeError,
|
UnsupportedImageTypeError,
|
||||||
)
|
)
|
||||||
|
from pikepdf.models.image import HifiPrintImageNotTranscodableError
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||||
@@ -91,6 +90,7 @@ def extract_image_filter(
|
|||||||
if (
|
if (
|
||||||
len(pim.filter_decodeparms) == 2
|
len(pim.filter_decodeparms) == 2
|
||||||
and first_filtdp[0] == Name.FlateDecode
|
and first_filtdp[0] == Name.FlateDecode
|
||||||
|
and first_filtdp[1] is not None
|
||||||
and first_filtdp[1].get(Name.Predictor, 1) == 1
|
and first_filtdp[1].get(Name.Predictor, 1) == 1
|
||||||
and second_filtdp[0] == Name.DCTDecode
|
and second_filtdp[0] == Name.DCTDecode
|
||||||
and not second_filtdp[1]
|
and not second_filtdp[1]
|
||||||
@@ -201,7 +201,7 @@ def extract_image_generic(
|
|||||||
with imgname.open('wb') as f:
|
with imgname.open('wb') as f:
|
||||||
ext = pim.extract_to(stream=f)
|
ext = pim.extract_to(stream=f)
|
||||||
imgname.rename(imgname.with_suffix(ext))
|
imgname.rename(imgname.with_suffix(ext))
|
||||||
except UnsupportedImageTypeError:
|
except (UnsupportedImageTypeError, HifiPrintImageNotTranscodableError):
|
||||||
return None
|
return None
|
||||||
return XrefExt(xref, ext)
|
return XrefExt(xref, ext)
|
||||||
elif (
|
elif (
|
||||||
@@ -257,6 +257,9 @@ def _find_image_xrefs_container(
|
|||||||
for _imname, image in dict(xobjs).items():
|
for _imname, image in dict(xobjs).items():
|
||||||
if image.objgen[1] != 0:
|
if image.objgen[1] != 0:
|
||||||
continue # Ignore images in an incremental PDF
|
continue # Ignore images in an incremental PDF
|
||||||
|
xref = Xref(image.objgen[0])
|
||||||
|
if xref in include_xrefs or xref in exclude_xrefs:
|
||||||
|
continue # Already processed
|
||||||
if Name.Subtype in image and image.Subtype == Name.Form:
|
if Name.Subtype in image and image.Subtype == Name.Form:
|
||||||
# Recurse into Form XObjects
|
# Recurse into Form XObjects
|
||||||
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
||||||
@@ -270,7 +273,6 @@ def _find_image_xrefs_container(
|
|||||||
depth + 1,
|
depth + 1,
|
||||||
)
|
)
|
||||||
continue
|
continue
|
||||||
xref = Xref(image.objgen[0])
|
|
||||||
if Name.SMask in image:
|
if Name.SMask in image:
|
||||||
# Ignore soft masks
|
# Ignore soft masks
|
||||||
smask_xref = Xref(image.SMask.objgen[0])
|
smask_xref = Xref(image.SMask.objgen[0])
|
||||||
|
|||||||
@@ -90,7 +90,7 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
|||||||
icc: ICC identifier such as 'sRGB'
|
icc: ICC identifier such as 'sRGB'
|
||||||
References:
|
References:
|
||||||
Adobe PDFMARK Reference:
|
Adobe PDFMARK Reference:
|
||||||
https://www.adobe.com/content/dam/acom/en/devnet/acrobat/pdfs/pdfmark_reference.pdf
|
https://opensource.adobe.com/dc-acrobat-sdk-docs/library/pdfmark/
|
||||||
"""
|
"""
|
||||||
if icc != 'sRGB':
|
if icc != 'sRGB':
|
||||||
raise NotImplementedError("Only supporting sRGB")
|
raise NotImplementedError("Only supporting sRGB")
|
||||||
|
|||||||
@@ -10,9 +10,8 @@ import atexit
|
|||||||
import logging
|
import logging
|
||||||
import re
|
import re
|
||||||
import statistics
|
import statistics
|
||||||
import sys
|
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from collections.abc import Container, Iterable, Iterator, Mapping, Sequence
|
from collections.abc import Callable, Container, Iterable, Iterator, Mapping, Sequence
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from enum import Enum, auto
|
from enum import Enum, auto
|
||||||
@@ -20,7 +19,7 @@ from functools import partial
|
|||||||
from math import hypot, inf, isclose
|
from math import hypot, inf, isclose
|
||||||
from os import PathLike
|
from os import PathLike
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Callable, NamedTuple
|
from typing import NamedTuple
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
from pdfminer.layout import LTPage, LTTextBox
|
from pdfminer.layout import LTPage, LTTextBox
|
||||||
@@ -240,7 +239,13 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
|||||||
# to do. Just pretend nothing happened, keep calm and carry on.
|
# to do. Just pretend nothing happened, keep calm and carry on.
|
||||||
warn("PDF graphics stack underflowed - PDF may be malformed")
|
warn("PDF graphics stack underflowed - PDF may be malformed")
|
||||||
elif operator == 'cm':
|
elif operator == 'cm':
|
||||||
ctm = Matrix(operands) @ ctm
|
try:
|
||||||
|
ctm = Matrix(operands) @ ctm
|
||||||
|
except ValueError:
|
||||||
|
raise InputFileError(
|
||||||
|
"PDF content stream is corrupt - this PDF is malformed. "
|
||||||
|
"Use a PDF editor that is capable of visually inspecting the PDF."
|
||||||
|
)
|
||||||
elif operator == 'Do':
|
elif operator == 'Do':
|
||||||
image_name = operands[0]
|
image_name = operands[0]
|
||||||
settings = XobjectSettings(
|
settings = XobjectSettings(
|
||||||
@@ -364,8 +369,18 @@ class ImageInfo:
|
|||||||
pim = PdfImage(pdfimage)
|
pim = PdfImage(pdfimage)
|
||||||
else:
|
else:
|
||||||
raise ValueError("Either pdfimage or inline must be set")
|
raise ValueError("Either pdfimage or inline must be set")
|
||||||
self._width = pim.width
|
if pim.obj.get(Name.SMask, None) is not None:
|
||||||
self._height = pim.height
|
# SMask is pretty much an alpha channel, but in PDF it's possible
|
||||||
|
# for channel to have different dimensions than the image
|
||||||
|
# itself. Some PDF writers use this to create a grayscale stencil
|
||||||
|
# mask. For our purposes, the effective size is the size of the
|
||||||
|
# larger component (image or smask).
|
||||||
|
smask = pim.obj[Name.SMask]
|
||||||
|
self._width = max(smask.get(Name.Width, 0), pim.width)
|
||||||
|
self._height = max(smask.get(Name.Height, 0), pim.height)
|
||||||
|
else:
|
||||||
|
self._width = pim.width
|
||||||
|
self._height = pim.height
|
||||||
|
|
||||||
# If /ImageMask is true, then this image is a stencil mask
|
# If /ImageMask is true, then this image is a stencil mask
|
||||||
# (Images that draw with this stencil mask will have a reference to
|
# (Images that draw with this stencil mask will have a reference to
|
||||||
@@ -486,7 +501,7 @@ class ImageInfo:
|
|||||||
"""Physical area of the image in square inches."""
|
"""Physical area of the image in square inches."""
|
||||||
if not self.renderable:
|
if not self.renderable:
|
||||||
return 0.0
|
return 0.0
|
||||||
return float(self.width * self.dpi.x * self.height * self.dpi.y)
|
return float((self.width / self.dpi.x) * (self.height / self.dpi.y))
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
"""Return a string representation of the image."""
|
"""Return a string representation of the image."""
|
||||||
@@ -568,7 +583,7 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
|||||||
xobjs = resources[Name.XObject].as_dict()
|
xobjs = resources[Name.XObject].as_dict()
|
||||||
for xobj in xobjs:
|
for xobj in xobjs:
|
||||||
candidate = xobjs[xobj]
|
candidate = xobjs[xobj]
|
||||||
if candidate is None or candidate[Name.Subtype] != Name.Form:
|
if candidate is None or candidate.get(Name.Subtype) != Name.Form:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
form_xobject = candidate
|
form_xobject = candidate
|
||||||
@@ -1060,18 +1075,12 @@ class PageInfo:
|
|||||||
|
|
||||||
weights = [area / total_drawn_area for area in image_areas]
|
weights = [area / total_drawn_area for area in image_areas]
|
||||||
# Calculate harmonic mean of DPIs weighted by area
|
# Calculate harmonic mean of DPIs weighted by area
|
||||||
if sys.version_info >= (3, 10):
|
weighted_dpi = statistics.harmonic_mean(image_dpis, weights)
|
||||||
weighted_dpi = statistics.harmonic_mean(image_dpis, weights)
|
|
||||||
else:
|
|
||||||
weighted_dpi = sum(weights) / sum(
|
|
||||||
weight / dpi for weight, dpi in zip(weights, image_dpis)
|
|
||||||
)
|
|
||||||
max_dpi = max(image_dpis)
|
max_dpi = max(image_dpis)
|
||||||
dpi_average_max_ratio = weighted_dpi / max_dpi
|
dpi_average_max_ratio = weighted_dpi / max_dpi
|
||||||
|
|
||||||
arg_max_dpi = image_dpis.index(max_dpi)
|
arg_max_dpi = image_dpis.index(max_dpi)
|
||||||
max_area_ratio = image_areas[arg_max_dpi] / total_drawn_area
|
max_area_ratio = image_areas[arg_max_dpi] / total_drawn_area
|
||||||
|
|
||||||
return PageResolutionProfile(
|
return PageResolutionProfile(
|
||||||
weighted_dpi,
|
weighted_dpi,
|
||||||
max_dpi,
|
max_dpi,
|
||||||
@@ -1176,7 +1185,7 @@ class PdfInfo:
|
|||||||
@property
|
@property
|
||||||
def filename(self) -> str | Path:
|
def filename(self) -> str | Path:
|
||||||
"""Return filename of PDF."""
|
"""Return filename of PDF."""
|
||||||
if not isinstance(self._infile, (str, Path)):
|
if not isinstance(self._infile, str | Path):
|
||||||
raise NotImplementedError("can't get filename from stream")
|
raise NotImplementedError("can't get filename from stream")
|
||||||
return self._infile
|
return self._infile
|
||||||
|
|
||||||
|
|||||||
@@ -5,12 +5,12 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import re
|
import re
|
||||||
from collections.abc import Mapping
|
from collections.abc import Iterator, Mapping
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from math import copysign
|
from math import copysign
|
||||||
from os import PathLike
|
from os import PathLike
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Iterator
|
from typing import Any
|
||||||
from unittest.mock import patch
|
from unittest.mock import patch
|
||||||
|
|
||||||
import pdfminer
|
import pdfminer
|
||||||
|
|||||||
@@ -3,7 +3,6 @@
|
|||||||
|
|
||||||
"""Utilities to measure OCR quality."""
|
"""Utilities to measure OCR quality."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import re
|
import re
|
||||||
|
|||||||
@@ -8,12 +8,11 @@ import logging
|
|||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
from collections.abc import Mapping, Sequence
|
from collections.abc import Callable, Mapping, Sequence
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||||
from subprocess import run as subprocess_run
|
from subprocess import run as subprocess_run
|
||||||
from typing import Callable, Union
|
|
||||||
|
|
||||||
from packaging.version import Version
|
from packaging.version import Version
|
||||||
|
|
||||||
@@ -23,7 +22,7 @@ from ocrmypdf.exceptions import MissingDependencyError
|
|||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
Args = Sequence[Union[Path, str]]
|
Args = Sequence[Path | str]
|
||||||
OsEnviron = os._Environ # pylint: disable=protected-access
|
OsEnviron = os._Environ # pylint: disable=protected-access
|
||||||
|
|
||||||
|
|
||||||
@@ -172,6 +171,7 @@ def get_version(
|
|||||||
) from e
|
) from e
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
if e.returncode != 0:
|
if e.returncode != 0:
|
||||||
|
log.exception(e)
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
f"Ran program '{program}' but it exited with an error:\n{e.output}"
|
f"Ran program '{program}' but it exited with an error:\n{e.output}"
|
||||||
) from e
|
) from e
|
||||||
|
|||||||
@@ -9,18 +9,13 @@ import os
|
|||||||
import re
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
from collections.abc import Iterable, Iterator
|
from collections.abc import Callable, Iterable, Iterator
|
||||||
from itertools import chain
|
from itertools import chain
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, TypeVar
|
from typing import Any, TypeAlias, TypeVar
|
||||||
|
|
||||||
from packaging.version import InvalidVersion, Version
|
from packaging.version import InvalidVersion, Version
|
||||||
|
|
||||||
if sys.version_info >= (3, 10):
|
|
||||||
from typing import TypeAlias
|
|
||||||
else:
|
|
||||||
from typing_extensions import TypeAlias # pragma: no cover
|
|
||||||
|
|
||||||
if sys.platform == 'win32':
|
if sys.platform == 'win32':
|
||||||
# mypy understands 'if sys.platform' better than try/except ModuleNotFoundError
|
# mypy understands 'if sys.platform' better than try/except ModuleNotFoundError
|
||||||
import winreg # pylint: disable=import-error
|
import winreg # pylint: disable=import-error
|
||||||
@@ -84,7 +79,7 @@ def registry_path_ghostscript(env=None) -> Iterator[Path]:
|
|||||||
registry_subkeys(k), key=ghostscript_version_key, default=(0, 0, 0)
|
registry_subkeys(k), key=ghostscript_version_key, default=(0, 0, 0)
|
||||||
)
|
)
|
||||||
with winreg.OpenKey(
|
with winreg.OpenKey(
|
||||||
winreg.HKEY_LOCAL_MACHINE, fr"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
|
winreg.HKEY_LOCAL_MACHINE, rf"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
|
||||||
) as k:
|
) as k:
|
||||||
for _, gs_path, _ in registry_values(k):
|
for _, gs_path, _ in registry_values(k):
|
||||||
yield Path(gs_path) / 'bin'
|
yield Path(gs_path) / 'bin'
|
||||||
|
|||||||
@@ -3,7 +3,6 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from pathlib import Path
|
|
||||||
from subprocess import CalledProcessError
|
from subprocess import CalledProcessError
|
||||||
from unittest.mock import patch
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
|||||||
@@ -169,22 +169,25 @@ class CacheOcrEngine(TesseractOcrEngine):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def get_orientation(input_file, options):
|
def get_orientation(input_file, options):
|
||||||
with CacheOcrEngine.lock, patch(
|
with (
|
||||||
'ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)
|
CacheOcrEngine.lock,
|
||||||
|
patch('ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)),
|
||||||
):
|
):
|
||||||
return TesseractOcrEngine.get_orientation(input_file, options)
|
return TesseractOcrEngine.get_orientation(input_file, options)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def get_deskew(input_file, options) -> float:
|
def get_deskew(input_file, options) -> float:
|
||||||
with CacheOcrEngine.lock, patch(
|
with (
|
||||||
'ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)
|
CacheOcrEngine.lock,
|
||||||
|
patch('ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)),
|
||||||
):
|
):
|
||||||
return TesseractOcrEngine.get_deskew(input_file, options)
|
return TesseractOcrEngine.get_deskew(input_file, options)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||||
with CacheOcrEngine.lock, patch(
|
with (
|
||||||
'ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)
|
CacheOcrEngine.lock,
|
||||||
|
patch('ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)),
|
||||||
):
|
):
|
||||||
TesseractOcrEngine.generate_hocr(
|
TesseractOcrEngine.generate_hocr(
|
||||||
input_file, output_hocr, output_text, options
|
input_file, output_hocr, output_text, options
|
||||||
@@ -192,8 +195,9 @@ class CacheOcrEngine(TesseractOcrEngine):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def generate_pdf(input_file, output_pdf, output_text, options):
|
def generate_pdf(input_file, output_pdf, output_text, options):
|
||||||
with CacheOcrEngine.lock, patch(
|
with (
|
||||||
'ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)
|
CacheOcrEngine.lock,
|
||||||
|
patch('ocrmypdf._exec.tesseract.run', new=partial(cached_run, options)),
|
||||||
):
|
):
|
||||||
TesseractOcrEngine.generate_pdf(
|
TesseractOcrEngine.generate_pdf(
|
||||||
input_file, output_pdf, output_text, options
|
input_file, output_pdf, output_text, options
|
||||||
|
|||||||
@@ -72,9 +72,10 @@ class FixedRotateNoopOcrEngine(OcrEngine):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||||
with Image.open(input_file) as im, open(
|
with (
|
||||||
output_hocr, 'w', encoding='utf-8'
|
Image.open(input_file) as im,
|
||||||
) as f:
|
open(output_hocr, 'w', encoding='utf-8') as f,
|
||||||
|
):
|
||||||
w, h = im.size
|
w, h = im.size
|
||||||
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
|
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
|
||||||
with open(output_text, 'w') as f:
|
with open(output_text, 'w') as f:
|
||||||
|
|||||||
@@ -70,9 +70,10 @@ class NoopOcrEngine(OcrEngine):
|
|||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||||
with Image.open(input_file) as im, open(
|
with (
|
||||||
output_hocr, 'w', encoding='utf-8'
|
Image.open(input_file) as im,
|
||||||
) as f:
|
open(output_hocr, 'w', encoding='utf-8') as f,
|
||||||
|
):
|
||||||
w, h = im.size
|
w, h = im.size
|
||||||
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
|
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
|
||||||
with open(output_text, 'w') as f:
|
with open(output_text, 'w') as f:
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ ensure we fail with an error rather than deadlock in such cases.
|
|||||||
Page 4 was chosen because of this number's association with bad luck
|
Page 4 was chosen because of this number's association with bad luck
|
||||||
in many East Asian cultures.
|
in many East Asian cultures.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
# type: ignore
|
# type: ignore
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
|||||||
@@ -29,6 +29,18 @@ def test_stream_api(resources: Path):
|
|||||||
assert b'%PDF' in out.read(1024)
|
assert b'%PDF' in out.read(1024)
|
||||||
|
|
||||||
|
|
||||||
|
def test_sidecar_stringio(resources: Path, outdir: Path, outpdf: Path):
|
||||||
|
s = BytesIO()
|
||||||
|
ocrmypdf.ocr(
|
||||||
|
resources / 'ccitt.pdf',
|
||||||
|
outpdf,
|
||||||
|
plugins=['tests/plugins/tesseract_cache.py'],
|
||||||
|
sidecar=s
|
||||||
|
)
|
||||||
|
s.seek(0)
|
||||||
|
assert b'the' in s.getvalue()
|
||||||
|
|
||||||
|
|
||||||
def test_hocr_api_multipage(resources: Path, outdir: Path, outpdf: Path):
|
def test_hocr_api_multipage(resources: Path, outdir: Path, outpdf: Path):
|
||||||
ocrmypdf.api._pdf_to_hocr(
|
ocrmypdf.api._pdf_to_hocr(
|
||||||
resources / 'multipage.pdf',
|
resources / 'multipage.pdf',
|
||||||
@@ -62,3 +74,4 @@ def test_hocr_to_pdf_api(resources: Path, outdir: Path, outpdf: Path):
|
|||||||
|
|
||||||
text = extract_text(outpdf)
|
text = extract_text(outpdf)
|
||||||
assert 'hocr' in text and 'the' not in text
|
assert 'hocr' in text and 'the' not in text
|
||||||
|
|
||||||
|
|||||||
+5
-3
@@ -11,9 +11,11 @@ import ocrmypdf
|
|||||||
|
|
||||||
|
|
||||||
def test_no_glyphless_graft(resources, outdir):
|
def test_no_glyphless_graft(resources, outdir):
|
||||||
with pikepdf.open(resources / 'francais.pdf') as pdf, pikepdf.open(
|
with (
|
||||||
resources / 'aspect.pdf'
|
pikepdf.open(resources / 'francais.pdf') as pdf,
|
||||||
) as pdf_aspect, pikepdf.open(resources / 'cmyk.pdf') as pdf_cmyk:
|
pikepdf.open(resources / 'aspect.pdf') as pdf_aspect,
|
||||||
|
pikepdf.open(resources / 'cmyk.pdf') as pdf_cmyk,
|
||||||
|
):
|
||||||
pdf.pages.extend(pdf_aspect.pages)
|
pdf.pages.extend(pdf_aspect.pages)
|
||||||
pdf.pages.extend(pdf_cmyk.pages)
|
pdf.pages.extend(pdf_cmyk.pages)
|
||||||
pdf.save(outdir / 'test.pdf')
|
pdf.save(outdir / 'test.pdf')
|
||||||
|
|||||||
+1
-1
@@ -17,7 +17,7 @@ from PIL import Image
|
|||||||
|
|
||||||
import ocrmypdf
|
import ocrmypdf
|
||||||
from ocrmypdf._exec import tesseract
|
from ocrmypdf._exec import tesseract
|
||||||
from ocrmypdf.exceptions import ExitCode, MissingDependencyError, OutputFileAccessError
|
from ocrmypdf.exceptions import ExitCode, MissingDependencyError
|
||||||
from ocrmypdf.pdfa import file_claims_pdfa
|
from ocrmypdf.pdfa import file_claims_pdfa
|
||||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
||||||
from ocrmypdf.subprocess import get_version
|
from ocrmypdf.subprocess import get_version
|
||||||
|
|||||||
@@ -35,9 +35,10 @@ def test_preserve_docinfo(output_type, resources, outpdf):
|
|||||||
'--plugin',
|
'--plugin',
|
||||||
'tests/plugins/tesseract_noop.py',
|
'tests/plugins/tesseract_noop.py',
|
||||||
)
|
)
|
||||||
with pikepdf.open(resources / 'graph.pdf') as pdf_before, pikepdf.open(
|
with (
|
||||||
output
|
pikepdf.open(resources / 'graph.pdf') as pdf_before,
|
||||||
) as pdf_after:
|
pikepdf.open(output) as pdf_after,
|
||||||
|
):
|
||||||
for key in ('/Title', '/Author'):
|
for key in ('/Title', '/Author'):
|
||||||
assert pdf_before.docinfo[key] == pdf_after.docinfo[key]
|
assert pdf_before.docinfo[key] == pdf_after.docinfo[key]
|
||||||
pdfa_info = file_claims_pdfa(str(output))
|
pdfa_info = file_claims_pdfa(str(output))
|
||||||
|
|||||||
@@ -9,10 +9,11 @@ import pytest
|
|||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf._exec import ghostscript, tesseract
|
from ocrmypdf._exec import ghostscript, tesseract
|
||||||
|
from ocrmypdf.exceptions import ExitCode
|
||||||
from ocrmypdf.helpers import Resolution
|
from ocrmypdf.helpers import Resolution
|
||||||
from ocrmypdf.pdfinfo import PdfInfo
|
from ocrmypdf.pdfinfo import PdfInfo
|
||||||
|
|
||||||
from .conftest import check_ocrmypdf, have_unpaper
|
from .conftest import check_ocrmypdf, have_unpaper, run_ocrmypdf
|
||||||
|
|
||||||
RENDERERS = ['hocr', 'sandwich']
|
RENDERERS = ['hocr', 'sandwich']
|
||||||
|
|
||||||
@@ -107,7 +108,7 @@ def test_non_square_resolution(renderer, resources, outpdf):
|
|||||||
in_pageinfo = PdfInfo(resources / 'aspect.pdf')
|
in_pageinfo = PdfInfo(resources / 'aspect.pdf')
|
||||||
assert in_pageinfo[0].dpi.x != in_pageinfo[0].dpi.y
|
assert in_pageinfo[0].dpi.x != in_pageinfo[0].dpi.y
|
||||||
|
|
||||||
check_ocrmypdf(
|
proc = run_ocrmypdf(
|
||||||
resources / 'aspect.pdf',
|
resources / 'aspect.pdf',
|
||||||
outpdf,
|
outpdf,
|
||||||
'--pdf-renderer',
|
'--pdf-renderer',
|
||||||
@@ -115,6 +116,10 @@ def test_non_square_resolution(renderer, resources, outpdf):
|
|||||||
'--plugin',
|
'--plugin',
|
||||||
'tests/plugins/tesseract_cache.py',
|
'tests/plugins/tesseract_cache.py',
|
||||||
)
|
)
|
||||||
|
# PDF/A conversion can fail for this file if Ghostscript >= 10.3, so don't test
|
||||||
|
# exit code in that case
|
||||||
|
if proc.returncode != ExitCode.pdfa_conversion_failed:
|
||||||
|
proc.check_returncode()
|
||||||
|
|
||||||
out_pageinfo = PdfInfo(outpdf)
|
out_pageinfo = PdfInfo(outpdf)
|
||||||
|
|
||||||
|
|||||||
@@ -23,4 +23,4 @@ def test_semfree(resources, outpdf):
|
|||||||
'--plugin',
|
'--plugin',
|
||||||
'tests/plugins/tesseract_noop.py',
|
'tests/plugins/tesseract_noop.py',
|
||||||
)
|
)
|
||||||
assert exitcode == ExitCode.ok
|
assert exitcode in (ExitCode.ok, ExitCode.pdfa_conversion_failed)
|
||||||
|
|||||||
@@ -13,9 +13,9 @@ import pytest
|
|||||||
|
|
||||||
from ocrmypdf import pdfinfo
|
from ocrmypdf import pdfinfo
|
||||||
from ocrmypdf._exec import tesseract
|
from ocrmypdf._exec import tesseract
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import BadArgsError, ExitCode, MissingDependencyError
|
||||||
|
|
||||||
from .conftest import check_ocrmypdf
|
from .conftest import check_ocrmypdf, run_ocrmypdf_api
|
||||||
|
|
||||||
# pylint: disable=redefined-outer-name
|
# pylint: disable=redefined-outer-name
|
||||||
|
|
||||||
@@ -144,3 +144,10 @@ def test_tesseract_log_output_raises(caplog):
|
|||||||
with pytest.raises(tesseract.TesseractConfigError):
|
with pytest.raises(tesseract.TesseractConfigError):
|
||||||
tesseract.tesseract_log_output(b'parameter not found: moo')
|
tesseract.tesseract_log_output(b'parameter not found: moo')
|
||||||
assert 'not found' in caplog.text
|
assert 'not found' in caplog.text
|
||||||
|
|
||||||
|
|
||||||
|
def test_blocked_language(resources, no_outpdf):
|
||||||
|
infile = resources / 'masks.pdf'
|
||||||
|
for bad_lang in ['osd', 'equ']:
|
||||||
|
with pytest.raises(BadArgsError):
|
||||||
|
run_ocrmypdf_api(infile, no_outpdf, '-l', bad_lang)
|
||||||
|
|||||||
@@ -7,10 +7,9 @@ from math import isclose
|
|||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from ocrmypdf.exceptions import ExitCode
|
|
||||||
from ocrmypdf.pdfinfo import PdfInfo
|
from ocrmypdf.pdfinfo import PdfInfo
|
||||||
|
|
||||||
from .conftest import check_ocrmypdf, run_ocrmypdf_api
|
from .conftest import check_ocrmypdf
|
||||||
|
|
||||||
# pylint: disable=redefined-outer-name
|
# pylint: disable=redefined-outer-name
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user