Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8132a4ae10 | ||
|
|
d5128c5cf5 | ||
|
|
270e31fa67 | ||
|
|
85e31d0a19 | ||
|
|
ea36aedb5f | ||
|
|
bd4d44e182 | ||
|
|
8fcf358934 | ||
|
|
47b0f28564 | ||
|
|
7018e2b247 | ||
|
|
8d12ecb798 | ||
|
|
0ab29ec0ba | ||
|
|
179714770a | ||
|
|
f04f45545c | ||
|
|
a3a083c125 | ||
|
|
d855f63985 | ||
|
|
d4863cbf0f | ||
|
|
7b8f081fbf | ||
|
|
7d33039bcd | ||
|
|
fde886baf4 | ||
|
|
146da79c00 | ||
|
|
2fc3b0d973 | ||
|
|
5667424530 | ||
|
|
8add531ffd | ||
|
|
0388c23ae7 | ||
|
|
9b77daae7c | ||
|
|
3e1b3ec98d | ||
|
|
0f0ca6f517 | ||
|
|
c93349c350 | ||
|
|
0c287929c2 | ||
|
|
23a37fc35c | ||
|
|
162a47f98e | ||
|
|
0239f69912 | ||
|
|
2ad8961d0b | ||
|
|
eec8a2b574 | ||
|
|
6c78076bea | ||
|
|
2637e84691 | ||
|
|
e8c82ee4b6 | ||
|
|
de2bb5ce8c | ||
|
|
ec1c377532 | ||
|
|
173428e81a | ||
|
|
67ed29dcea | ||
|
|
3454c050ed | ||
|
|
5ee99b26e7 | ||
|
|
ac3aa67d8a | ||
|
|
1768a1eda9 | ||
|
|
5902fe45c1 | ||
|
|
78981641f0 | ||
|
|
c77ae4b34c | ||
|
|
b2cbbf0099 | ||
|
|
6b6c34af01 | ||
|
|
be12f7a728 | ||
|
|
e3c813fc67 | ||
|
|
35a1eaf62a | ||
|
|
d393d18c13 | ||
|
|
54e622ad10 | ||
|
|
330352aeed | ||
|
|
ac2fc49208 | ||
|
|
4bee7355e9 | ||
|
|
86f2b1f9a7 | ||
|
|
3002409e49 | ||
|
|
0cf6828c20 | ||
|
|
331c829b6e | ||
|
|
06a5e0c3f6 | ||
|
|
811f23381a | ||
|
|
a371655052 | ||
|
|
a6ce35b13a | ||
|
|
45added738 | ||
|
|
6e20439c91 | ||
|
|
72e056436c | ||
|
|
e02ba19097 | ||
|
|
d3b858f994 | ||
|
|
19045c4f21 | ||
|
|
f4d89fe6cc | ||
|
|
ab85c0f5a9 | ||
|
|
32693b683d | ||
|
|
b5dc276ba1 | ||
|
|
7c38c71794 | ||
|
|
a80e7a127b | ||
|
|
cf3309555f | ||
|
|
f80dd0d86a | ||
|
|
1ba2bce486 | ||
|
|
050dd1f5a8 | ||
|
|
e44a57aec0 | ||
|
|
d94d2671c3 | ||
|
|
5124daa79f | ||
|
|
7293847da7 | ||
|
|
59fd0ac587 | ||
|
|
90619b308c | ||
|
|
d0d49ce989 | ||
|
|
bf0224faa4 | ||
|
|
ae2f8ed8f1 | ||
|
|
14ac9b0560 | ||
|
|
dbe6148d41 | ||
|
|
36d4c2dbbc | ||
|
|
adbffb7bd9 | ||
|
|
05ecb6ca46 | ||
|
|
0a7b60cda5 | ||
|
|
5f211ecf6f | ||
|
|
417ee067a2 | ||
|
|
c4649dabef | ||
|
|
6eadd65dfb | ||
|
|
e8ed510543 | ||
|
|
22d35c199d | ||
|
|
5a82ad63c9 | ||
|
|
4769a6c50b | ||
|
|
9004009adc | ||
|
|
11221f9912 | ||
|
|
177349cc84 | ||
|
|
070c9772ce | ||
|
|
1bc09045a5 | ||
|
|
e46a18dd2f | ||
|
|
c64871c2ed | ||
|
|
de909fb99a | ||
|
|
731b2fc477 | ||
|
|
214f6ec759 | ||
|
|
080aa4dbd1 | ||
|
|
7af5dcd4a4 | ||
|
|
fe9f52fbe7 | ||
|
|
fcbdeb8dbe | ||
|
|
cb251a8d03 | ||
|
|
3731fdfd72 | ||
|
|
b2e6a6431e | ||
|
|
9ff1e56bf6 | ||
|
|
2b30f74fce | ||
|
|
10f4c48e0b | ||
|
|
37d5c086bb | ||
|
|
91830627e5 | ||
|
|
f2fc37b257 | ||
|
|
a99e40fa84 | ||
|
|
9ce692a6f1 | ||
|
|
a3c49b8f31 | ||
|
|
33b70be7d5 | ||
|
|
4924b11b6b | ||
|
|
1d0e4e7c9f | ||
|
|
9b8d14d16e | ||
|
|
b7eb93eb79 | ||
|
|
42c0d0f48f | ||
|
|
5fce50ff7c | ||
|
|
765ed4c386 | ||
|
|
1b2849ec0a | ||
|
|
4f604591b4 | ||
|
|
01dc8e23ff | ||
|
|
b432770cfc | ||
|
|
5502fb8d9f | ||
|
|
e66922b030 | ||
|
|
00e9759b16 | ||
|
|
ba10c5345b | ||
|
|
8a5f94988a | ||
|
|
aa73e3c69f | ||
|
|
9d5fa05a00 | ||
|
|
997380e567 | ||
|
|
2685f910b1 | ||
|
|
bfcc586032 | ||
|
|
2d77b95fd9 | ||
|
|
f072e91120 | ||
|
|
efa2bca8a3 | ||
|
|
b039010d3e | ||
|
|
3f7cefcf5d | ||
|
|
45f97d1565 | ||
|
|
1281f8eb68 | ||
|
|
9ef61864fb | ||
|
|
90b2119ad3 | ||
|
|
f0cc7f2230 | ||
|
|
d60a384aab | ||
|
|
14a6093636 | ||
|
|
54b42d73ab | ||
|
|
9abed14f1f | ||
|
|
d09f61d4fe | ||
|
|
4a0130649e | ||
|
|
a0224d94ca | ||
|
|
9e7b9de830 | ||
|
|
08fc5fc01d | ||
|
|
110c75cba2 | ||
|
|
5afca3f342 | ||
|
|
d9eb0ba7ef | ||
|
|
46d0978a09 | ||
|
|
4e35100978 | ||
|
|
7bd0e43243 | ||
|
|
9cd97da5f2 | ||
|
|
d002703c41 | ||
|
|
b1fecf3b05 | ||
|
|
c2ccc7f29d | ||
|
|
36dfd12e2c | ||
|
|
7da4e6ca7f | ||
|
|
16fc52079d | ||
|
|
f37decf3b3 | ||
|
|
4ad4a13ef0 | ||
|
|
6f61f69a8d | ||
|
|
88831e8ab1 | ||
|
|
2ebc36fcec | ||
|
|
1709e23701 | ||
|
|
2e55cb5980 | ||
|
|
6dbaebdc0c | ||
|
|
5156fe7662 | ||
|
|
2c99c89e45 | ||
|
|
74286e7e1e | ||
|
|
2e937dee9f | ||
|
|
23f3830533 | ||
|
|
305e6594be | ||
|
|
f4155dca77 | ||
|
|
545cd031b0 | ||
|
|
a1c7826336 | ||
|
|
c5359bd990 | ||
|
|
7f77308846 | ||
|
|
8e7c5f3001 | ||
|
|
79db985181 | ||
|
|
7d23a661fc | ||
|
|
99e94807c5 | ||
|
|
8412de9344 | ||
|
|
b458b1422b | ||
|
|
76bd8cab13 | ||
|
|
ef70c9499e | ||
|
|
47dcb6fcd0 | ||
|
|
88d2949e6b | ||
|
|
c9389c7713 | ||
|
|
4d2f499f97 | ||
|
|
4104904a1e | ||
|
|
1a0a797ca6 | ||
|
|
670ce2b969 | ||
|
|
7e97981114 | ||
|
|
53db866ef9 | ||
|
|
d591a3e059 | ||
|
|
37c050aa4f | ||
|
|
4b9ea40a0c | ||
|
|
165432486b | ||
|
|
d619fac0bd | ||
|
|
acc70036cc | ||
|
|
80b7cf6330 | ||
|
|
8a8c06c79c | ||
|
|
67773da309 | ||
|
|
d5a9861d5c | ||
|
|
9ffe829a10 | ||
|
|
8a3b82e364 | ||
|
|
580822a6a2 | ||
|
|
9f3a52fd12 | ||
|
|
52e829d845 | ||
|
|
2b2e5c271a | ||
|
|
5fe3102e4e | ||
|
|
5b57520c98 |
+13
-4
@@ -1,5 +1,5 @@
|
|||||||
# OCRmyPDF
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
#
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
FROM ubuntu:22.04 as base
|
FROM ubuntu:22.04 as base
|
||||||
|
|
||||||
@@ -25,7 +25,9 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
|||||||
libffi-dev \
|
libffi-dev \
|
||||||
ca-certificates \
|
ca-certificates \
|
||||||
curl \
|
curl \
|
||||||
git
|
git \
|
||||||
|
libcairo2-dev \
|
||||||
|
pkg-config
|
||||||
|
|
||||||
# Get the latest pip (Ubuntu version doesn't support manylinux2010)
|
# Get the latest pip (Ubuntu version doesn't support manylinux2010)
|
||||||
RUN \
|
RUN \
|
||||||
@@ -50,8 +52,15 @@ RUN pip3 install --no-cache-dir .[test,webservice,watcher]
|
|||||||
|
|
||||||
FROM base
|
FROM base
|
||||||
|
|
||||||
|
# For Tesseract 5
|
||||||
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
|
software-properties-common gpg-agent
|
||||||
|
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr-devel
|
||||||
|
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
|
fonts-droid-fallback \
|
||||||
|
jbig2dec \
|
||||||
img2pdf \
|
img2pdf \
|
||||||
libsm6 libxext6 libxrender-dev \
|
libsm6 libxext6 libxrender-dev \
|
||||||
pngquant \
|
pngquant \
|
||||||
@@ -74,7 +83,7 @@ COPY --from=builder /app/misc/webservice.py /app/
|
|||||||
COPY --from=builder /app/misc/watcher.py /app/
|
COPY --from=builder /app/misc/watcher.py /app/
|
||||||
|
|
||||||
# Copy minimal project files to get the test suite.
|
# Copy minimal project files to get the test suite.
|
||||||
COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/
|
COPY --from=builder /app/pyproject.toml /app/README.md /app/
|
||||||
COPY --from=builder /app/tests /app/tests
|
COPY --from=builder /app/tests /app/tests
|
||||||
|
|
||||||
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
||||||
|
|||||||
@@ -1,3 +1,6 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
# dotfiles
|
# dotfiles
|
||||||
.*
|
.*
|
||||||
!.coveragerc
|
!.coveragerc
|
||||||
|
|||||||
@@ -1,3 +1,6 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
# Always use Unix convention for new lines
|
# Always use Unix convention for new lines
|
||||||
* text eol=lf
|
* text eol=lf
|
||||||
|
|
||||||
|
|||||||
@@ -1,3 +1,6 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
# These are supported funding model platforms
|
# These are supported funding model platforms
|
||||||
|
|
||||||
github: # Replace with up to 4 GitHub Sponsors-enabled usernames e.g., [user1, user2]
|
github: # Replace with up to 4 GitHub Sponsors-enabled usernames e.g., [user1, user2]
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
name: General issues
|
||||||
|
description: Installation, packages, dependencies, "nothing works", test suite failures...
|
||||||
|
title: "[Bug]: "
|
||||||
|
labels: ["bug", "triage"]
|
||||||
|
assignees:
|
||||||
|
- jbarlow83
|
||||||
|
body:
|
||||||
|
- type: markdown
|
||||||
|
attributes:
|
||||||
|
value: |
|
||||||
|
Thanks for taking the time to fill out this bug report!
|
||||||
|
- type: textarea
|
||||||
|
id: what-happened
|
||||||
|
attributes:
|
||||||
|
label: What were you trying to do?
|
||||||
|
description: Also tell us, what did you expect to happen?
|
||||||
|
placeholder: Tell us what you see!
|
||||||
|
validations:
|
||||||
|
required: true
|
||||||
|
- type: dropdown
|
||||||
|
id: packaging-system
|
||||||
|
attributes:
|
||||||
|
label: Where are you installing from?
|
||||||
|
multiple: true
|
||||||
|
options:
|
||||||
|
- PyPI (pip, poetry, pipx, etc.)
|
||||||
|
- Linux package manager (apt, dnf, etc.)
|
||||||
|
- Wndows package manager (chocolatey, etc.)
|
||||||
|
- Homebrew
|
||||||
|
- Docker container
|
||||||
|
- Ubuntu snap
|
||||||
|
- Conda
|
||||||
|
- source build
|
||||||
|
validations:
|
||||||
|
required: true
|
||||||
|
- type: dropdown
|
||||||
|
id: operating-system
|
||||||
|
attributes:
|
||||||
|
label: What operating system are you working on?
|
||||||
|
multiple: true
|
||||||
|
options:
|
||||||
|
- Linux
|
||||||
|
- Windows
|
||||||
|
- macOS
|
||||||
|
- BSD
|
||||||
|
- type: textarea
|
||||||
|
id: logs
|
||||||
|
attributes:
|
||||||
|
label: Relevant log output
|
||||||
|
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||||
|
render: plain text
|
||||||
@@ -1,32 +0,0 @@
|
|||||||
---
|
|
||||||
name: General issues
|
|
||||||
about: Installation, packages, dependencies, "nothing works", test suite failures...
|
|
||||||
title: ''
|
|
||||||
labels: ''
|
|
||||||
assignees: ''
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Describe the bug**
|
|
||||||
What's the problem?
|
|
||||||
|
|
||||||
**To Reproduce**
|
|
||||||
Steps to reproduce the behavior.
|
|
||||||
|
|
||||||
**Expected behavior**
|
|
||||||
What did you expected to happen?
|
|
||||||
|
|
||||||
**Screenshots**
|
|
||||||
If applicable, add screenshots to help explain your problem.
|
|
||||||
|
|
||||||
**System (please complete the following information):**
|
|
||||||
- OS:
|
|
||||||
- Python version:
|
|
||||||
- OCRmyPDF version:
|
|
||||||
|
|
||||||
**Installation**
|
|
||||||
How did you install OCRmyPDF? Did you install it from your operating system's
|
|
||||||
package manager, or using pip?
|
|
||||||
|
|
||||||
**Additional context**
|
|
||||||
Add any other context about the problem here.
|
|
||||||
@@ -1,40 +0,0 @@
|
|||||||
---
|
|
||||||
name: Problem with a specific input file
|
|
||||||
about: Something went wrong while trying to OCR a specific file
|
|
||||||
title: ''
|
|
||||||
labels: ''
|
|
||||||
assignees: ''
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Describe the bug**
|
|
||||||
A clear and concise description of what the bug is.
|
|
||||||
|
|
||||||
**To Reproduce**
|
|
||||||
What command line or API call were you trying to run?
|
|
||||||
|
|
||||||
```bash
|
|
||||||
ocrmypdf ...arguments... input.pdf output.pdf
|
|
||||||
```
|
|
||||||
|
|
||||||
Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful.
|
|
||||||
|
|
||||||
**Example file**
|
|
||||||
If your issue is a problem that affects only certain files, and we will require an input file (PDF or image) that demonstrates your issue.
|
|
||||||
|
|
||||||
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/ocrmypdf/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
|
||||||
|
|
||||||
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
|
||||||
|
|
||||||
*(Issues without example files usually cannot be resolved. It's like reporting an issue against a web browser without providing a URL.)*
|
|
||||||
|
|
||||||
**Expected behavior**
|
|
||||||
A clear and concise description of what you expected to happen.
|
|
||||||
|
|
||||||
**Screenshots**
|
|
||||||
If applicable, add screenshots to help explain your problem.
|
|
||||||
|
|
||||||
**System**
|
|
||||||
- OS: [e.g. Linux, Windows, macOS]
|
|
||||||
- OCRmyPDF Version: ``ocrmypdf --version``
|
|
||||||
- How did you install ocrmypdf? Did you use a system package manager, `pip`, or a Docker image?
|
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
name: Problem with specific file
|
||||||
|
description: Something went wrong while trying to OCR a specific file
|
||||||
|
title: "[Bug]: "
|
||||||
|
labels: ["bug", "triage"]
|
||||||
|
assignees:
|
||||||
|
- jbarlow83
|
||||||
|
body:
|
||||||
|
- type: markdown
|
||||||
|
attributes:
|
||||||
|
value: |
|
||||||
|
Thanks for taking the time to describe this issue with a particular file.
|
||||||
|
- type: textarea
|
||||||
|
id: what-happened
|
||||||
|
attributes:
|
||||||
|
label: Describe the bug
|
||||||
|
description: A clear and concise description of what the bug is.
|
||||||
|
placeholder: Tell us what you see!
|
||||||
|
validations:
|
||||||
|
required: true
|
||||||
|
- type: textarea
|
||||||
|
id: reproduce
|
||||||
|
attributes:
|
||||||
|
label: Steps to reproduce
|
||||||
|
description: Please include steps to reproduce
|
||||||
|
value: |
|
||||||
|
1. Run ocrmypdf -v1 ...arguments... input.pdf output.pdf
|
||||||
|
2. Open output.pdf
|
||||||
|
3. ...
|
||||||
|
render: plain text
|
||||||
|
- type: textarea
|
||||||
|
id: files
|
||||||
|
attributes:
|
||||||
|
label: Files
|
||||||
|
description: Please attach the input and output files, or any screenshots that may be helpful.
|
||||||
|
placeholder: Drag and drop files here
|
||||||
|
- type: dropdown
|
||||||
|
id: packaging-system
|
||||||
|
attributes:
|
||||||
|
label: How did you download and install the software?
|
||||||
|
multiple: true
|
||||||
|
options:
|
||||||
|
- PyPI (pip, poetry, pipx, etc.)
|
||||||
|
- Linux package manager (apt, dnf, etc.)
|
||||||
|
- Windows package manager (chocolatey, etc.)
|
||||||
|
- Homebrew
|
||||||
|
- Docker container
|
||||||
|
- Ubuntu snap
|
||||||
|
- Conda
|
||||||
|
- source build
|
||||||
|
- type: input
|
||||||
|
id: version
|
||||||
|
attributes:
|
||||||
|
label: OCRmyPDF version
|
||||||
|
description: Paste "ocrmypdf --version" here
|
||||||
|
placeholder: ocrmypdf --version
|
||||||
|
- type: textarea
|
||||||
|
id: logs
|
||||||
|
attributes:
|
||||||
|
label: Relevant log output
|
||||||
|
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||||
|
placeholder: Run OCRmyPDF with verbosity `-v1` to get more detailed logging output.
|
||||||
|
render: plain text
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
name: Feature request
|
||||||
|
description: Suggest an idea for this project
|
||||||
|
title: "[Feature]: "
|
||||||
|
labels: ["enhancement", "triage"]
|
||||||
|
assignees:
|
||||||
|
- jbarlow83
|
||||||
|
body:
|
||||||
|
- type: textarea
|
||||||
|
id: feature
|
||||||
|
attributes:
|
||||||
|
label: Describe the proposed feature
|
||||||
|
description: A clear and concise description of what the desired is.
|
||||||
@@ -1,27 +0,0 @@
|
|||||||
---
|
|
||||||
name: Feature request
|
|
||||||
about: Suggest an idea for this project
|
|
||||||
title: ''
|
|
||||||
labels: ''
|
|
||||||
assignees: ''
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Is your feature request related to a problem? Please describe.**
|
|
||||||
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
|
||||||
|
|
||||||
**Describe the solution you'd like**
|
|
||||||
A clear and concise description of what you want to happen.
|
|
||||||
|
|
||||||
**Describe alternatives you've considered**
|
|
||||||
A clear and concise description of any alternative solutions or features you've considered. Please include the versions of OCRmyPDF and other supporting programs (Tesseract OCR, Ghostscript) - maybe an alternative already exists in a newer version.
|
|
||||||
|
|
||||||
**Example file**
|
|
||||||
If your issue concerns how OCRmyPDF processes certain files, and please provide an example file that helps illustrate how OCRmyPDF's output could be improve.
|
|
||||||
|
|
||||||
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/ocrmypdf/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
|
||||||
|
|
||||||
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
|
||||||
|
|
||||||
**Additional context**
|
|
||||||
Add any other context or screenshots about the feature request here.
|
|
||||||
@@ -1,3 +1,6 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
# To get started with Dependabot version updates, you'll need to specify which
|
# To get started with Dependabot version updates, you'll need to specify which
|
||||||
# package ecosystems to update and where the package manifests are located.
|
# package ecosystems to update and where the package manifests are located.
|
||||||
# Please see the documentation for all configuration options:
|
# Please see the documentation for all configuration options:
|
||||||
|
|||||||
+65
-52
@@ -1,9 +1,11 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
name: Test and deploy
|
name: Test and deploy
|
||||||
|
|
||||||
on:
|
on:
|
||||||
push:
|
push:
|
||||||
branches:
|
branches:
|
||||||
- master
|
- main
|
||||||
- ci
|
- ci
|
||||||
- release/*
|
- release/*
|
||||||
- feature/*
|
- feature/*
|
||||||
@@ -20,19 +22,15 @@ jobs:
|
|||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
include:
|
include:
|
||||||
- os: ubuntu-18.04
|
- os: ubuntu-22.04
|
||||||
python: "3.7"
|
|
||||||
- os: ubuntu-20.04
|
|
||||||
python: "3.8"
|
|
||||||
- os: ubuntu-20.04
|
|
||||||
python: "3.9"
|
python: "3.9"
|
||||||
- os: ubuntu-20.04
|
- os: ubuntu-22.04
|
||||||
python: "3.10"
|
python: "3.10"
|
||||||
- os: ubuntu-latest
|
- os: ubuntu-22.04
|
||||||
python: "3.9"
|
python: "3.11"
|
||||||
- os: ubuntu-latest
|
#- os: ubuntu-latest
|
||||||
python: "pypy-3.8"
|
# python: "pypy3.9"
|
||||||
- os: ubuntu-latest
|
- os: ubuntu-22.04
|
||||||
python: "3.9"
|
python: "3.9"
|
||||||
tesseract5: true
|
tesseract5: true
|
||||||
|
|
||||||
@@ -41,7 +39,7 @@ jobs:
|
|||||||
PYTHON: ${{ matrix.python }}
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
@@ -62,6 +60,7 @@ jobs:
|
|||||||
curl \
|
curl \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
img2pdf \
|
img2pdf \
|
||||||
|
libexempi8 \
|
||||||
libffi-dev \
|
libffi-dev \
|
||||||
libsm6 libxext6 libxrender-dev \
|
libsm6 libxext6 libxrender-dev \
|
||||||
pngquant \
|
pngquant \
|
||||||
@@ -73,18 +72,6 @@ jobs:
|
|||||||
unpaper \
|
unpaper \
|
||||||
zlib1g
|
zlib1g
|
||||||
|
|
||||||
- name: Install Ubuntu 18.04 packages
|
|
||||||
if: matrix.os == 'ubuntu-18.04'
|
|
||||||
run: |
|
|
||||||
sudo apt-get install -y --no-install-recommends \
|
|
||||||
libexempi3
|
|
||||||
|
|
||||||
- name: Install Ubuntu 20.04 packages
|
|
||||||
if: matrix.os == 'ubuntu-20.04' || matrix.os == 'ubuntu-latest'
|
|
||||||
run: |
|
|
||||||
sudo apt-get install -y --no-install-recommends \
|
|
||||||
libexempi8
|
|
||||||
|
|
||||||
- name: Install Ubuntu packages for PyPy
|
- name: Install Ubuntu packages for PyPy
|
||||||
if: startsWith(matrix.python, 'pypy')
|
if: startsWith(matrix.python, 'pypy')
|
||||||
run: |
|
run: |
|
||||||
@@ -96,7 +83,7 @@ jobs:
|
|||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip wheel
|
python -m pip install --upgrade pip wheel
|
||||||
python -m pip install .[test]
|
python -m pip install --prefer-binary .[test]
|
||||||
|
|
||||||
- name: Report versions
|
- name: Report versions
|
||||||
run: |
|
run: |
|
||||||
@@ -122,23 +109,19 @@ jobs:
|
|||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
os: [macos-latest]
|
os: [macos-latest]
|
||||||
python: ["3.9", "3.10"]
|
python: ["3.10", "3.11"]
|
||||||
|
|
||||||
env:
|
env:
|
||||||
OS: ${{ matrix.os }}
|
OS: ${{ matrix.os }}
|
||||||
PYTHON: ${{ matrix.python }}
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v4
|
|
||||||
name: Install Python
|
|
||||||
with:
|
|
||||||
python-version: ${{ matrix.python }}
|
|
||||||
|
|
||||||
- name: Install Homebrew deps
|
- name: Install Homebrew deps
|
||||||
|
continue-on-error: true
|
||||||
run: |
|
run: |
|
||||||
brew update
|
brew update
|
||||||
brew install \
|
brew install \
|
||||||
@@ -149,10 +132,15 @@ jobs:
|
|||||||
pngquant \
|
pngquant \
|
||||||
tesseract
|
tesseract
|
||||||
|
|
||||||
|
- uses: actions/setup-python@v4
|
||||||
|
name: Install Python
|
||||||
|
with:
|
||||||
|
python-version: ${{ matrix.python }}
|
||||||
|
|
||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip wheel
|
python -m pip install --upgrade pip wheel
|
||||||
python -m pip install .[test]
|
python -m pip install --prefer-binary .[test]
|
||||||
|
|
||||||
- name: Report versions
|
- name: Report versions
|
||||||
run: |
|
run: |
|
||||||
@@ -177,14 +165,14 @@ jobs:
|
|||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
os: [windows-latest]
|
os: [windows-latest]
|
||||||
python: ["3.9", "3.10"]
|
python: ["3.10", "3.11"]
|
||||||
|
|
||||||
env:
|
env:
|
||||||
OS: ${{ matrix.os }}
|
OS: ${{ matrix.os }}
|
||||||
PYTHON: ${{ matrix.python }}
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
@@ -201,7 +189,7 @@ jobs:
|
|||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip wheel
|
python -m pip install --upgrade pip wheel
|
||||||
python -m pip install .[test]
|
python -m pip install --prefer-binary .[test]
|
||||||
|
|
||||||
- name: Test
|
- name: Test
|
||||||
run: |
|
run: |
|
||||||
@@ -217,20 +205,19 @@ jobs:
|
|||||||
name: Build sdist and wheels
|
name: Build sdist and wheels
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v4
|
- uses: actions/setup-python@v4
|
||||||
name: Install Python
|
name: Install Python
|
||||||
with:
|
with:
|
||||||
python-version: "3.7"
|
python-version: "3.9"
|
||||||
|
|
||||||
- name: Make wheels and sdist
|
- name: Make wheels and sdist
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip wheel
|
python -m pip install --upgrade pip wheel build
|
||||||
python setup.py sdist
|
python -m build --sdist --wheel
|
||||||
python setup.py bdist_wheel
|
|
||||||
|
|
||||||
- uses: actions/upload-artifact@v3
|
- uses: actions/upload-artifact@v3
|
||||||
with:
|
with:
|
||||||
@@ -242,6 +229,9 @@ jobs:
|
|||||||
name: Deploy artifacts to PyPI
|
name: Deploy artifacts to PyPI
|
||||||
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
|
environment: release
|
||||||
|
permissions:
|
||||||
|
id-token: write # mandatory for PyPI publishing
|
||||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/download-artifact@v3
|
- uses: actions/download-artifact@v3
|
||||||
@@ -249,11 +239,34 @@ jobs:
|
|||||||
name: artifact
|
name: artifact
|
||||||
path: dist
|
path: dist
|
||||||
|
|
||||||
- uses: pypa/gh-action-pypi-publish@master
|
- name: Publish to PyPI
|
||||||
|
uses: pypa/gh-action-pypi-publish@release/v1
|
||||||
|
|
||||||
|
create_release:
|
||||||
|
name: Create GitHub release
|
||||||
|
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||||
|
permissions:
|
||||||
|
# Required to create a release
|
||||||
|
contents: write
|
||||||
|
steps:
|
||||||
|
- uses: actions/download-artifact@v3
|
||||||
with:
|
with:
|
||||||
user: __token__
|
name: artifact
|
||||||
password: ${{ secrets.TOKEN_PYPI }}
|
path: dist
|
||||||
# repository_url: https://test.pypi.org/legacy/
|
|
||||||
|
- name: Create Release
|
||||||
|
id: create-release
|
||||||
|
uses: shogo82148/actions-create-release@v1
|
||||||
|
|
||||||
|
- name: Upload Assets
|
||||||
|
uses: shogo82148/actions-upload-release-asset@v1
|
||||||
|
with:
|
||||||
|
upload_url: ${{ steps.create-release.outputs.upload_url }}
|
||||||
|
asset_path: |
|
||||||
|
./dist/*.whl
|
||||||
|
./dist/*.tar.gz
|
||||||
|
|
||||||
docker:
|
docker:
|
||||||
name: Build Docker images
|
name: Build Docker images
|
||||||
@@ -264,9 +277,9 @@ jobs:
|
|||||||
- name: Set image tag to release or branch
|
- name: Set image tag to release or branch
|
||||||
run: echo "DOCKER_IMAGE_TAG=${GITHUB_REF##*/}" >> $GITHUB_ENV
|
run: echo "DOCKER_IMAGE_TAG=${GITHUB_REF##*/}" >> $GITHUB_ENV
|
||||||
|
|
||||||
- name: If master, set to latest
|
- name: If main, set to latest
|
||||||
run: echo 'DOCKER_IMAGE_TAG=latest' >> $GITHUB_ENV
|
run: echo 'DOCKER_IMAGE_TAG=latest' >> $GITHUB_ENV
|
||||||
if: env.DOCKER_IMAGE_TAG == 'master'
|
if: env.DOCKER_IMAGE_TAG == 'main'
|
||||||
|
|
||||||
- name: Set Docker Hub repository to username
|
- name: Set Docker Hub repository to username
|
||||||
run: echo "DOCKER_REPOSITORY=jbarlow83" >> $GITHUB_ENV
|
run: echo "DOCKER_REPOSITORY=jbarlow83" >> $GITHUB_ENV
|
||||||
@@ -274,22 +287,22 @@ jobs:
|
|||||||
- name: Set image name
|
- name: Set image name
|
||||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
||||||
|
|
||||||
- uses: actions/checkout@v3
|
- uses: actions/checkout@v4
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- name: Login to Docker Hub
|
- name: Login to Docker Hub
|
||||||
uses: docker/login-action@v2
|
uses: docker/login-action@v3
|
||||||
with:
|
with:
|
||||||
username: jbarlow83
|
username: jbarlow83
|
||||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||||
|
|
||||||
- name: Set up QEMU
|
- name: Set up QEMU
|
||||||
uses: docker/setup-qemu-action@v2
|
uses: docker/setup-qemu-action@v3
|
||||||
|
|
||||||
- name: Set up Docker Buildx
|
- name: Set up Docker Buildx
|
||||||
id: buildx
|
id: buildx
|
||||||
uses: docker/setup-buildx-action@v2
|
uses: docker/setup-buildx-action@v3
|
||||||
|
|
||||||
- name: Print image tag
|
- name: Print image tag
|
||||||
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
||||||
|
|||||||
+12
-11
@@ -1,14 +1,15 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: CC-BY-SA-4.0
|
||||||
|
|
||||||
# dotfiles
|
# dotfiles
|
||||||
.*
|
.coverage
|
||||||
!.coveragerc
|
.venv*/
|
||||||
!.dockerignore
|
.tox/
|
||||||
!.git_archival.txt
|
.vscode/
|
||||||
!.gitattributes
|
.hypothesis/
|
||||||
!.gitignore
|
.ipynb_checkpoints/
|
||||||
!.pre-commit-config.yaml
|
.mypy_cache/
|
||||||
!.readthedocs.yaml
|
.pytest_cache/
|
||||||
!.github/
|
|
||||||
!.docker/
|
|
||||||
|
|
||||||
# Dev scratch
|
# Dev scratch
|
||||||
*.ipynb
|
*.ipynb
|
||||||
@@ -26,6 +27,7 @@ venv*/
|
|||||||
*.traineddata
|
*.traineddata
|
||||||
/private
|
/private
|
||||||
/coverage.xml
|
/coverage.xml
|
||||||
|
/issuepdf
|
||||||
|
|
||||||
# Package building
|
# Package building
|
||||||
*.egg-info/
|
*.egg-info/
|
||||||
@@ -42,4 +44,3 @@ docs/_build/
|
|||||||
docs/_static/
|
docs/_static/
|
||||||
docs/_templates/
|
docs/_templates/
|
||||||
docs/Makefile
|
docs/Makefile
|
||||||
ocrmypdf/lib/_*.py
|
|
||||||
|
|||||||
+11
-16
@@ -1,33 +1,28 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
repos:
|
repos:
|
||||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||||
rev: v4.3.0
|
rev: v4.4.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: check-case-conflict
|
- id: check-case-conflict
|
||||||
- id: check-merge-conflict
|
- id: check-merge-conflict
|
||||||
- id: check-toml
|
- id: check-toml
|
||||||
- id: check-yaml
|
- id: check-yaml
|
||||||
- id: debug-statements
|
- id: debug-statements
|
||||||
- repo: https://github.com/pycqa/isort
|
- repo: https://github.com/charliermarsh/ruff-pre-commit
|
||||||
rev: 5.10.1
|
rev: "v0.0.261"
|
||||||
hooks:
|
hooks:
|
||||||
- id: isort
|
- id: ruff
|
||||||
args: ["--profile", "black", "-a", "from __future__ import annotations"]
|
files: "src/.*\\.pyi?$"
|
||||||
|
args: [--fix, --exit-non-zero-on-fix]
|
||||||
- repo: https://github.com/psf/black
|
- repo: https://github.com/psf/black
|
||||||
rev: 22.6.0
|
rev: 23.3.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: black
|
- id: black
|
||||||
language_version: python
|
language_version: python
|
||||||
- repo: https://github.com/asottile/setup-cfg-fmt
|
|
||||||
rev: v1.20.2
|
|
||||||
hooks:
|
|
||||||
- id: setup-cfg-fmt
|
|
||||||
- repo: https://github.com/asottile/pyupgrade
|
|
||||||
rev: v2.37.2
|
|
||||||
hooks:
|
|
||||||
- id: pyupgrade
|
|
||||||
args: ["--py37-plus"]
|
|
||||||
- repo: https://github.com/pre-commit/mirrors-mypy
|
- repo: https://github.com/pre-commit/mirrors-mypy
|
||||||
rev: v0.971
|
rev: v1.2.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: mypy
|
- id: mypy
|
||||||
additional_dependencies:
|
additional_dependencies:
|
||||||
|
|||||||
+8
-1
@@ -1,3 +1,6 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
# Read the Docs configuration file
|
# Read the Docs configuration file
|
||||||
# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details
|
# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details
|
||||||
|
|
||||||
@@ -13,8 +16,12 @@ formats:
|
|||||||
- pdf
|
- pdf
|
||||||
|
|
||||||
# Optionally set the version of Python and requirements required to build your docs
|
# Optionally set the version of Python and requirements required to build your docs
|
||||||
|
build:
|
||||||
|
os: ubuntu-22.04
|
||||||
|
tools:
|
||||||
|
python: "3.9"
|
||||||
|
|
||||||
python:
|
python:
|
||||||
version: "3.7"
|
|
||||||
install:
|
install:
|
||||||
- method: pip
|
- method: pip
|
||||||
path: .
|
path: .
|
||||||
|
|||||||
+133
@@ -0,0 +1,133 @@
|
|||||||
|
Format: https://www.debian.org/doc/packaging-manuals/copyright-format/1.0/
|
||||||
|
Upstream-Name: OCRmyPDF
|
||||||
|
Upstream-Contact: James R. Barlow <james@purplerock.ca>
|
||||||
|
Source: https://github.com/ocrmypdf/OCRmyPDF
|
||||||
|
|
||||||
|
|
||||||
|
Files:
|
||||||
|
.git_archival.txt
|
||||||
|
docs/images/logo-social.png
|
||||||
|
docs/images/logo-square-256.svg
|
||||||
|
docs/images/logo-square.png
|
||||||
|
docs/images/logo-square.svg
|
||||||
|
docs/images/logo.svg
|
||||||
|
setup.cfg
|
||||||
|
Copyright: (C) 2022 James R. Barlow
|
||||||
|
License: MPL-2.0
|
||||||
|
|
||||||
|
Files:
|
||||||
|
.github/ISSUE_TEMPLATE/*.md
|
||||||
|
docs/images/macos-workflow.png
|
||||||
|
Copyright: (C) 2022 James R. Barlow
|
||||||
|
License: CC-BY-SA-4.0
|
||||||
|
|
||||||
|
Files:
|
||||||
|
tests/resources/acroform.pdf
|
||||||
|
tests/resources/aspect.pdf
|
||||||
|
tests/resources/blank.pdf
|
||||||
|
tests/resources/cmyk.pdf
|
||||||
|
tests/resources/crom.png
|
||||||
|
tests/resources/enormous.pdf
|
||||||
|
tests/resources/formxobject.pdf
|
||||||
|
tests/resources/francais.pdf
|
||||||
|
tests/resources/hugemono.pdf
|
||||||
|
tests/resources/invalid.pdf
|
||||||
|
tests/resources/kcs.pdf
|
||||||
|
tests/resources/livecycle.pdf
|
||||||
|
tests/resources/missing_docinfo.pdf
|
||||||
|
tests/resources/negzero.pdf
|
||||||
|
tests/resources/no_contents.pdf
|
||||||
|
tests/resources/toc.pdf
|
||||||
|
tests/resources/trivial.pdf
|
||||||
|
tests/resources/truetype_font_nomapping.pdf
|
||||||
|
tests/resources/type3_font_nomapping.pdf
|
||||||
|
misc/screencast/*
|
||||||
|
Copyright: (C) 2022 James R. Barlow
|
||||||
|
License: CC-BY-SA-4.0
|
||||||
|
|
||||||
|
Files:
|
||||||
|
tests/resources/graph.pdf
|
||||||
|
tests/resources/graph_ocred.pdf
|
||||||
|
Copyright: (C) 2012 SmokeyJoe
|
||||||
|
License: GFDL-1.2-or-later or CC-BY-SA-3.0
|
||||||
|
|
||||||
|
Files: tests/resources/c02-22.pdf
|
||||||
|
tests/resources/congress.jpg
|
||||||
|
tests/resources/multipage.pdf
|
||||||
|
Copyright: Public domain
|
||||||
|
License: public-domain
|
||||||
|
Copyright on these files has expired.
|
||||||
|
|
||||||
|
Files: docs/images/bitmap_vs_svg.svg
|
||||||
|
Copyright: (C) 2006 Yug
|
||||||
|
License: CC-BY-SA-2.5
|
||||||
|
|
||||||
|
Files: tests/cache/*
|
||||||
|
Copyright: (C) 2022 James R. Barlow
|
||||||
|
License: CC-BY-SA-4.0
|
||||||
|
|
||||||
|
Files: tests/resources/linn.png
|
||||||
|
tests/resources/linn.pdf
|
||||||
|
tests/resources/linn.txt
|
||||||
|
tests/resources/ccitt.pdf
|
||||||
|
tests/resources/cardinal.pdf
|
||||||
|
tests/resources/jbig2.pdf
|
||||||
|
tests/resources/skew.pdf
|
||||||
|
tests/resources/rotated_skew.pdf
|
||||||
|
tests/resources/poster.pdf
|
||||||
|
Copyright: (C) 1985 Forat Electronics
|
||||||
|
License: GFDL-1.2-or-later or CC-BY-SA-3.0
|
||||||
|
|
||||||
|
Files: tests/resources/lichtenstein.pdf
|
||||||
|
Copyright: (C) 2001 Andreas Tille
|
||||||
|
(C) 2007 Alessio Damato
|
||||||
|
License: GFDL-1.2-or-later or CC-BY-SA-3.0
|
||||||
|
|
||||||
|
Files: tests/resources/masks.pdf
|
||||||
|
Copyright: held by the contributors to the German Wikipedia article "Linux"
|
||||||
|
see: https://de.wikipedia.org/w/index.php?title=Linux&action=history
|
||||||
|
(masks.pdf generated from Wikipedia article as of 2016-08-24)
|
||||||
|
License: CC-BY-SA-3.0
|
||||||
|
|
||||||
|
Files: tests/resources/epson.pdf
|
||||||
|
Copyright: held by the contributors to the Wikipedia article "Optical character recognition"
|
||||||
|
see: https://en.wikipedia.org/w/index.php?title=Optical_character_recognition&action=history
|
||||||
|
(epson.pdf generated from Wikipedia article as of 2016-09-14)
|
||||||
|
License: CC-BY-SA-3.0
|
||||||
|
|
||||||
|
Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf
|
||||||
|
Copyright: (C) 2005 Ellywa
|
||||||
|
License: GFDL-1.2-or-later or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0
|
||||||
|
Comment:
|
||||||
|
Obtained from: https://commons.wikimedia.org/wiki/File:Triumph.typewriter_text_Linzensoep.gif
|
||||||
|
|
||||||
|
Files: tests/resources/overlay.pdf
|
||||||
|
Copyright: (C) 2017 Max Anderson
|
||||||
|
License: MIT
|
||||||
|
|
||||||
|
Files:
|
||||||
|
tests/resources/baiona*.png
|
||||||
|
tests/resources/baiona*.jpg
|
||||||
|
tests/resources/link.pdf
|
||||||
|
tests/resources/palette.pdf
|
||||||
|
Copyright: (C) 2014 Euskaldunaa
|
||||||
|
License: CC-BY-SA-4.0
|
||||||
|
|
||||||
|
Files: tests/resources/vector.pdf
|
||||||
|
Copyright: (C) 2018 Catscratch
|
||||||
|
License: MIT
|
||||||
|
|
||||||
|
Files: src/ocrmypdf/data/sRGB.icc
|
||||||
|
Copyright: Kai-Uwe Behrmann <www.behrmann.name>
|
||||||
|
Marti Maria <www.littlecms.com>
|
||||||
|
Photogamut <www.photogamut.org>
|
||||||
|
Graeme Gill <www.argyllcms.com>
|
||||||
|
ColorSolutions <www.basICColor.com>
|
||||||
|
License: Zlib
|
||||||
|
|
||||||
|
Files: tests/resources/3small.pdf
|
||||||
|
Copyright: (C) 2014 Euskaldunaa
|
||||||
|
(C) 2017 James R. Barlow
|
||||||
|
(C) 2005 Ellywa
|
||||||
|
License: CC-BY-SA-4.0 and (GFDL-1.2-or-later or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0)
|
||||||
|
Comment: concatenation of baiona_gray.png, crom.png and typewriter.png/2400dpi.pdf
|
||||||
@@ -1,3 +1,7 @@
|
|||||||
|
<!-- SPDX-FileCopyrightText: 2014 Julien Pfefferkorn -->
|
||||||
|
<!-- SPDX-FileCopyrightText: 2015 James R. Barlow -->
|
||||||
|
<!-- SPDX-License-Identifier: CC-BY-SA-4.0 -->
|
||||||
|
|
||||||
<img src="docs/images/logo.svg" width="240" alt="OCRmyPDF">
|
<img src="docs/images/logo.svg" width="240" alt="OCRmyPDF">
|
||||||
|
|
||||||
[](https://github.com/ocrmypdf/OCRmyPDF/actions/workflows/build.yml) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions]
|
[](https://github.com/ocrmypdf/OCRmyPDF/actions/workflows/build.yml) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions]
|
||||||
@@ -38,6 +42,8 @@ ocrmypdf # it's a scriptable command line program
|
|||||||
- Scales properly to handle files with thousands of pages
|
- Scales properly to handle files with thousands of pages
|
||||||
- Battle-tested on millions of PDFs
|
- Battle-tested on millions of PDFs
|
||||||
|
|
||||||
|
<img src="misc/screencast/demo.svg" alt="Demo of OCRmyPDF in a terminal session">
|
||||||
|
|
||||||
For details: please consult the [documentation](https://ocrmypdf.readthedocs.io/en/latest/).
|
For details: please consult the [documentation](https://ocrmypdf.readthedocs.io/en/latest/).
|
||||||
|
|
||||||
## Motivation
|
## Motivation
|
||||||
@@ -64,9 +70,9 @@ Linux, Windows, macOS and FreeBSD are supported. Docker images are also availabl
|
|||||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||||
| Fedora | ``dnf install ocrmypdf`` |
|
| Fedora | ``dnf install ocrmypdf`` |
|
||||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
| FreeBSD | ``pkg install py37-ocrmypdf`` |
|
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||||
| Conda | ``conda install ocrmypdf`` |
|
| Conda | ``conda install ocrmypdf`` |
|
||||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||||
|
|
||||||
@@ -92,10 +98,7 @@ brew install tesseract-lang
|
|||||||
|
|
||||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested.
|
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested.
|
||||||
|
|
||||||
OCRmyPDF supports Tesseract 4.0 and the beta versions of Tesseract 5.0. It will
|
OCRmyPDF supports Tesseract 4.1.1+. It will automatically use whichever version it finds first on the `PATH` environment variable. On Windows, if `PATH` does not provide a Tesseract binary, we use the highest version number that is installed according to the Windows Registry.
|
||||||
automatically use whichever version it finds first on the `PATH` environment
|
|
||||||
variable. On Windows, if `PATH` does not provide a Tesseract binary, we use
|
|
||||||
the highest version number that is installed according to the Windows Registry.
|
|
||||||
|
|
||||||
## Documentation and support
|
## Documentation and support
|
||||||
|
|
||||||
@@ -111,7 +114,7 @@ Please report issues on our [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF
|
|||||||
|
|
||||||
## Requirements
|
## Requirements
|
||||||
|
|
||||||
In addition to the required Python version (3.7+), OCRmyPDF requires external program installations of Ghostscript and Tesseract OCR. OCRmyPDF is pure Python, and runs on pretty much everything: Linux, macOS, Windows and FreeBSD.
|
In addition to the required Python version (3.8+), OCRmyPDF requires external program installations of Ghostscript and Tesseract OCR. OCRmyPDF is pure Python, and runs on pretty much everything: Linux, macOS, Windows and FreeBSD.
|
||||||
|
|
||||||
## Press & Media
|
## Press & Media
|
||||||
|
|
||||||
@@ -122,6 +125,7 @@ In addition to the required Python version (3.7+), OCRmyPDF requires external pr
|
|||||||
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||||
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||||
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||||
|
- [Y Combinator discussion](https://news.ycombinator.com/item?id=32028752)
|
||||||
|
|
||||||
## Business enquiries
|
## Business enquiries
|
||||||
|
|
||||||
|
|||||||
+81
-16
@@ -10,7 +10,7 @@ Control of unpaper
|
|||||||
|
|
||||||
OCRmyPDF uses ``unpaper`` to provide the implementation of the
|
OCRmyPDF uses ``unpaper`` to provide the implementation of the
|
||||||
``--clean`` and ``--clean-final`` arguments.
|
``--clean`` and ``--clean-final`` arguments.
|
||||||
`unpaper <https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md>`__
|
`unpaper <https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md>`__
|
||||||
provides a variety of image processing filters to improve images.
|
provides a variety of image processing filters to improve images.
|
||||||
|
|
||||||
By default, OCRmyPDF uses only ``unpaper`` arguments that were found to
|
By default, OCRmyPDF uses only ``unpaper`` arguments that were found to
|
||||||
@@ -47,8 +47,8 @@ and clean up the margins of both.
|
|||||||
|
|
||||||
Some ``unpaper`` features cause multiple input or output files to be
|
Some ``unpaper`` features cause multiple input or output files to be
|
||||||
consumed or produced. OCRmyPDF requires ``unpaper`` to consume one
|
consumed or produced. OCRmyPDF requires ``unpaper`` to consume one
|
||||||
file and produce one file. An deviation from that condition will
|
file and produce one file; errors will result if this assumption is not
|
||||||
result in errors.
|
met.
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
|
|
||||||
@@ -82,14 +82,17 @@ is stripped out. Then an image of each page is created with visible text
|
|||||||
masked out. The page image is sent for OCR, and any additional text is
|
masked out. The page image is sent for OCR, and any additional text is
|
||||||
inserted as OCR. If a file contains a mix of text and bitmap images that
|
inserted as OCR. If a file contains a mix of text and bitmap images that
|
||||||
contain text, OCRmyPDF will locate the additional text in images without
|
contain text, OCRmyPDF will locate the additional text in images without
|
||||||
disrupting the existing text.
|
disrupting the existing text. Some PDF OCR solutions render text as
|
||||||
|
technically printable or visible in some way, perhaps by drawing it and
|
||||||
|
then painting over it. OCRmyPDF cannot distinguish this type of OCR
|
||||||
|
text from real text, so it will not be "redone".
|
||||||
|
|
||||||
If ``--force-ocr`` is issued, then all pages will be rasterized to
|
If ``--force-ocr`` is issued, then all pages will be rasterized to
|
||||||
images, discarding any hidden OCR text, and rasterizing any printable
|
images, discarding any hidden OCR text, rasterizing any printable
|
||||||
text. This is useful for redoing OCR, for fixing OCR text with a damaged
|
text, and flattening form fields or interactive objects into their visual
|
||||||
character map (text is selectable but not searchable), and destroying
|
representation. This is useful for redoing OCR, for fixing OCR text
|
||||||
redacted information. Any forms and vector graphics will be rasterized
|
with a damaged character map (text is selectable but not searchable),
|
||||||
as well.
|
and destroying redacted information.
|
||||||
|
|
||||||
Time and image size limits
|
Time and image size limits
|
||||||
--------------------------
|
--------------------------
|
||||||
@@ -104,13 +107,48 @@ was requested, the preprocessed image layer will be inserted.
|
|||||||
If you want to adjust the amount of time spent on OCR, change
|
If you want to adjust the amount of time spent on OCR, change
|
||||||
``--tesseract-timeout``. You can also automatically skip images that
|
``--tesseract-timeout``. You can also automatically skip images that
|
||||||
exceed a certain number of megapixels with ``--skip-big``. (A 300 DPI,
|
exceed a certain number of megapixels with ``--skip-big``. (A 300 DPI,
|
||||||
8.5×11" page is 8.4 megapixels.)
|
8.5×11" page image is 8.4 megapixels.)
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
# Allow 300 seconds for OCR; skip any page larger than 50 megapixels
|
# Allow 300 seconds for OCR; skip any page larger than 50 megapixels
|
||||||
ocrmypdf --tesseract-timeout 300 --skip-big 50 bigfile.pdf output.pdf
|
ocrmypdf --tesseract-timeout 300 --skip-big 50 bigfile.pdf output.pdf
|
||||||
|
|
||||||
|
OCR for huge images
|
||||||
|
-------------------
|
||||||
|
|
||||||
|
Separate from these settings, Tesseract has internal limits on the size
|
||||||
|
of images it will process. If you issue
|
||||||
|
``--tesseract-downsample-large-images``, OCRmyPDF will downsample images
|
||||||
|
to fit Tesseract limits. (The limits are usually entered only for scanned
|
||||||
|
images of oversized media, such as large maps or blueprints exceeding
|
||||||
|
110 cm or 43 inches in either dimension, and at high DPI.)
|
||||||
|
|
||||||
|
``--tesseract-downsample-above`` adjusts the threshold at which images
|
||||||
|
will be downsampled. By default, only images that exceed any of Tesseract's
|
||||||
|
internal limits are downsampled.
|
||||||
|
|
||||||
|
You will also need to set ``--tesseract-timeout`` high enough to allow
|
||||||
|
for processing.
|
||||||
|
|
||||||
|
Only the image sent for OCR is downsampled. The original image is
|
||||||
|
preserved.
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
# Allow 600 seconds for OCR on huge images
|
||||||
|
ocrmypdf --tesseract-timeout 600 \
|
||||||
|
--tesseract-downsample-large-images \
|
||||||
|
bigfile.pdf output.pdf
|
||||||
|
|
||||||
|
# Downsample images above 5000 pixels on the longest dimension to
|
||||||
|
# 5000 pixels
|
||||||
|
ocrmypdf --tesseract-timeout 120 \
|
||||||
|
--tesseract-downsample-large-images \
|
||||||
|
--tesseract-downsample-above 5000 \
|
||||||
|
bigfile.pdf output_downsampled_ocr.pdf
|
||||||
|
|
||||||
|
|
||||||
Overriding default tesseract
|
Overriding default tesseract
|
||||||
----------------------------
|
----------------------------
|
||||||
|
|
||||||
@@ -154,13 +192,14 @@ In addition to tesseract, OCRmyPDF uses the following external binaries:
|
|||||||
- ``jbig2``
|
- ``jbig2``
|
||||||
|
|
||||||
In each case OCRmyPDF will search the ``PATH`` environment variable to
|
In each case OCRmyPDF will search the ``PATH`` environment variable to
|
||||||
locate the binaries.
|
locate the binaries. By modifying the ``PATH`` environment variable, you
|
||||||
|
can override the binaries that OCRmyPDF uses.
|
||||||
|
|
||||||
Changing tesseract configuration variables
|
Changing tesseract configuration variables
|
||||||
------------------------------------------
|
------------------------------------------
|
||||||
|
|
||||||
You can override tesseract's default `control
|
You can override tesseract's default `control
|
||||||
parameters <https://github.com/tesseract-ocr/tesseract/wiki/ControlParams>`__
|
parameters <https://tesseract-ocr.github.io/tessdoc/tess3/ControlParams.html>`__
|
||||||
with a configuration file.
|
with a configuration file.
|
||||||
|
|
||||||
As an example, this configuration will disable Tesseract's dictionary
|
As an example, this configuration will disable Tesseract's dictionary
|
||||||
@@ -241,6 +280,17 @@ PDF.js viewer.
|
|||||||
|
|
||||||
This works in all versions of Tesseract.
|
This works in all versions of Tesseract.
|
||||||
|
|
||||||
|
Rendering and rasterizing options
|
||||||
|
=================================
|
||||||
|
|
||||||
|
.. versionadded:: 14.3.0
|
||||||
|
|
||||||
|
The ``--continue-on-soft-render-error`` option allows OCRmyPDF to
|
||||||
|
proceed if a page cannot be rasterized rendered. This is useful if you are
|
||||||
|
trying to get the best possible OCR from a PDF that is not well-formed,
|
||||||
|
and you are willing to accept some pages that may not visually match the
|
||||||
|
input, and that may not OCR well.
|
||||||
|
|
||||||
Return code policy
|
Return code policy
|
||||||
==================
|
==================
|
||||||
|
|
||||||
@@ -299,16 +349,31 @@ stable user interface. They may be imported from
|
|||||||
- The program was interrupted by pressing Ctrl+C.
|
- The program was interrupted by pressing Ctrl+C.
|
||||||
|
|
||||||
|
|
||||||
|
.. _tmpdir:
|
||||||
|
|
||||||
|
Changing temporary storage location
|
||||||
|
===================================
|
||||||
|
|
||||||
|
OCRmyPDF generates many temporary files during processing.
|
||||||
|
|
||||||
|
To change where temporary files are stored, change the ``TMPDIR``
|
||||||
|
environment variable for ocrmypdf's environment. (Python's
|
||||||
|
``tempfile.gettempdir()`` returns the root directory in which temporary
|
||||||
|
files will be stored.) For example, one could redirect ``TMPDIR`` to a
|
||||||
|
large RAM disk to avoid wear on HDD/SSD and potentially improve
|
||||||
|
performance.
|
||||||
|
|
||||||
|
On Windows, the ``TEMP`` environment variable is used instead.
|
||||||
|
|
||||||
Debugging the intermediate files
|
Debugging the intermediate files
|
||||||
================================
|
================================
|
||||||
|
|
||||||
OCRmyPDF normally saves its intermediate results to a temporary folder
|
OCRmyPDF normally saves its intermediate results to a temporary folder
|
||||||
and deletes this folder when it exits, whether it succeeded or failed.
|
and deletes this folder when it exits, whether it succeeded or failed.
|
||||||
|
|
||||||
If the ``-k`` argument is issued on the command line, OCRmyPDF will keep
|
If the ``--keep-temporary-files`` (``-k```) argument is issued on the
|
||||||
the temporary folder and print the location, whether it succeeded or
|
command line, OCRmyPDF will keep the temporary folder and print the location,
|
||||||
failed (provided the Python interpreter did not crash). An example
|
whether it succeeded or failed. An example message is:
|
||||||
message is:
|
|
||||||
|
|
||||||
.. code-block:: none
|
.. code-block:: none
|
||||||
|
|
||||||
|
|||||||
@@ -72,14 +72,6 @@ OCRmyPDF, use processes.
|
|||||||
not take at least one of these steps, process semantics will prevent
|
not take at least one of these steps, process semantics will prevent
|
||||||
OCRmyPDF from working correctly.
|
OCRmyPDF from working correctly.
|
||||||
|
|
||||||
.. warning::
|
|
||||||
|
|
||||||
On macOS with Python 3.7, you must call
|
|
||||||
:func:`multiprocessing.set_start_method("spawn")`. Without this, multiprocessing
|
|
||||||
will be unstable. From the command line, OCRmyPDF does this automatically,
|
|
||||||
but as an API user you must do this. See Python bpo-33725 for details.
|
|
||||||
Python 3.8+ also resolve this automatically.
|
|
||||||
|
|
||||||
Logging
|
Logging
|
||||||
-------
|
-------
|
||||||
|
|
||||||
|
|||||||
+30
-31
@@ -21,8 +21,8 @@ processors. To maximize parallelism without overloading your system with
|
|||||||
processes, consider using ``parallel -j 2`` to limit parallel to running
|
processes, consider using ``parallel -j 2`` to limit parallel to running
|
||||||
two jobs at once.
|
two jobs at once.
|
||||||
|
|
||||||
This command will run all ocrmypdf all files named ``*.pdf`` in the
|
This command will run ``ocrmypdf`` on all files named ``*.pdf`` in the
|
||||||
current directory and write them to the previous created ``output/``
|
current directory and write them to the previously created ``output/``
|
||||||
folder. It will not search subdirectories.
|
folder. It will not search subdirectories.
|
||||||
|
|
||||||
The ``--tag`` argument tells parallel to print the filename as a prefix
|
The ``--tag`` argument tells parallel to print the filename as a prefix
|
||||||
@@ -46,16 +46,6 @@ place, and printing each filename in between runs:
|
|||||||
|
|
||||||
find . -printf '%p\n' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
|
find . -printf '%p\n' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
|
||||||
|
|
||||||
Alternatively, with a Docker container and streaming the file through
|
|
||||||
standard input and output:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
find . -name '*.pdf' -print0 | xargs -0 | while read pdf; do
|
|
||||||
pdfout=$(mktemp)
|
|
||||||
docker run --rm -i jbarlow83/ocrmypdf - - <$pdf >$pdfout && cp $pdfout $pdf
|
|
||||||
done
|
|
||||||
|
|
||||||
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
||||||
``find`` to create a directory list and ``parallel`` to parallelize runs
|
``find`` to create a directory list and ``parallel`` to parallelize runs
|
||||||
of ``ocrmypdf``, again updating files in place.
|
of ``ocrmypdf``, again updating files in place.
|
||||||
@@ -70,6 +60,15 @@ In a Windows batch file, use
|
|||||||
|
|
||||||
for /r %%f in (*.pdf) do ocrmypdf %%f %%f
|
for /r %%f in (*.pdf) do ocrmypdf %%f %%f
|
||||||
|
|
||||||
|
With a Docker container, you will need to stream through standard input and output:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
find . -name '*.pdf' -print0 | xargs -0 | while read pdf; do
|
||||||
|
pdfout=$(mktemp)
|
||||||
|
docker run --rm -i jbarlow83/ocrmypdf - - <$pdf >$pdfout && cp $pdfout $pdf
|
||||||
|
done
|
||||||
|
|
||||||
Sample script
|
Sample script
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
@@ -88,9 +87,9 @@ package <https://www.synology.com/en-global/dsm/packages/Docker>`__ is
|
|||||||
installed. Attached is a script to address particular quirks of using
|
installed. Attached is a script to address particular quirks of using
|
||||||
OCRmyPDF on one of these devices.
|
OCRmyPDF on one of these devices.
|
||||||
|
|
||||||
This is only possible for x86-based Synology products. Some Synology
|
At the time this script was written, it only worked for x86-based Synology
|
||||||
products use ARM or Power processors and do not support Docker. Further
|
products. It is not known if it will work on ARM-based Synology products.
|
||||||
adjustments might be needed to deal with the Synology's relatively
|
Further adjustments might be needed to deal with the Synology's relatively
|
||||||
limited CPU and RAM.
|
limited CPU and RAM.
|
||||||
|
|
||||||
.. literalinclude:: ../misc/synology.py
|
.. literalinclude:: ../misc/synology.py
|
||||||
@@ -133,7 +132,7 @@ Users may need to customize the script to meet their requirements.
|
|||||||
"OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)"
|
"OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)"
|
||||||
"OCR_ARCHIVE_DIRECTORY", "Set archive directory for processed originals (should not be under input, requires ``OCR_ON_SUCCESS_ARCHIVE`` to be set)"
|
"OCR_ARCHIVE_DIRECTORY", "Set archive directory for processed originals (should not be under input, requires ``OCR_ON_SUCCESS_ARCHIVE`` to be set)"
|
||||||
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
|
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
|
||||||
"OCR_ON_SUCCESS_ARCHIVE", "This will move the processed orignal file to ``OCR_ARCHIVE_DIRECTORY`` if the exit code is 0 (OK). Note that ``OCR_ON_SUCCESS_DELETE`` takes precedence over this option, i.e. if both options are set, the input file will be deleted."
|
"OCR_ON_SUCCESS_ARCHIVE", "This will move the processed original file to ``OCR_ARCHIVE_DIRECTORY`` if the exit code is 0 (OK). Note that ``OCR_ON_SUCCESS_DELETE`` takes precedence over this option, i.e. if both options are set, the input file will be deleted."
|
||||||
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||||
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
||||||
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
|
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
|
||||||
@@ -151,14 +150,14 @@ The watcher service is included in the OCRmyPDF Docker image. To run it:
|
|||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
docker run \
|
docker run \
|
||||||
-v <path to files to convert>:/input \
|
--volume <path to files to convert>:/input \
|
||||||
-v <path to store results>:/output \
|
--volume <path to store results>:/output \
|
||||||
-v <path to store processed originals>:/archive \
|
--volume <path to store processed originals>:/archive \
|
||||||
-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
||||||
-e OCR_ON_SUCCESS_ARCHIVE=1 \
|
--env OCR_ON_SUCCESS_ARCHIVE=1 \
|
||||||
-e OCR_DESKEW=1 \
|
--env OCR_DESKEW=1 \
|
||||||
-e PYTHONUNBUFFERED=1 \
|
--env PYTHONUNBUFFERED=1 \
|
||||||
-it --entrypoint python3 \
|
--interactive --tty --entrypoint python3 \
|
||||||
jbarlow83/ocrmypdf \
|
jbarlow83/ocrmypdf \
|
||||||
watcher.py
|
watcher.py
|
||||||
|
|
||||||
@@ -170,13 +169,13 @@ original to ``/archive``. The parameters to this image are:
|
|||||||
:header: "Parameter", "Description"
|
:header: "Parameter", "Description"
|
||||||
:widths: 50, 50
|
:widths: 50, 50
|
||||||
|
|
||||||
"``-v <path to files to convert>:/input``", "Files placed in this location will be OCRed"
|
"``--volume <path to files to convert>:/input``", "Files placed in this location will be OCRed"
|
||||||
"``-v <path to store results>:/output``", "This is where OCRed files will be stored"
|
"``--volume <path to store results>:/output``", "This is where OCRed files will be stored"
|
||||||
"``-v <path to store processed originals>:/archive``", "Archive processed originals here"
|
"``--volume <path to store processed originals>:/archive``", "Archive processed originals here"
|
||||||
"``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``"
|
"``--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||||
"``-e OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals"
|
"``--env OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals"
|
||||||
"``-e OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs"
|
"``--env OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs"
|
||||||
"``-e PYTHONBUFFERED=1``", "This will force ``STDOUT`` to be unbuffered and allow you to see messages in docker logs"
|
"``--env PYTHONBUFFERED=1``", "This will force ``STDOUT`` to be unbuffered and allow you to see messages in docker logs"
|
||||||
|
|
||||||
This service relies on polling to check for changes to the filesystem. It
|
This service relies on polling to check for changes to the filesystem. It
|
||||||
may not be suitable for some environments, such as filesystems shared on a
|
may not be suitable for some environments, such as filesystems shared on a
|
||||||
|
|||||||
@@ -0,0 +1,87 @@
|
|||||||
|
.. _ocr-service:
|
||||||
|
|
||||||
|
==================
|
||||||
|
Online deployments
|
||||||
|
==================
|
||||||
|
|
||||||
|
OCRmyPDF is designed to be used as a command line tool, but it can be
|
||||||
|
used in a web service. This document describes some considerations for
|
||||||
|
doing so.
|
||||||
|
|
||||||
|
A basic web service implementation is provided in the source code
|
||||||
|
repository, as ``misc/webservice.py``. It is only demonstration quality
|
||||||
|
and is not intended for production use.
|
||||||
|
|
||||||
|
OCRmyPDF is not designed for use as a public web service where a
|
||||||
|
malicious user could upload a chosen PDF. In particular, it is not
|
||||||
|
necessarily secure against PDF malware or PDFs that cause denial of
|
||||||
|
service. For further discussino of security, see :ref:`security`.
|
||||||
|
|
||||||
|
OCRmyPDF relies on Ghostscript, and therefore, if deployed
|
||||||
|
online one should be prepared to comply with Ghostscript's Affero GPL
|
||||||
|
license, and any other licenses.
|
||||||
|
|
||||||
|
Setting aside these concerns, a side effect of OCRmyPDF is that it may
|
||||||
|
incidentally sanitize PDFs containing certain types of malware. It
|
||||||
|
repairs the PDF with pikepdf/libqpdf, which could correct malformed PDF
|
||||||
|
structures that are part of an attack. When PDF/A output is selected
|
||||||
|
(the default), the input PDF is partially reconstructed by Ghostscript.
|
||||||
|
When ``--force-ocr`` is used, all pages are rasterized and reconverted
|
||||||
|
to PDF, which could remove malware in embedded images.
|
||||||
|
|
||||||
|
Limiting CPU usage
|
||||||
|
------------------
|
||||||
|
|
||||||
|
OCRmyPDF will attempt to use all available CPUs and storage, so
|
||||||
|
executing ``nice ocrmypdf`` or limiting the number of jobs with the
|
||||||
|
``--jobs`` argument may ensure the server remains responsive. Another option
|
||||||
|
would be to run OCRmyPDF jobs inside a Docker container, a virtual machine,
|
||||||
|
or a cloud instance, which can impose its own limits on CPU usage and be
|
||||||
|
terminated "from orbit" if it fails to complete.
|
||||||
|
|
||||||
|
Temporary storage requirements
|
||||||
|
------------------------------
|
||||||
|
|
||||||
|
OCRmyPDF will use a large amount of temporary storage for its work,
|
||||||
|
proportional to the total number of pixels needed to rasterize the PDF.
|
||||||
|
The raster image of a 8.5×11" color page at 300 DPI takes 25 MB
|
||||||
|
uncompressed; OCRmyPDF saves its intermediates as PNG, but that still
|
||||||
|
means it requires about 9 MB per intermediate based on average
|
||||||
|
compression ratios. Multiple intermediates per page are also required,
|
||||||
|
depending on the command line given. A rule of thumb would be to allow
|
||||||
|
100 MB of temporary storage per page in a file – meaning that a small
|
||||||
|
cloud servers or small VM partitions should be provisioned with plenty
|
||||||
|
of extra space, if say, a 500 page file might be sent.
|
||||||
|
|
||||||
|
To change the temporary directory, see :ref:`tmpdir`.
|
||||||
|
|
||||||
|
On Amazon Web Services or other cloud vendors, consider setting your
|
||||||
|
temporary directory to `empheral
|
||||||
|
storage <https://docs.aws.amazon.com/AWSEC2/latest/UserGuide/InstanceStorage.html>`__.
|
||||||
|
|
||||||
|
Timeouts
|
||||||
|
--------
|
||||||
|
|
||||||
|
To prevent excessively long OCR jobs consider setting
|
||||||
|
``--tesseract-timeout`` and/or ``--skip-big`` arguments. ``--skip-big``
|
||||||
|
is particularly helpful if your PDFs include documents such as reports
|
||||||
|
on standard page sizes with large images attached - often large images
|
||||||
|
are not worth OCR'ing anyway.
|
||||||
|
|
||||||
|
Document management systems
|
||||||
|
---------------------------
|
||||||
|
|
||||||
|
If you are looking for a full document management system, consider
|
||||||
|
`paperless-ngx <https://github.com/paperless-ngx/paperless-ngx>`__,
|
||||||
|
which is a web application that uses OCRmyPDF to automatically OCR and
|
||||||
|
archive documents.
|
||||||
|
|
||||||
|
Commercial OCR alternatives
|
||||||
|
---------------------------
|
||||||
|
|
||||||
|
The author also provides professional services that include OCR and
|
||||||
|
building databases around PDFs, and is happy to provide consultation.
|
||||||
|
|
||||||
|
Abbyy Cloud OCR is viable commercial alternative with a web services
|
||||||
|
API. Amazon Textract, Google Cloud Vision, and Microsoft Azure
|
||||||
|
Computer Vision provide advanced OCR but have less PDF rendering capability.
|
||||||
+4
-7
@@ -2,6 +2,8 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: CC-BY-SA-4.0
|
# SPDX-License-Identifier: CC-BY-SA-4.0
|
||||||
|
|
||||||
|
# ruff: noqa: E402
|
||||||
|
|
||||||
# ocrmypdf documentation build configuration file, created by
|
# ocrmypdf documentation build configuration file, created by
|
||||||
# sphinx-quickstart on Sun Sep 4 14:29:43 2016.
|
# sphinx-quickstart on Sun Sep 4 14:29:43 2016.
|
||||||
#
|
#
|
||||||
@@ -22,8 +24,6 @@
|
|||||||
# import sys
|
# import sys
|
||||||
# sys.path.insert(0, os.path.abspath('.'))
|
# sys.path.insert(0, os.path.abspath('.'))
|
||||||
|
|
||||||
"""isort:skip_file"""
|
|
||||||
|
|
||||||
# -- General configuration ------------------------------------------------
|
# -- General configuration ------------------------------------------------
|
||||||
|
|
||||||
# If your documentation needs a minimal Sphinx version, state it here.
|
# If your documentation needs a minimal Sphinx version, state it here.
|
||||||
@@ -65,7 +65,7 @@ master_doc = 'index'
|
|||||||
# General information about the project.
|
# General information about the project.
|
||||||
project = 'ocrmypdf'
|
project = 'ocrmypdf'
|
||||||
copyright = (
|
copyright = (
|
||||||
'2022, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.'
|
'2023, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.'
|
||||||
)
|
)
|
||||||
author = 'James R. Barlow'
|
author = 'James R. Barlow'
|
||||||
|
|
||||||
@@ -76,6 +76,7 @@ author = 'James R. Barlow'
|
|||||||
# The short X.Y version.
|
# The short X.Y version.
|
||||||
|
|
||||||
import os
|
import os
|
||||||
|
from importlib.metadata import version as package_version
|
||||||
|
|
||||||
on_rtd = os.environ.get('READTHEDOCS') == 'True'
|
on_rtd = os.environ.get('READTHEDOCS') == 'True'
|
||||||
|
|
||||||
@@ -96,10 +97,6 @@ if on_rtd:
|
|||||||
]
|
]
|
||||||
sys.modules.update((mod_name, Mock()) for mod_name in MOCK_MODULES)
|
sys.modules.update((mod_name, Mock()) for mod_name in MOCK_MODULES)
|
||||||
|
|
||||||
try:
|
|
||||||
from importlib_metadata import version as package_version
|
|
||||||
except ModuleNotFoundError:
|
|
||||||
from importlib.metadata import version as package_version
|
|
||||||
|
|
||||||
# The full version, including alpha/beta/rc tags.
|
# The full version, including alpha/beta/rc tags.
|
||||||
release = package_version('ocrmypdf')
|
release = package_version('ocrmypdf')
|
||||||
|
|||||||
+26
-9
@@ -18,7 +18,7 @@ work you're contemplating is already half-done in a development branch.
|
|||||||
Code style
|
Code style
|
||||||
==========
|
==========
|
||||||
|
|
||||||
We use PEP8, ``black`` for code formatting and ``isort`` for import sorting. The
|
We use PEP8, ``black`` for code formatting and ``ruff`` for everything else. The
|
||||||
settings for these programs are in ``pyproject.toml`` and ``setup.cfg``. Pull
|
settings for these programs are in ``pyproject.toml`` and ``setup.cfg``. Pull
|
||||||
requests should follow the style guide. One difference we use from "black" style
|
requests should follow the style guide. One difference we use from "black" style
|
||||||
is that strings shown to the user are always in double quotes (``"``) and strings
|
is that strings shown to the user are always in double quotes (``"``) and strings
|
||||||
@@ -29,12 +29,17 @@ Tests
|
|||||||
|
|
||||||
New features should come with tests that confirm their correctness.
|
New features should come with tests that confirm their correctness.
|
||||||
|
|
||||||
New Python dependencies
|
New dependencies
|
||||||
=======================
|
================
|
||||||
|
|
||||||
If you are proposing a change that will require a new Python dependency, we
|
If you are proposing a change that will require a new dependency, we
|
||||||
prefer dependencies that are already packaged by Debian or Red Hat. This makes
|
prefer dependencies that are already packaged by Debian or Red Hat. This makes
|
||||||
life much easier for our downstream package maintainers.
|
life much easier for our downstream package maintainers. A package that is only
|
||||||
|
available on PyPI or GitHub, and not more widely packaged, may not be accepted.
|
||||||
|
|
||||||
|
We are unlikely to accept a dependency on CUDA or other GPU-based libraries,
|
||||||
|
because these are still difficult to package and install on many systems.
|
||||||
|
We recommend implementing these changes as plugins.
|
||||||
|
|
||||||
Python dependencies must also be license-compatible. GPLv3 or AGPLv3 are likely
|
Python dependencies must also be license-compatible. GPLv3 or AGPLv3 are likely
|
||||||
incompatible with the project's license, but LGPLv3 is compatible.
|
incompatible with the project's license, but LGPLv3 is compatible.
|
||||||
@@ -43,7 +48,19 @@ New non-Python dependencies
|
|||||||
===========================
|
===========================
|
||||||
|
|
||||||
OCRmyPDF uses several external programs (Tesseract, Ghostscript and others) for
|
OCRmyPDF uses several external programs (Tesseract, Ghostscript and others) for
|
||||||
its functionality. In general we prefer to avoid adding new external programs.
|
its functionality. In general we prefer to avoid adding new external programs,
|
||||||
|
and if we are to add external programs, we prefer those that are already
|
||||||
|
packaged by Debian or Red Hat.
|
||||||
|
|
||||||
|
Plugins
|
||||||
|
=======
|
||||||
|
|
||||||
|
Some new features may be a good fit for a plugin. Plugins are a way to add
|
||||||
|
features to OCRmyPDF without adding them to the core program. Plugins are
|
||||||
|
installed separately from OCRmyPDF. They are written in Python and can be
|
||||||
|
installed from PyPI. See the `plugin documentation <https://ocrmypdf.readthedocs.io/en/latest/plugins.html>`_.
|
||||||
|
|
||||||
|
We are happy to link users to your plugin from the documentation.
|
||||||
|
|
||||||
Style guide: Is it OCRmyPDF or ocrmypdf?
|
Style guide: Is it OCRmyPDF or ocrmypdf?
|
||||||
========================================
|
========================================
|
||||||
@@ -53,8 +70,8 @@ The program/project is OCRmyPDF and the name of the executable or library is ocr
|
|||||||
Copyright and license
|
Copyright and license
|
||||||
=====================
|
=====================
|
||||||
|
|
||||||
For contributions over 10 lines of code, please include your name to list of
|
For contributions over 10 lines of code, please add your name to list of
|
||||||
copyright holders for that file. The core program is licensed under MPL-2.0,
|
copyright holders for that file. The core program is licensed under MPL-2.0,
|
||||||
test files and documentation under CC-BY-SA 4.0, and miscellaneous files under
|
test files and documentation under CC-BY-SA 4.0, and miscellaneous files under
|
||||||
MIT. Please contribute code only that you wrote and you have the permission to
|
MIT, with a few minor exceptions. Please contribute only content that you own
|
||||||
contribute or license to us.
|
or have the right to contribute under these licenses.
|
||||||
|
|||||||
+28
-6
@@ -231,13 +231,20 @@ Don't actually OCR my PDF
|
|||||||
=========================
|
=========================
|
||||||
|
|
||||||
If you set ``--tesseract-timeout 0`` OCRmyPDF will apply its image
|
If you set ``--tesseract-timeout 0`` OCRmyPDF will apply its image
|
||||||
processing without performing OCR, if all you want to is to apply image
|
processing without performing OCR (by causing OCR to time out). This works
|
||||||
processing or PDF/A conversion.
|
if all you want to is to apply image processing or PDF/A conversion.
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
ocrmypdf --tesseract-timeout=0 --remove-background input.pdf output.pdf
|
ocrmypdf --tesseract-timeout=0 --remove-background input.pdf output.pdf
|
||||||
|
|
||||||
|
.. versionchanged:: v14.1.0
|
||||||
|
|
||||||
|
Prior to this version, ``--tesseract-timeout 0`` would prevent other
|
||||||
|
uses of Tesseract, such as deskewing, from working. This is no longer
|
||||||
|
the case. Use ``--tesseract-non-ocr-timeout`` to control the timeout
|
||||||
|
for non-OCR operations, if needed.
|
||||||
|
|
||||||
Optimize images without performing OCR
|
Optimize images without performing OCR
|
||||||
--------------------------------------
|
--------------------------------------
|
||||||
|
|
||||||
@@ -261,9 +268,10 @@ Hyphens denote a range of pages and commas separate page numbers. If you prefer
|
|||||||
to use spaces, quote all of the page numbers: ``--pages '2, 3, 5, 7'``.
|
to use spaces, quote all of the page numbers: ``--pages '2, 3, 5, 7'``.
|
||||||
|
|
||||||
OCRmyPDF will warn if your list of page numbers contains duplicates or
|
OCRmyPDF will warn if your list of page numbers contains duplicates or
|
||||||
overlap pages. OCRmyPDF does not currently account for document page numbers,
|
overlapping pages. OCRmyPDF does not currently account for document page numbers,
|
||||||
such as an introduction section of a book that uses Roman numerals. It simply
|
such as an introduction section of a book that uses Roman numerals. It simply
|
||||||
counts the number of virtual pieces of paper since the start.
|
counts the number of virtual pieces of paper since the start. If your list of
|
||||||
|
pages is out of numerical order, OCRmyPDF will sort it for you.
|
||||||
|
|
||||||
Regardless of the argument to ``--pages``, OCRmyPDF will optimize all pages/images
|
Regardless of the argument to ``--pages``, OCRmyPDF will optimize all pages/images
|
||||||
in the file and convert it to PDF/A, unless you disable those options. Both of these
|
in the file and convert it to PDF/A, unless you disable those options. Both of these
|
||||||
@@ -283,7 +291,7 @@ argument. (Normally, OCRmyPDF will exit with an error if asked to modify
|
|||||||
a file with OCR.)
|
a file with OCR.)
|
||||||
|
|
||||||
This may be helpful for users who want to take advantage of accuracy
|
This may be helpful for users who want to take advantage of accuracy
|
||||||
improvements in Tesseract 4.0 for files they previously OCRed with an
|
improvements in Tesseract for files they previously OCRed with an
|
||||||
earlier version of Tesseract and OCRmyPDF.
|
earlier version of Tesseract and OCRmyPDF.
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
@@ -360,7 +368,7 @@ The types of optimization available may expand over time. By default,
|
|||||||
OCRmyPDF compresses data streams inside PDFs, and will change
|
OCRmyPDF compresses data streams inside PDFs, and will change
|
||||||
inefficient compression modes to more modern versions. A program like
|
inefficient compression modes to more modern versions. A program like
|
||||||
``qpdf`` can be used to change encodings, e.g. to inspect the internals
|
``qpdf`` can be used to change encodings, e.g. to inspect the internals
|
||||||
fo a PDF.
|
for a PDF.
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -372,3 +380,17 @@ Some users may consider enabling lossy JBIG2. See: :ref:`jbig2-lossy`.
|
|||||||
|
|
||||||
Image processing and PDF/A conversion can also introduce lossy transformations
|
Image processing and PDF/A conversion can also introduce lossy transformations
|
||||||
to your PDF images, even when ``--optimize 1`` is in use.
|
to your PDF images, even when ``--optimize 1`` is in use.
|
||||||
|
|
||||||
|
|
||||||
|
Digitally signed PDFs
|
||||||
|
=====================
|
||||||
|
|
||||||
|
OCRmyPDF cannot preserve digital signatures in PDFs and also add to OCR to them.
|
||||||
|
By default, it will refuse to modify a signed PDF regardless of other settings. You can
|
||||||
|
override this behavior with ``--invalidate-digital-signatures``; as the name suggests,
|
||||||
|
any digital signatures will be invalidated.
|
||||||
|
|
||||||
|
OCRmyPDF cannot open documents that are encrypted with a digital certificate.
|
||||||
|
|
||||||
|
Versions of OCRmyPDF prior to 14.4.0 would invalidate existing digital signatures
|
||||||
|
without warning.
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
.. SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||||
|
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||||
|
|
||||||
|
============
|
||||||
|
Design notes
|
||||||
|
============
|
||||||
|
|
||||||
|
Why doesn't OCRmyPDF use PyTesseract?
|
||||||
|
=====================================
|
||||||
|
|
||||||
|
PyTesseract is a Python wrapper around the Tesseract OCR engine. When OCRmyPDF was
|
||||||
|
first written, PyTesseract used ABI bindings to call the Tesseract library. This
|
||||||
|
was not a good fit for OCRmyPDF because ABI bindings can be fragile.
|
||||||
|
|
||||||
|
PyTesseract has since evolved calling the Tesseract executable, abandoning the ABI
|
||||||
|
approach and using the CLI instead, just like OCRmyPDF does. If it were written from
|
||||||
|
scratch today, OCRmyPDF might use PyTesseract.
|
||||||
|
|
||||||
|
PyTesseract has more features don't particularly need PDF output, but less features
|
||||||
|
than OCRmyPDF's API for creating PDFs.
|
||||||
|
|
||||||
|
What is ``executor()``?
|
||||||
|
=======================
|
||||||
|
|
||||||
|
OCRmyPDF uses a custom concurrent executor which can support either threads or
|
||||||
|
processes with the same interface. This is useful because OCRmyPDF can use
|
||||||
|
either threads or processes to parallelize work, whichever is more appropriate
|
||||||
|
for the task at hand.
|
||||||
|
|
||||||
|
The interface is currently private and subject to change. In particular, if
|
||||||
|
experiments with asyncio and anyio are successful, the interface will change.
|
||||||
|
|
||||||
+3
-1
@@ -20,7 +20,6 @@ image processing and OCR to existing PDFs.
|
|||||||
introduction
|
introduction
|
||||||
release_notes
|
release_notes
|
||||||
installation
|
installation
|
||||||
optimizer
|
|
||||||
languages
|
languages
|
||||||
jbig2
|
jbig2
|
||||||
|
|
||||||
@@ -29,9 +28,11 @@ image processing and OCR to existing PDFs.
|
|||||||
:maxdepth: 2
|
:maxdepth: 2
|
||||||
|
|
||||||
cookbook
|
cookbook
|
||||||
|
optimizer
|
||||||
docker
|
docker
|
||||||
advanced
|
advanced
|
||||||
batch
|
batch
|
||||||
|
cloud
|
||||||
performance
|
performance
|
||||||
pdfsecurity
|
pdfsecurity
|
||||||
errors
|
errors
|
||||||
@@ -43,6 +44,7 @@ image processing and OCR to existing PDFs.
|
|||||||
api
|
api
|
||||||
plugins
|
plugins
|
||||||
apiref
|
apiref
|
||||||
|
design_notes
|
||||||
contributing
|
contributing
|
||||||
maintainers
|
maintainers
|
||||||
|
|
||||||
|
|||||||
+43
-106
@@ -21,7 +21,7 @@ These platforms have one-liner installs:
|
|||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| Fedora | ``dnf install ocrmypdf`` |
|
| Fedora | ``dnf install ocrmypdf tesseract-osd`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| macOS | ``brew install ocrmypdf`` |
|
| macOS | ``brew install ocrmypdf`` |
|
||||||
+-------------------------------+-----------------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
@@ -44,7 +44,7 @@ install, or install a more recent version than your platform provides, read on.
|
|||||||
Installing on Linux
|
Installing on Linux
|
||||||
===================
|
===================
|
||||||
|
|
||||||
Debian and Ubuntu 18.04 or newer
|
Debian and Ubuntu 20.04 or newer
|
||||||
--------------------------------
|
--------------------------------
|
||||||
|
|
||||||
.. |deb-11| image:: https://repology.org/badge/version-for-repo/debian_11/ocrmypdf.svg
|
.. |deb-11| image:: https://repology.org/badge/version-for-repo/debian_11/ocrmypdf.svg
|
||||||
@@ -56,9 +56,6 @@ Debian and Ubuntu 18.04 or newer
|
|||||||
.. |deb-unstable| image:: https://repology.org/badge/version-for-repo/debian_unstable/ocrmypdf.svg
|
.. |deb-unstable| image:: https://repology.org/badge/version-for-repo/debian_unstable/ocrmypdf.svg
|
||||||
:alt: Debian unstable
|
:alt: Debian unstable
|
||||||
|
|
||||||
.. |ubu-1804| image:: https://repology.org/badge/version-for-repo/ubuntu_18_04/ocrmypdf.svg
|
|
||||||
:alt: Ubuntu 18.04 LTS
|
|
||||||
|
|
||||||
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
||||||
:alt: Ubuntu 20.04 LTS
|
:alt: Ubuntu 20.04 LTS
|
||||||
|
|
||||||
@@ -72,15 +69,14 @@ Debian and Ubuntu 18.04 or newer
|
|||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |deb-11| |deb-12| |deb-unstable| |
|
| |deb-11| |deb-12| |deb-unstable| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |ubu-1804| |ubu-2004| |ubu-2204| |
|
| |ubu-2004| |ubu-2204| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
|
|
||||||
Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users
|
Users of Debian 11, or Ubuntu 20.04 LTS, or newer may simply
|
||||||
of Windows Subsystem for Linux, may simply
|
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
apt-get install ocrmypdf
|
apt install ocrmypdf
|
||||||
|
|
||||||
As indicated in the table above, Debian and Ubuntu releases may lag
|
As indicated in the table above, Debian and Ubuntu releases may lag
|
||||||
behind the latest version. If the version available for your platform is
|
behind the latest version. If the version available for your platform is
|
||||||
@@ -124,7 +120,7 @@ Users of Fedora 29 or later may simply
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
dnf install ocrmypdf
|
dnf install ocrmypdf tesseract-osd
|
||||||
|
|
||||||
For full details on version availability, check the `Fedora Package
|
For full details on version availability, check the `Fedora Package
|
||||||
Tracker <https://apps.fedoraproject.org/packages/ocrmypdf>`__.
|
Tracker <https://apps.fedoraproject.org/packages/ocrmypdf>`__.
|
||||||
@@ -198,46 +194,6 @@ To install for the current user only:
|
|||||||
|
|
||||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||||
|
|
||||||
Ubuntu 18.04 LTS
|
|
||||||
----------------
|
|
||||||
|
|
||||||
Ubuntu 18.04 includes ocrmypdf 6.1.2 - you can install that with ``apt``, but
|
|
||||||
it is quite old now. To install a more recent version, uninstall the old version
|
|
||||||
of ocrmypdf, and install the following dependencies:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
sudo apt-get -y remove ocrmypdf
|
|
||||||
sudo apt-get -y update
|
|
||||||
sudo apt-get -y install \
|
|
||||||
ghostscript \
|
|
||||||
icc-profiles-free \
|
|
||||||
libxml2 \
|
|
||||||
pngquant \
|
|
||||||
python3-distutils \
|
|
||||||
python3-pkg-resources \
|
|
||||||
python3-reportlab \
|
|
||||||
qpdf \
|
|
||||||
tesseract-ocr \
|
|
||||||
zlib1g \
|
|
||||||
unpaper
|
|
||||||
|
|
||||||
We will need a newer version of ``pip`` then was available for Ubuntu 18.04:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
wget https://bootstrap.pypa.io/get-pip.py && python3 get-pip.py
|
|
||||||
|
|
||||||
Then install the most recent ocrmypdf for the local user and set the
|
|
||||||
user's ``PATH`` to check for the user's Python packages.
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
export PATH=$HOME/.local/bin:$PATH
|
|
||||||
python3 -m pip install --user ocrmypdf
|
|
||||||
|
|
||||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
|
||||||
|
|
||||||
Arch Linux (AUR)
|
Arch Linux (AUR)
|
||||||
----------------
|
----------------
|
||||||
|
|
||||||
@@ -318,6 +274,21 @@ To install OCRmyPDF for Alpine Linux:
|
|||||||
|
|
||||||
apk add ocrmypdf
|
apk add ocrmypdf
|
||||||
|
|
||||||
|
Gentoo Linux
|
||||||
|
------------
|
||||||
|
|
||||||
|
.. image:: https://repology.org/badge/version-for-repo/gentoo_ovl_guru/ocrmypdf.svg
|
||||||
|
:alt: Gentoo Linux
|
||||||
|
:target: https://repology.org/metapackage/ocrmypdf
|
||||||
|
|
||||||
|
To install OCRmyPDF on Gentoo Linux, use the following commands:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
eselect repository enable guru
|
||||||
|
emaint sync --repo guru
|
||||||
|
emerge --ask app-text/OCRmyPDF
|
||||||
|
|
||||||
Other Linux packages
|
Other Linux packages
|
||||||
--------------------
|
--------------------
|
||||||
|
|
||||||
@@ -336,7 +307,7 @@ Homebrew
|
|||||||
|
|
||||||
.. image:: https://img.shields.io/homebrew/v/ocrmypdf.svg
|
.. image:: https://img.shields.io/homebrew/v/ocrmypdf.svg
|
||||||
:alt: homebrew
|
:alt: homebrew
|
||||||
:target: http://brewformulas.org/Ocrmypdf
|
:target: https://formulae.brew.sh/formula/ocrmypdf
|
||||||
|
|
||||||
OCRmyPDF is now a standard `Homebrew <https://brew.sh>`__ formula. To
|
OCRmyPDF is now a standard `Homebrew <https://brew.sh>`__ formula. To
|
||||||
install on macOS:
|
install on macOS:
|
||||||
@@ -387,18 +358,12 @@ Update the homebrew pip:
|
|||||||
|
|
||||||
pip install --upgrade pip
|
pip install --upgrade pip
|
||||||
|
|
||||||
You can then install OCRmyPDF from PyPI, for the current user:
|
You can then install OCRmyPDF from PyPI for the current user:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip install --user ocrmypdf
|
pip install --user ocrmypdf
|
||||||
|
|
||||||
or system-wide:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
pip install ocrmypdf
|
|
||||||
|
|
||||||
The command line program should now be available:
|
The command line program should now be available:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
@@ -417,9 +382,9 @@ Native Windows
|
|||||||
|
|
||||||
You must install the following for Windows:
|
You must install the following for Windows:
|
||||||
|
|
||||||
* Python 3.7 (64-bit) or later
|
* Python 3.9 (64-bit) or later
|
||||||
* Tesseract 4.0 or later
|
* Tesseract 4.1.1 (64-bit) or later
|
||||||
* Ghostscript 9.50 or later
|
* Ghostscript 9.50 (64-bit) or later
|
||||||
|
|
||||||
Using the `Chocolatey <https://chocolatey.org/>`_ package manager, install the
|
Using the `Chocolatey <https://chocolatey.org/>`_ package manager, install the
|
||||||
following when running in an Administrator command prompt:
|
following when running in an Administrator command prompt:
|
||||||
@@ -438,10 +403,8 @@ Administrator.):
|
|||||||
|
|
||||||
* ``pip install ocrmypdf``
|
* ``pip install ocrmypdf``
|
||||||
|
|
||||||
Chocolatey automatically selects appropriate versions of these applications. If you
|
Chocolatey automatically selects appropriate versions of these applications. Please make sure
|
||||||
are installing them manually, please install 64-bit versions of all applications for
|
you are installing the 64-bit versions.
|
||||||
64-bit Windows, or 32-bit versions of all applications for 32-bit Windows. Mixing
|
|
||||||
the "bitness" of these programs will lead to errors.
|
|
||||||
|
|
||||||
OCRmyPDF will check the Windows Registry and standard locations in your Program Files
|
OCRmyPDF will check the Windows Registry and standard locations in your Program Files
|
||||||
for third party software it needs (specifically, Tesseract and Ghostscript). To
|
for third party software it needs (specifically, Tesseract and Ghostscript). To
|
||||||
@@ -456,6 +419,10 @@ to change the PATH.
|
|||||||
Please download Python from Python.org or Chocolatey instead, and do not use the
|
Please download Python from Python.org or Chocolatey instead, and do not use the
|
||||||
Microsoft Store version.
|
Microsoft Store version.
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
|
||||||
|
32-bit Windows might work, but is not supported.
|
||||||
|
|
||||||
Windows Subsystem for Linux
|
Windows Subsystem for Linux
|
||||||
---------------------------
|
---------------------------
|
||||||
|
|
||||||
@@ -481,7 +448,7 @@ Cygwin64
|
|||||||
|
|
||||||
First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``::
|
First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``::
|
||||||
|
|
||||||
python37 (or later)
|
python38 (or later)
|
||||||
python3?-devel
|
python3?-devel
|
||||||
python3?-pip
|
python3?-pip
|
||||||
python3?-lxml
|
python3?-lxml
|
||||||
@@ -559,21 +526,6 @@ the latest version. However, PyPI and ``pip`` cannot address the fact
|
|||||||
that ``ocrmypdf`` depends on certain non-Python system libraries and
|
that ``ocrmypdf`` depends on certain non-Python system libraries and
|
||||||
programs being installed.
|
programs being installed.
|
||||||
|
|
||||||
.. warning::
|
|
||||||
|
|
||||||
Debian and Ubuntu users: unfortunately, Debian and Ubuntu customize
|
|
||||||
Python in non-standard ways, and the nature of these customizations
|
|
||||||
varies from release to release. This can make for a frustrating
|
|
||||||
user experience. The instructions below work on almost all platforms that
|
|
||||||
have Python installed, except for Debian and Ubuntu, where you may need
|
|
||||||
to take additional steps. For best results on Debian and Ubuntu, use the
|
|
||||||
``apt`` packages; or if these are too old, run
|
|
||||||
``apt install python3-pip python3-venv``, create a virtual environment,
|
|
||||||
and install OCRmyPDF in that environment.
|
|
||||||
|
|
||||||
`See here for more inforation on Debian-Python issues
|
|
||||||
<https://gist.github.com/tiran/2dec9e03c6f901814f6d1e8dad09528e>`__.
|
|
||||||
|
|
||||||
For best results, first install `your platform's
|
For best results, first install `your platform's
|
||||||
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
|
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
|
||||||
``ocrmypdf``, using the instructions elsewhere in this document. Then
|
``ocrmypdf``, using the instructions elsewhere in this document. Then
|
||||||
@@ -592,21 +544,6 @@ try:
|
|||||||
You should then be able to run ``ocrmypdf --version`` and see that the
|
You should then be able to run ``ocrmypdf --version`` and see that the
|
||||||
latest version was located.
|
latest version was located.
|
||||||
|
|
||||||
Since ``pip install --user`` does not work correctly on some platforms,
|
|
||||||
notably Ubuntu 16.04 and older, and the Homebrew version of Python,
|
|
||||||
instead use this for a system wide installation:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
pip install ocrmypdf
|
|
||||||
|
|
||||||
.. note::
|
|
||||||
|
|
||||||
AArch64 (ARM64) users: this process will be difficult because most
|
|
||||||
Python packages are not available as binary wheels for your platform.
|
|
||||||
You're probably better off using a platform install on Debian, Ubuntu,
|
|
||||||
or Fedora.
|
|
||||||
|
|
||||||
Requirements for pip and HEAD install
|
Requirements for pip and HEAD install
|
||||||
-------------------------------------
|
-------------------------------------
|
||||||
|
|
||||||
@@ -616,9 +553,9 @@ manager. ``pip`` cannot provide them.
|
|||||||
|
|
||||||
The following versions are required:
|
The following versions are required:
|
||||||
|
|
||||||
- Python 3.7 or newer
|
- Python 3.9 or newer
|
||||||
- Ghostscript 9.23 or newer
|
- Ghostscript 9.50 or newer
|
||||||
- Tesseract 4.0.0 or newer
|
- Tesseract 4.1.1 or newer
|
||||||
- jbig2enc 0.29 or newer
|
- jbig2enc 0.29 or newer
|
||||||
- pngquant 2.5 or newer
|
- pngquant 2.5 or newer
|
||||||
- unpaper 6.1
|
- unpaper 6.1
|
||||||
@@ -649,7 +586,7 @@ unfortunately, the ``pip install`` command cannot satisfy all of them.
|
|||||||
Installing HEAD revision from sources
|
Installing HEAD revision from sources
|
||||||
=====================================
|
=====================================
|
||||||
|
|
||||||
If you have ``git`` and Python 3.7 or newer installed, you can install
|
If you have ``git`` and Python 3.9 or newer installed, you can install
|
||||||
from source. When the ``pip`` installer runs, it will alert you if
|
from source. When the ``pip`` installer runs, it will alert you if
|
||||||
dependencies are missing.
|
dependencies are missing.
|
||||||
|
|
||||||
@@ -678,9 +615,9 @@ system-wide:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
git clone -b master https://github.com/ocrmypdf/OCRmyPDF.git
|
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
python3 -m venv
|
python3 -m venv .venv
|
||||||
source venv/bin/activate
|
source .venv/bin/activate
|
||||||
cd OCRmyPDF
|
cd OCRmyPDF
|
||||||
pip install .
|
pip install .
|
||||||
|
|
||||||
@@ -705,9 +642,9 @@ To install all of the development and test requirements:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
git clone -b master https://github.com/ocrmypdf/OCRmyPDF.git
|
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
python -m venv
|
python -m .venv
|
||||||
source venv/bin/activate
|
source .venv/bin/activate
|
||||||
cd OCRmyPDF
|
cd OCRmyPDF
|
||||||
pip install -e .[test]
|
pip install -e .[test]
|
||||||
|
|
||||||
|
|||||||
@@ -85,7 +85,7 @@ OCRmyPDF analyzes each page of a PDF to determine the colorspace and
|
|||||||
resolution (DPI) needed to capture all of the information on that page
|
resolution (DPI) needed to capture all of the information on that page
|
||||||
without losing content. It uses
|
without losing content. It uses
|
||||||
`Ghostscript <http://ghostscript.com/>`__ to rasterize the page, and
|
`Ghostscript <http://ghostscript.com/>`__ to rasterize the page, and
|
||||||
then performs on OCR the rasterized image to create an OCR "layer".
|
then performs OCR on the rasterized image to create an OCR "layer".
|
||||||
The layer is then grafted back onto the original PDF.
|
The layer is then grafted back onto the original PDF.
|
||||||
|
|
||||||
While one can use a program like Ghostscript or ImageMagick to get an
|
While one can use a program like Ghostscript or ImageMagick to get an
|
||||||
@@ -190,11 +190,10 @@ Ghostscript also imposes some limitations:
|
|||||||
behavior can be suppressed by setting ``--pdfa-image-compression`` to
|
behavior can be suppressed by setting ``--pdfa-image-compression`` to
|
||||||
``jpeg`` or ``lossless`` to set all images to one type or the other.
|
``jpeg`` or ``lossless`` to set all images to one type or the other.
|
||||||
Ghostscript has no option to maintain the input image's format.
|
Ghostscript has no option to maintain the input image's format.
|
||||||
(Ghostscript 9.25+ can copy JPEG images without transcoding them;
|
(Modern Ghostscript can copy JPEG images without transcoding them.)
|
||||||
earlier versions will transcode.)
|
|
||||||
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
||||||
one of the standard XMP metadata namespaces for PDFs. In particular,
|
one of the standard XMP metadata namespaces for PDFs. In particular,
|
||||||
PRISM Metdata is removed.
|
PRISM Metadata is removed.
|
||||||
- Ghostscript's PDF/A conversion seems to remove or deactivate
|
- Ghostscript's PDF/A conversion seems to remove or deactivate
|
||||||
hyperlinks and other active content.
|
hyperlinks and other active content.
|
||||||
|
|
||||||
|
|||||||
+2
-1
@@ -37,7 +37,8 @@ For all other Linux, you must build a JBIG2 encoder from source:
|
|||||||
.. _jbig2-lossy:
|
.. _jbig2-lossy:
|
||||||
|
|
||||||
Dependencies include libtoolize and libleptonica, which on Ubuntu systems
|
Dependencies include libtoolize and libleptonica, which on Ubuntu systems
|
||||||
are packaged as libtool and libleptonica-dev.
|
are packaged as libtool and libleptonica-dev. On Fedora (35) they are packaged
|
||||||
|
as libtool and leptonica-devel.
|
||||||
|
|
||||||
Lossy mode JBIG2
|
Lossy mode JBIG2
|
||||||
================
|
================
|
||||||
|
|||||||
+8
-8
@@ -12,7 +12,7 @@ OCRmyPDF uses Tesseract for OCR, and relies on its language packs for all langua
|
|||||||
On most platforms, English is installed with Tesseract by default, but not always.
|
On most platforms, English is installed with Tesseract by default, but not always.
|
||||||
|
|
||||||
Tesseract supports `most
|
Tesseract supports `most
|
||||||
languages <https://github.com/tesseract-ocr/tesseract/blob/master/doc/tesseract.1.asc#languages>`__.
|
languages <https://github.com/tesseract-ocr/tesseract/blob/main/doc/tesseract.1.asc#languages>`__.
|
||||||
Languages are identified by standardized three-letter codes (called ISO 639-2 Alpha-3).
|
Languages are identified by standardized three-letter codes (called ISO 639-2 Alpha-3).
|
||||||
Tesseract's documentation also lists the three-letter code for your language.
|
Tesseract's documentation also lists the three-letter code for your language.
|
||||||
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
||||||
@@ -70,13 +70,13 @@ This enables these languages for all packages (e.g. including aspell).
|
|||||||
|
|
||||||
# Display a list of all Tesseract language packs
|
# Display a list of all Tesseract language packs
|
||||||
equery uses app-text/tessdata_fast
|
equery uses app-text/tessdata_fast
|
||||||
|
|
||||||
# Add English and German language support for Tesseract only
|
# Add English and German language support for Tesseract only
|
||||||
echo 'app-text/tessdata_fast l10n_de l10n_en' >> /etc/portage/package.use
|
echo 'app-text/tessdata_fast l10n_de l10n_en' >> /etc/portage/package.use
|
||||||
|
|
||||||
# Add global English and German language support (the `l10n_` from equery has to be omited)
|
# Add global English and German language support (the `l10n_` from equery has to be omitted)
|
||||||
echo L10N="de en" >> /etc/portage/make.conf
|
echo L10N="de en" >> /etc/portage/make.conf
|
||||||
|
|
||||||
# update system to reflect changed USE flags
|
# update system to reflect changed USE flags
|
||||||
emerge --update --deep --newuse @world
|
emerge --update --deep --newuse @world
|
||||||
|
|
||||||
@@ -101,7 +101,7 @@ derived Docker image as
|
|||||||
Windows users
|
Windows users
|
||||||
=============
|
=============
|
||||||
|
|
||||||
The Tesseract installer provided by Chocolatey currently includes only English language.
|
The Tesseract installer provided by Chocolatey currently includes only English language.
|
||||||
To install other languages, download the respective language pack (``.traineddata`` file)
|
To install other languages, download the respective language pack (``.traineddata`` file)
|
||||||
from https://github.com/tesseract-ocr/tessdata/ and place it in
|
from https://github.com/tesseract-ocr/tessdata/ and place it in
|
||||||
``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed).
|
``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed).
|
||||||
|
|||||||
@@ -45,11 +45,6 @@ to indicate that your distribution modifies OCRmyPDF in some way.
|
|||||||
You can patch the ``__version__`` variable in ``src/ocrmypdf/_version.py`` if
|
You can patch the ``__version__`` variable in ``src/ocrmypdf/_version.py`` if
|
||||||
necessary.
|
necessary.
|
||||||
|
|
||||||
OCRmyPDF uses setuptools-scm-git-archive to ensure that tarballs downloaded from
|
|
||||||
GitHub contain version information. Unfortunately, these tarballs are not always
|
|
||||||
deterministic. See this
|
|
||||||
`issue <https://github.com/ocrmypdf/OCRmyPDF/issues/841#issuecomment-936562696>`_.
|
|
||||||
|
|
||||||
jbig2enc
|
jbig2enc
|
||||||
--------
|
--------
|
||||||
|
|
||||||
|
|||||||
+30
-9
@@ -13,14 +13,33 @@ tuned. Optimization occurs after OCR, and only if OCR succeeded. It does not
|
|||||||
perform other possible optimizations such as deduplicating resources,
|
perform other possible optimizations such as deduplicating resources,
|
||||||
consolidating fonts, simplifying vector drawings, or anything of that nature.
|
consolidating fonts, simplifying vector drawings, or anything of that nature.
|
||||||
|
|
||||||
Optimization ranges from ``-O0`` through ``-O3``, where ``0`` disables
|
.. list-table:: Title
|
||||||
optimization and ``3`` implements all options. ``1``, the default, performs only
|
:widths: 33 6 60
|
||||||
safe and lossless optimizations. (This is similar to GCC's optimization
|
:header-rows: 1
|
||||||
parameter.) The exact type of optimizations performed will vary over time.
|
|
||||||
|
|
||||||
PDF optimization requires third-party, optional tools for certain optimizations.
|
* - Optimization level
|
||||||
If these are not installed or cannot be found by OCRmyPDF, optimization will not
|
- Shorthand
|
||||||
be as good.
|
- Description
|
||||||
|
* - ``--optimize 0``
|
||||||
|
- ``-O0``
|
||||||
|
- Disable most optimizations.
|
||||||
|
* - ``--optimize 1`` (default)
|
||||||
|
- ``-O1``
|
||||||
|
- Safe and lossless optimizations.
|
||||||
|
* - ``--optimize 2``
|
||||||
|
- ``-O2``
|
||||||
|
- Safe and lossy optimizations.
|
||||||
|
* - ``--optimize 3``
|
||||||
|
- ``-O3``
|
||||||
|
- Aggressive lossy optimizations.
|
||||||
|
|
||||||
|
The exact type of optimizations performed will vary over time, and depend on
|
||||||
|
the availability of third-party tools.
|
||||||
|
|
||||||
|
Despite optimizations, OCRmyPDF might still increase the overall file size,
|
||||||
|
since it must embed information about the recognized text, and depending on the
|
||||||
|
settings chosen, may not be able to represent the output file as compactly as
|
||||||
|
the input file.
|
||||||
|
|
||||||
Optimizations that always occurs
|
Optimizations that always occurs
|
||||||
================================
|
================================
|
||||||
@@ -37,12 +56,14 @@ Fast web view
|
|||||||
OCRmyPDF automatically optimizes PDFs for "fast web view" in Adobe Acrobat's
|
OCRmyPDF automatically optimizes PDFs for "fast web view" in Adobe Acrobat's
|
||||||
parlance, or equivalently, linearizes PDFs so that the resources they reference
|
parlance, or equivalently, linearizes PDFs so that the resources they reference
|
||||||
are presented in the order a viewer needs them for sequential display. This
|
are presented in the order a viewer needs them for sequential display. This
|
||||||
reduces the latency of viewing a PDF both online and from local storage. This
|
reduces the latency of viewing a PDF both online and from local storage, in
|
||||||
actually slightly increases the file size.
|
exchange for a slight increase in file size.
|
||||||
|
|
||||||
To disable this optimization and all others, use ``ocrmypdf --optimize 0 ...``
|
To disable this optimization and all others, use ``ocrmypdf --optimize 0 ...``
|
||||||
or the shorthand ``-O0``.
|
or the shorthand ``-O0``.
|
||||||
|
|
||||||
|
Adobe Acrobat might not report the file as being "fast web view".
|
||||||
|
|
||||||
Lossless optimizations
|
Lossless optimizations
|
||||||
======================
|
======================
|
||||||
|
|
||||||
|
|||||||
+61
-102
@@ -29,13 +29,9 @@ attack vectors.
|
|||||||
In short, PDFs `may contain
|
In short, PDFs `may contain
|
||||||
viruses <https://security.stackexchange.com/questions/64052/can-a-pdf-file-contain-a-virus>`__.
|
viruses <https://security.stackexchange.com/questions/64052/can-a-pdf-file-contain-a-virus>`__.
|
||||||
|
|
||||||
This
|
If you do not trust a PDF or its source, do not open it or use OCRmyPDF
|
||||||
`article <https://theinvisiblethings.blogspot.ca/2013/02/converting-untrusted-pdfs-into-trusted.html>`__
|
on it. Consider using a Docker container or virtual machine to isolate
|
||||||
describes a high-paranoia method which allows potentially hostile PDFs
|
an untrusted PDF from your system.
|
||||||
to be viewed and rasterized safely in a disposable virtual machine. A
|
|
||||||
trusted PDF created in this manner is converted to images and loses all
|
|
||||||
information making it searchable and losing all compression. OCRmyPDF
|
|
||||||
could be used to restore searchability.
|
|
||||||
|
|
||||||
How OCRmyPDF processes PDFs
|
How OCRmyPDF processes PDFs
|
||||||
===========================
|
===========================
|
||||||
@@ -43,11 +39,11 @@ How OCRmyPDF processes PDFs
|
|||||||
OCRmyPDF must open and interpret your PDF in order to insert an OCR
|
OCRmyPDF must open and interpret your PDF in order to insert an OCR
|
||||||
layer. First, it runs all PDFs through
|
layer. First, it runs all PDFs through
|
||||||
`pikepdf <https://github.com/pikepdf/pikepdf>`__, a library based on
|
`pikepdf <https://github.com/pikepdf/pikepdf>`__, a library based on
|
||||||
`qpdf <https://github.com/qpdf/qpdf>`__, a program that repairs PDFs
|
`QPDF <https://github.com/qpdf/qpdf>`__, a program that repairs PDFs
|
||||||
with syntax errors. This is done because, in the author's experience, a
|
with syntax errors. This is done because, in the author's experience, a
|
||||||
significant number of PDFs in the wild, especially those created by
|
significant number of PDFs in the wild, especially those created by
|
||||||
scanners, are not well-formed files. qpdf makes it more likely that
|
scanners, are not well-formed files. QPDF makes it more likely that
|
||||||
OCRmyPDF will succeed, but offers no security guarantees. qpdf is also
|
OCRmyPDF will succeed, but offers no security guarantees. QPDF is also
|
||||||
used to split the PDF into single page PDFs.
|
used to split the PDF into single page PDFs.
|
||||||
|
|
||||||
Finally, OCRmyPDF rasterizes each page of the PDF using
|
Finally, OCRmyPDF rasterizes each page of the PDF using
|
||||||
@@ -58,109 +54,72 @@ into the existing PDF or it may essentially reconstruct ("re-fry") a
|
|||||||
visually identical PDF that may be quite different at the binary level.
|
visually identical PDF that may be quite different at the binary level.
|
||||||
That said, OCRmyPDF is not a tool designed for sanitizing PDFs.
|
That said, OCRmyPDF is not a tool designed for sanitizing PDFs.
|
||||||
|
|
||||||
.. _ocr-service:
|
Password protected PDFs
|
||||||
|
=======================
|
||||||
Using OCRmyPDF online or as a service
|
|
||||||
=====================================
|
|
||||||
|
|
||||||
OCRmyPDF is not designed for use as a public web service where a
|
|
||||||
malicious user could upload a chosen PDF. In particular, it is not
|
|
||||||
necessarily secure against PDF malware or PDFs that cause denial of
|
|
||||||
service. OCRmyPDF relies on Ghostscript, and therefore, if deployed
|
|
||||||
online one should be prepared to comply with Ghostscript's Affero GPL
|
|
||||||
license, and any other licenses.
|
|
||||||
|
|
||||||
Setting aside these concerns, a side effect of OCRmyPDF is that it may
|
|
||||||
incidentally sanitize PDFs containing certain types of malware. It
|
|
||||||
repairs the PDF with pikepdf/libqpdf, which could correct malformed PDF
|
|
||||||
structures that are part of an attack. When PDF/A output is selected
|
|
||||||
(the default), the input PDF is partially reconstructed by Ghostscript.
|
|
||||||
When ``--force-ocr`` is used, all pages are rasterized and reconverted
|
|
||||||
to PDF, which could remove malware in embedded images.
|
|
||||||
|
|
||||||
OCRmyPDF should be relatively safe to use in a trusted intranet, with
|
|
||||||
some considerations:
|
|
||||||
|
|
||||||
Limiting CPU usage
|
|
||||||
------------------
|
|
||||||
|
|
||||||
OCRmyPDF will attempt to use all available CPUs and storage, so
|
|
||||||
executing ``nice ocrmypdf`` or limiting the number of jobs with the
|
|
||||||
``-j`` argument may ensure the server remains available. Another option
|
|
||||||
would be to run OCRmyPDF jobs inside a Docker container, a virtual machine,
|
|
||||||
or a cloud instance, which can impose its own limits on CPU usage and be
|
|
||||||
terminated "from orbit" if it fails to complete.
|
|
||||||
|
|
||||||
Temporary storage requirements
|
|
||||||
------------------------------
|
|
||||||
|
|
||||||
OCRmyPDF will use a large amount of temporary storage for its work,
|
|
||||||
proportional to the total number of pixels needed to rasterize the PDF.
|
|
||||||
The raster image of a 8.5×11" color page at 300 DPI takes 25 MB
|
|
||||||
uncompressed; OCRmyPDF saves its intermediates as PNG, but that still
|
|
||||||
means it requires about 9 MB per intermediate based on average
|
|
||||||
compression ratios. Multiple intermediates per page are also required,
|
|
||||||
depending on the command line given. A rule of thumb would be to allow
|
|
||||||
100 MB of temporary storage per page in a file – meaning that a small
|
|
||||||
cloud servers or small VM partitions should be provisioned with plenty
|
|
||||||
of extra space, if say, a 500 page file might be sent.
|
|
||||||
|
|
||||||
To check temporary storage usage on actual files, run
|
|
||||||
``ocrmypdf -k ...`` which will preserve and print the path to temporary
|
|
||||||
storage when the job is done.
|
|
||||||
|
|
||||||
To change where temporary files are stored, change the ``TMPDIR``
|
|
||||||
environment variable for ocrmypdf's environment. (Python's
|
|
||||||
``tempfile.gettempdir()`` returns the root directory in which temporary
|
|
||||||
files will be stored.) For example, one could redirect ``TMPDIR`` to a
|
|
||||||
large RAM disk to avoid wear on HDD/SSD and potentially improve
|
|
||||||
performance. On Amazon Web Services, ``TMPDIR`` can be set to `empheral
|
|
||||||
storage <https://docs.aws.amazon.com/AWSEC2/latest/UserGuide/InstanceStorage.html>`__.
|
|
||||||
|
|
||||||
Timeouts
|
|
||||||
--------
|
|
||||||
|
|
||||||
To prevent excessively long OCR jobs consider setting
|
|
||||||
``--tesseract-timeout`` and/or ``--skip-big`` arguments. ``--skip-big``
|
|
||||||
is particularly helpful if your PDFs include documents such as reports
|
|
||||||
on standard page sizes with large images attached - often large images
|
|
||||||
are not worth OCR'ing anyway.
|
|
||||||
|
|
||||||
Commercial alternatives
|
|
||||||
-----------------------
|
|
||||||
|
|
||||||
The author also provides professional services that include OCR and
|
|
||||||
building databases around PDFs, and is happy to provide consultation.
|
|
||||||
|
|
||||||
Abbyy Cloud OCR is viable commercial alternative with a web services
|
|
||||||
API. Amazon Textract, Google Cloud Vision, and Microsoft Azure
|
|
||||||
Computer Vision provide advanced OCR but have less PDF rendering capability.
|
|
||||||
|
|
||||||
Password protection, digital signatures and certification
|
|
||||||
=========================================================
|
|
||||||
|
|
||||||
Password protected PDFs usually have two passwords, and owner and user
|
Password protected PDFs usually have two passwords, and owner and user
|
||||||
password. When the user password is set to empty, PDF readers will open
|
password. When the user password is set to empty, PDF readers will open
|
||||||
the file automatically and marked it as "(SECURED)". While not as
|
the file automatically and mark it as "(SECURED)". Password security can
|
||||||
reliable as a digital signature, this indicates that whoever set the
|
also request certain restrictions on the PDF, but anyone can remove these
|
||||||
password approved of the file at that time. When the user password is
|
restrictions if they have either the owner *or* user password. Passwords
|
||||||
set, the document cannot be viewed without the password.
|
mainly present a barrier for casual users.
|
||||||
|
|
||||||
Either way, OCRmyPDF does not remove passwords from PDFs and exits with
|
OCRmyPDF cannot remove passwords from PDFs. If you want to remove a
|
||||||
an error on encountering them.
|
password from a PDF, you must use other software, such as ``qpdf``.
|
||||||
|
|
||||||
``qpdf`` can remove passwords. If the owner and user password are set, a
|
If the owner and user password are set, a
|
||||||
password is required for ``qpdf``. If only the owner password is set, then the
|
password is required for ``qpdf``. If only the owner password is set, then the
|
||||||
password can be stripped, even if one does not have the owner password.
|
password can be stripped, even if one does not have the owner password. To
|
||||||
|
remove the password from a using QPDF, use:
|
||||||
|
|
||||||
After OCR is applied, password protection is not permitted on PDF/A
|
.. code-block:: bash
|
||||||
documents but the file can be converted to regular PDF.
|
|
||||||
|
qpdf --decrypt --password='abc123' input.pdf no_password.pdf
|
||||||
|
|
||||||
|
Then you can run OCRmyPDF on the file.
|
||||||
|
|
||||||
|
In its default mode, OCRmyPDF generates PDF/A. Passwords may not be set on PDF/A
|
||||||
|
documents. If you want to set a password on the output PDF, you must
|
||||||
|
specify ``--output-type pdf``.
|
||||||
|
|
||||||
|
Signature images
|
||||||
|
================
|
||||||
|
|
||||||
Many programs exist which are capable of inserting an image of someone's
|
Many programs exist which are capable of inserting an image of someone's
|
||||||
signature. On its own, this offers no security guarantees. It is trivial
|
signature. On its own, this offers no security guarantees. It is trivial
|
||||||
to remove the signature image and apply it to other files. This practice
|
to remove the signature image and apply it to other files. This practice
|
||||||
offers no real security.
|
offers no real security.
|
||||||
|
|
||||||
|
Digital signatures
|
||||||
|
==================
|
||||||
|
|
||||||
Important documents can be digitally signed and certified to attest to
|
Important documents can be digitally signed and certified to attest to
|
||||||
their authorship. OCRmyPDF cannot do this. Open source tools such as
|
their authorship, approval or execution of a legal agreement. OCRmyPDF
|
||||||
pdfbox (Java) have this capability as does Adobe Acrobat.
|
will detect signed PDFs and will not modify them, unless the
|
||||||
|
``--invalidate-digital-signatures`` option is used, which will
|
||||||
|
invalidate any signatures. (The signature may still be present in the PDF
|
||||||
|
if opened, but PDF readers will not validate it.)
|
||||||
|
|
||||||
|
A digital signature adds a cryptographic hash of the document to the
|
||||||
|
document, so tamper protection is provided. That also precludes OCRmyPDF
|
||||||
|
from modifying the document and preserving the signature.
|
||||||
|
|
||||||
|
Digital signatures are not the same as a signature image. A digital
|
||||||
|
signature is a cryptographic hash of the document that is encrypted with
|
||||||
|
the author's private key. The signature is decrypted with the author's
|
||||||
|
public key. The public key is usually distributed by a certificate
|
||||||
|
authority. The signature is then verified by the PDF reader. If the
|
||||||
|
document is modified, the signature will be invalidated.
|
||||||
|
|
||||||
|
Certificate-encrypted PDFs
|
||||||
|
==========================
|
||||||
|
|
||||||
|
PDFs can be encrypted with a certificate. This is a more secure form of
|
||||||
|
encryption than a password. The certificate is usually issued by a
|
||||||
|
certificate authority. A certificate is used to encrypt the document using
|
||||||
|
the public key for the benefit of a specific recipient who possesses
|
||||||
|
the private key.
|
||||||
|
|
||||||
|
OCRmyPDF cannot open certificate-encrypted PDFs. If you have the
|
||||||
|
certificate, you can use other PDF software, such as Acrobat, to
|
||||||
|
decrypt the PDF.
|
||||||
+6
-10
@@ -76,20 +76,16 @@ Setuptools plugins
|
|||||||
You can also create a plugin that OCRmyPDF will always automatically load if both are
|
You can also create a plugin that OCRmyPDF will always automatically load if both are
|
||||||
installed in the same virtual environment, using a setuptools entrypoint.
|
installed in the same virtual environment, using a setuptools entrypoint.
|
||||||
|
|
||||||
Your package's ``setup.py`` would need to contain the following, for a plugin
|
Your package's ``pyproject.toml`` would need to contain the following, for a plugin
|
||||||
named ``ocrmypdf-exampleplugin``:
|
named ``ocrmypdf-exampleplugin``:
|
||||||
|
|
||||||
.. code-block:: python
|
.. code-block:: toml
|
||||||
|
|
||||||
# sample ./setup.py file
|
[project]
|
||||||
from setuptools import setup
|
name = "ocrmypdf-exampleplugin"
|
||||||
|
|
||||||
setup(
|
[project.entry-points."ocrmypdf"]
|
||||||
name="ocrmypdf-exampleplugin",
|
exampleplugin = "exampleplugin.pluginmodule"
|
||||||
packages=["exampleplugin"],
|
|
||||||
# the following makes a plugin available to pytest
|
|
||||||
entry_points={"ocrmypdf": ["exampleplugin = exampleplugin.pluginmodule"]},
|
|
||||||
)
|
|
||||||
|
|
||||||
.. code-block:: ini
|
.. code-block:: ini
|
||||||
|
|
||||||
|
|||||||
+129
-1
@@ -28,6 +28,134 @@ tagged yet.
|
|||||||
|
|
||||||
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||||
|
|
||||||
|
v15.0.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Dropped support for Python 3.8.
|
||||||
|
- Dropped support many older dependencies - see ``pyproject.toml`` for details.
|
||||||
|
Generally speaking, Ubuntu 22.04 is our baseline system.
|
||||||
|
- Dropped support 32-bit Windows and Linux. You must use a 64-bit operating system,
|
||||||
|
and 64-bit versions of Python, Tesseract and Ghostscript to use OCRmyPDF. Many of
|
||||||
|
our dependencies are dropping 32-bit support (e.g. Pillow), and we are following
|
||||||
|
suit.
|
||||||
|
- Changed to trusted release for PyPI publishing.
|
||||||
|
- pikepdf memory mapping is enabled again for improved performance, now an issue
|
||||||
|
with pikepdf has been fixed.
|
||||||
|
- ``ocrmypdf.helpers.calculate_downsample`` previously had two variants, one
|
||||||
|
that took a ``PIL.Image`` and one that took a ``tuple[int, int]``. The latter
|
||||||
|
was removed.
|
||||||
|
- The snap version of ocrmypdf is now based on Ubuntu core22.
|
||||||
|
- We now account situations where a small portion of an image on a page reports a
|
||||||
|
high DPI (resolution). Previously, the entire page would be rasterized at the
|
||||||
|
highest resolution, which caused performance problems. Now, the page is rasterized
|
||||||
|
at a resolution based on the average DPI of the page, weighted by the area that
|
||||||
|
each feature occupies. Typically, small areas of high resolution in PDFs are
|
||||||
|
errors or quirks from the repeated use of assets and high resolution is not
|
||||||
|
beneficial. :issue:`1010,1104,1004,1079,1010`
|
||||||
|
- Ghostscript color conversion strategy is now configurable. :issue:`1143`
|
||||||
|
|
||||||
|
v14.4.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Digitally signed PDFs are now detected. If the PDF is signed, OCRmyPDF will
|
||||||
|
refuse to modify it. Previously, only encrypted PDFs were detected, not
|
||||||
|
those that were signed but not encrypted. :issue:`1040`
|
||||||
|
- In addition, ``--invalidate-digital-signatures`` can be used to override the
|
||||||
|
above behavior and modify the PDF anyway. :issue:`1040`
|
||||||
|
- tqdm progress bars replaced with "rich" progress bars. The rich library is
|
||||||
|
a new dependency. Certain APIs that used tqdm are now deprecated and will
|
||||||
|
be removed in the next major release.
|
||||||
|
- Improved integration with GitHub Releases. Thanks to @stumpylog.
|
||||||
|
|
||||||
|
v14.3.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Renamed master branch to main.
|
||||||
|
- Improve PDF rasterization accuracy by using the ``-dPDFSTOPONERROR`` option
|
||||||
|
to Ghostscript. Use ``--continue-on-soft-render-error`` if you want to render
|
||||||
|
the PDF anyway. The plugin specification was adjusted to support this feature;
|
||||||
|
plugin authors may want to adapt PDF rasterizing and rendering
|
||||||
|
plugins. :issue:`1083`
|
||||||
|
- The calculated deskew angle is now recorded in the logged output. :issue:`1101`
|
||||||
|
- Metadata can now be unset by setting a metadata type such as ``--title`` to an
|
||||||
|
empty string. :issue:`1117,1059`
|
||||||
|
- Fixed random order of languages due to use of a set. This may have caused output
|
||||||
|
to vary when multiple languages were set for OCR. :issue:`1113`
|
||||||
|
- Clarified the optimization ratio reported in the log output.
|
||||||
|
- Documentation improvements.
|
||||||
|
|
||||||
|
v14.2.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed :issue:`977`, where images inside Form XObjects were always excluded
|
||||||
|
from image optimization.
|
||||||
|
|
||||||
|
v14.2.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Added ``--tesseract-downsample-above`` to downsample larger images even when
|
||||||
|
they do not exceed Tesseract's internal limits. This can be used to speed
|
||||||
|
up OCR, possibly sacrificing accuracy.
|
||||||
|
- Fixed resampling AttributeError on older Pillow. :issue:`1096`
|
||||||
|
- Removed an error about using Ghostscript on PDFs with that have the /UserUnit
|
||||||
|
feature in use. Previously, Ghostscript would fail to process these PDFs,
|
||||||
|
but in all supported versions it is now supported, so the error is no longer
|
||||||
|
needed.
|
||||||
|
- Improved documentation around installing other language packs for Tesseract.
|
||||||
|
|
||||||
|
v14.1.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Added ``--tesseract-non-ocr-timeout``. This allows using Tesseract's deskew
|
||||||
|
and other non-OCR features while disabling OCR using ``--tesseract-timeout 0``.
|
||||||
|
- Added ``--tesseract-downsample-large-images``. This downsamples larges images
|
||||||
|
that exceed the maximum image size Tesseract can handle. Large images may still
|
||||||
|
take a long time to process, but this allows them to be processed if that
|
||||||
|
is desired.
|
||||||
|
- Fixed :issue:`1082`, an issue with snap packaged building.
|
||||||
|
- Change linter to ruff, fix lint errors, update documentation.
|
||||||
|
|
||||||
|
v14.0.4
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed :issue:`1066, 1075`, an exception when processing certain malformed PDFs.
|
||||||
|
|
||||||
|
v14.0.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed :issue:`1068`, avoid deleting /dev/null when running as root.
|
||||||
|
- Other documentation fixes.
|
||||||
|
|
||||||
|
v14.0.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed :issue:`1052`, an exception on attempting to process certain nonconforming PDFs.
|
||||||
|
- Explicitly documented that Windows 32-bit is no longer supported.
|
||||||
|
- Fixed source installation instructions.
|
||||||
|
- Other documentation fixes.
|
||||||
|
|
||||||
|
v14.0.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed some version checks done with smart version comparison.
|
||||||
|
- Added missing jbig2dec to Docker image.
|
||||||
|
|
||||||
|
v14.0.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Dropped support for Python 3.7.
|
||||||
|
- Dropped support generally speaking, all dependencies older than what Ubuntu 20.04
|
||||||
|
provides.
|
||||||
|
- Ghostscript 9.50 or newer is now required. Shims to support old versions were
|
||||||
|
removed.
|
||||||
|
- Tesseract 4.1.1 or newer is now required. Shims to support old versions were
|
||||||
|
removed.
|
||||||
|
- Docker image now uses Tesseract 5.
|
||||||
|
- Dropped setup.cfg configuration for pyproject.toml.
|
||||||
|
- Removed deprecation exception PdfMergeFailedError.
|
||||||
|
- A few more public domain test files were removed or replaced. We are aiming for
|
||||||
|
100% compliance with SPDX and generally towards simplifying copyright.
|
||||||
|
|
||||||
v13.7.0
|
v13.7.0
|
||||||
=======
|
=======
|
||||||
|
|
||||||
@@ -681,7 +809,7 @@ v10.3.2
|
|||||||
v10.3.1
|
v10.3.1
|
||||||
=======
|
=======
|
||||||
|
|
||||||
- Fixed a number of test suite failures with pdfminer.six older than veresion 20200402.
|
- Fixed a number of test suite failures with pdfminer.six older than version 20200402.
|
||||||
- Enabled support for pdfminer.six 20200720.
|
- Enabled support for pdfminer.six 20200720.
|
||||||
|
|
||||||
v10.3.0
|
v10.3.0
|
||||||
|
|||||||
+9
-2
@@ -2,11 +2,18 @@
|
|||||||
# SPDX-FileCopyrightText: 2016 findingorder <https://github.com/findingorder>
|
# SPDX-FileCopyrightText: 2016 findingorder <https://github.com/findingorder>
|
||||||
# SPDX-License-Identifier: MIT
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
"""Example of using ocrmypdf as a library in a script.
|
||||||
|
|
||||||
|
This script will recursively search a directory for PDF files and run OCR on
|
||||||
|
them. It will log the results. It runs OCR on every file, even if it already
|
||||||
|
has text. OCRmyPDF will detect files that already have text.
|
||||||
|
|
||||||
|
You should edit this script to meet your needs.
|
||||||
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
# This script must be edited to meet your needs.
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
|||||||
@@ -6,52 +6,56 @@ set -o errexit
|
|||||||
|
|
||||||
__ocrmypdf_arguments()
|
__ocrmypdf_arguments()
|
||||||
{
|
{
|
||||||
local arguments="--help (show help message)
|
local arguments="\
|
||||||
--language (language(s) of the file to be OCRed)
|
--help (show help message)
|
||||||
--image-dpi (assume this DPI if input image DPI is unknown)
|
--language (language(s) of the file to be OCRed)
|
||||||
--output-type (select PDF output options)
|
--image-dpi (assume this DPI if input image DPI is unknown)
|
||||||
--sidecar (write OCR to text file)
|
--output-type (select PDF output options)
|
||||||
--version (print program version and exit)
|
--sidecar (write OCR to text file)
|
||||||
--jobs (how many worker processes to use)
|
--version (print program version and exit)
|
||||||
--quiet (suppress INFO messages)
|
--jobs (how many worker processes to use)
|
||||||
--verbose (set verbosity level)
|
--quiet (suppress INFO messages)
|
||||||
--title (set metadata)
|
--verbose (set verbosity level)
|
||||||
--author (set metadata)
|
--title (set metadata)
|
||||||
--subject (set metadata)
|
--author (set metadata)
|
||||||
--keywords (set metadata)
|
--subject (set metadata)
|
||||||
--rotate-pages (rotate pages to correct orientation)
|
--keywords (set metadata)
|
||||||
--remove-background (attempt to remove background from pages)
|
--rotate-pages (rotate pages to correct orientation)
|
||||||
--deskew (fix small horizontal alignment skew)
|
--remove-background (attempt to remove background from pages)
|
||||||
--clean (clean document images before OCR)
|
--deskew (fix small horizontal alignment skew)
|
||||||
--clean-final (clean document images and keep result)
|
--clean (clean document images before OCR)
|
||||||
--unpaper-args (a quoted string of arguments to pass to unpaper)
|
--clean-final (clean document images and keep result)
|
||||||
--oversample (oversample images to this DPI)
|
--unpaper-args (a quoted string of arguments to pass to unpaper)
|
||||||
--remove-vectors (don\'t send vector objects to OCR)
|
--oversample (oversample images to this DPI)
|
||||||
--threshold (threshold images before OCR)
|
--remove-vectors (don\'t send vector objects to OCR)
|
||||||
--force-ocr (OCR documents that already have printable text)
|
--threshold (threshold images before OCR)
|
||||||
--skip-text (skip OCR on any pages that already contain text)
|
--force-ocr (OCR documents that already have printable text)
|
||||||
--redo-ocr (redo OCR on any pages that seem to have OCR already)
|
--skip-text (skip OCR on any pages that already contain text)
|
||||||
--skip-big (skip OCR on pages larger than this many MPixels)
|
--redo-ocr (redo OCR on any pages that seem to have OCR already)
|
||||||
--optimize (select optimization level)
|
--invalidate-digital-signatures (remove digital signatures from PDF)
|
||||||
--jpeg-quality (JPEG quality [0..100])
|
--skip-big (skip OCR on pages larger than this many MPixels)
|
||||||
--png-quality (PNG quality [0..100])
|
--optimize (select optimization level)
|
||||||
--jbig2-lossy (enable lossy JBIG2 (see docs))
|
--jpeg-quality (JPEG quality [0..100])
|
||||||
--pages (apply OCR to only the specified pages)
|
--png-quality (PNG quality [0..100])
|
||||||
--max-image-mpixels (image decompression bomb threshold)
|
--jbig2-lossy (enable lossy JBIG2 (see docs))
|
||||||
--pdf-renderer (select PDF renderer options)
|
--jbig2-threshold (set JBIG2 threshold (see docs))
|
||||||
--rotate-pages-threshold (page rotation confidence)
|
--pages (apply OCR to only the specified pages)
|
||||||
--pdfa-image-compression (set PDF/A image compression options)
|
--max-image-mpixels (image decompression bomb threshold)
|
||||||
--fast-web-view (if file size if above this amount in MB linearize PDF)
|
--pdf-renderer (select PDF renderer options)
|
||||||
--plugin (name of plugin to import)
|
--rotate-pages-threshold (page rotation confidence)
|
||||||
--keep-temporary-files (keep temporary files (debug)
|
--pdfa-image-compression (set PDF/A image compression options)
|
||||||
--tesseract-config (set custom tesseract config file)
|
--fast-web-view (if file size if above this amount in MB linearize PDF)
|
||||||
--tesseract-pagesegmode (set tesseract --psm)
|
--plugin (name of plugin to import)
|
||||||
--tesseract-oem (set tesseract --oem)
|
--keep-temporary-files (keep temporary files (debug)
|
||||||
--tesseract-thresholding (set tesseract image thresholding)
|
--tesseract-config (set custom tesseract config file)
|
||||||
--tesseract-timeout (maximum number of seconds to wait for OCR)
|
--tesseract-pagesegmode (set tesseract --psm)
|
||||||
--user-words (specify location of user words file)
|
--tesseract-oem (set tesseract --oem)
|
||||||
--user-patterns (specify location of user patterns file)
|
--tesseract-thresholding (set tesseract image thresholding)
|
||||||
--no-progress-bar (disable the progress bar)
|
--tesseract-timeout (maximum number of seconds to wait for OCR)
|
||||||
|
--user-words (specify location of user words file)
|
||||||
|
--user-patterns (specify location of user patterns file)
|
||||||
|
--no-progress-bar (disable the progress bar)
|
||||||
|
--color-conversion-strategy (select color conversion strategy)
|
||||||
"
|
"
|
||||||
|
|
||||||
COMPREPLY=( $( compgen -W "$arguments" -- "$cur") )
|
COMPREPLY=( $( compgen -W "$arguments" -- "$cur") )
|
||||||
@@ -191,6 +195,20 @@ sauvola (use Sauvola thresholding)"
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
__ocrmypdf_color-conversion-strategy()
|
||||||
|
{
|
||||||
|
local choices="LeaveColorUnchanged (default)
|
||||||
|
CMYK (convert to CMYK)
|
||||||
|
Gray (convert to grayscale)
|
||||||
|
RGB (convert to RGB)
|
||||||
|
UseDeviceIndependentColor (convert with device independent color)"
|
||||||
|
|
||||||
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
|
# Remove description if only one completion exists
|
||||||
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
__ocrmypdf_check_previous()
|
__ocrmypdf_check_previous()
|
||||||
{
|
{
|
||||||
@@ -250,6 +268,10 @@ __ocrmypdf_check_previous()
|
|||||||
_filedir
|
_filedir
|
||||||
return 0
|
return 0
|
||||||
;;
|
;;
|
||||||
|
--color-conversion-strategy)
|
||||||
|
__ocrmypdf_color-conversion-strategy
|
||||||
|
return 0
|
||||||
|
;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
return 1
|
return 1
|
||||||
|
|||||||
@@ -14,8 +14,9 @@ complete -c ocrmypdf -s i -l clean-final -d "clean document images and keep resu
|
|||||||
complete -c ocrmypdf -l remove-vectors -d "don't send vector objects to OCR"
|
complete -c ocrmypdf -l remove-vectors -d "don't send vector objects to OCR"
|
||||||
|
|
||||||
complete -c ocrmypdf -s f -l force-ocr -d "OCR documents that already have printable text"
|
complete -c ocrmypdf -s f -l force-ocr -d "OCR documents that already have printable text"
|
||||||
complete -c ocrmypdf -s s -l skip-ocr -d "skip OCR on pages that text, otherwise try OCR"
|
complete -c ocrmypdf -s s -l skip-text -d "skip OCR on any pages that already contain text"
|
||||||
complete -c ocrmypdf -l redo-ocr -d "redo OCR on any pages that seem to have OCR already"
|
complete -c ocrmypdf -l redo-ocr -d "redo OCR on any pages that seem to have OCR already"
|
||||||
|
complete -c ocrmypdf -l invalidate-digital-signatures -d "invalidate digital signatures and allow OCR to proceed"
|
||||||
|
|
||||||
complete -c ocrmypdf -s k -l keep-temporary-files -d "keep temporary files (debug)"
|
complete -c ocrmypdf -s k -l keep-temporary-files -d "keep temporary files (debug)"
|
||||||
|
|
||||||
@@ -83,6 +84,7 @@ complete -c ocrmypdf -x -l skip-big -d "skip OCR on pages larger than this many
|
|||||||
complete -c ocrmypdf -x -l jpeg-quality -d "JPEG quality [0..100]"
|
complete -c ocrmypdf -x -l jpeg-quality -d "JPEG quality [0..100]"
|
||||||
complete -c ocrmypdf -x -l png-quality -d "PNG quality [0..100]"
|
complete -c ocrmypdf -x -l png-quality -d "PNG quality [0..100]"
|
||||||
complete -c ocrmypdf -x -l jbig2-lossy -d "enable lossy JBIG2 (see docs)"
|
complete -c ocrmypdf -x -l jbig2-lossy -d "enable lossy JBIG2 (see docs)"
|
||||||
|
complete -c ocrmypdf -x -l jbig2-threshold -d "JBIG2 compression threshold (see docs)"
|
||||||
complete -c ocrmypdf -x -l max-image-mpixels -d "image decompression bomb threshold"
|
complete -c ocrmypdf -x -l max-image-mpixels -d "image decompression bomb threshold"
|
||||||
complete -c ocrmypdf -x -l pages -d "apply OCR to only the specified pages"
|
complete -c ocrmypdf -x -l pages -d "apply OCR to only the specified pages"
|
||||||
complete -c ocrmypdf -x -l tesseract-config -d "set custom tesseract config file"
|
complete -c ocrmypdf -x -l tesseract-config -d "set custom tesseract config file"
|
||||||
@@ -128,4 +130,27 @@ complete -c ocrmypdf -r -l user-words -d "specify location of user words file"
|
|||||||
complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file"
|
complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file"
|
||||||
complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF"
|
complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF"
|
||||||
|
|
||||||
complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf; __fish_complete_suffix .PDF; __fish_complete_suffix .jpg; __fish_complete_suffix .png)"
|
function __fish_ocrmypdf_color_conversion_strategy
|
||||||
|
echo -e "LeaveColorUnchanged\t"(_ "do not convert color spaces (default)")
|
||||||
|
echo -e "CMYK\t"(_ "convert all color spaces to CMYK")
|
||||||
|
echo -e "Gray\t"(_ "convert all color spaces to grayscale")
|
||||||
|
echo -e "RGB\t"(_ "convert all color spaces to RGB")
|
||||||
|
echo -e "UseDeviceIndependentColor\t"(_ "convert all color spaces to ICC-based color spaces")
|
||||||
|
end
|
||||||
|
|
||||||
|
complete -c ocrmypdf -x -l color-conversion-strategy -a '(__fish_ocrmypdf_color_conversion_strategy)' -d "set color conversion strategy"
|
||||||
|
|
||||||
|
function __fish_ocrmypdf_input_file_given
|
||||||
|
set -l tokens (commandline -opc)
|
||||||
|
for token in $tokens
|
||||||
|
if string match -q -r '^-' -- $token
|
||||||
|
continue
|
||||||
|
end
|
||||||
|
if test -f "$token"
|
||||||
|
return 0
|
||||||
|
end
|
||||||
|
end
|
||||||
|
return 1
|
||||||
|
end
|
||||||
|
|
||||||
|
complete -c ocrmypdf -x -n 'not __fish_ocrmypdf_input_file_given' -a "(__fish_complete_suffix .pdf)" -d "input file"
|
||||||
|
|||||||
@@ -1,3 +1,5 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MIT
|
||||||
---
|
---
|
||||||
version: "3.3"
|
version: "3.3"
|
||||||
services:
|
services:
|
||||||
|
|||||||
@@ -1,8 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R Barlow: https://github.com/jbarlow83
|
# SPDX-FileCopyrightText: 2022 James R Barlow: https://github.com/jbarlow83
|
||||||
# SPDX-License-Identifier: MIT
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
"""
|
"""An example of an OCRmyPDF plugin.
|
||||||
An example of an OCRmyPDF plugin.
|
|
||||||
|
|
||||||
This plugin adds two new command line arguments
|
This plugin adds two new command line arguments
|
||||||
--grayscale-ocr: converts the image to grayscale before performing OCR on it
|
--grayscale-ocr: converts the image to grayscale before performing OCR on it
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
<!-- SPDX-License-Identifier: CC-BY-SA-4.0 -->
|
||||||
|
|
||||||
|
To regenerate
|
||||||
|
=============
|
||||||
|
|
||||||
|
Using asciinema and svg-term (`npm install -g svg-term-cli`).
|
||||||
|
|
||||||
|
Create `~/.config/asciinema/config` to disable prompt.
|
||||||
|
|
||||||
|
```
|
||||||
|
[record]
|
||||||
|
|
||||||
|
command = fish --init-command 'alias fish_prompt="echo \>\ "'
|
||||||
|
```
|
||||||
|
|
||||||
|
Run asciinema
|
||||||
|
|
||||||
|
```
|
||||||
|
asciinema rec new_input.cast
|
||||||
|
```
|
||||||
|
|
||||||
|
Re-record faster version with fewer pauses
|
||||||
|
|
||||||
|
```
|
||||||
|
asciinema rec demo.cast -c "asciinema play new_input.cast --speed 2 --idle-time-limit 0.5"
|
||||||
|
```
|
||||||
|
|
||||||
|
Convert to SVG
|
||||||
|
```
|
||||||
|
svg-term --in=misc/screencast/demo.cast --out=misc/screencast/demo.svg --window
|
||||||
|
```
|
||||||
@@ -0,0 +1,65 @@
|
|||||||
|
{"version": 2, "width": 131, "height": 24, "timestamp": 1687247006, "env": {"SHELL": "/usr/bin/fish", "TERM": "xterm-256color"}}
|
||||||
|
[0.103649, "o", "\u001b[?2004h\u001b]7; \u0007"]
|
||||||
|
[0.104223, "o", "\u001b]0;fish \u0007\u001b[30m\u001b(B\u001b[m\r> \u001b[K\r\u001b[C\u001b[C"]
|
||||||
|
[0.604542, "o", "o\r\u001b[3C\b\u001b[38;2;255;0;0mo\r\u001b[3C\u001b[30m\u001b(B\u001b[m\u001b[38;2;85;85;85mcrmypdf multipage.pdf multipage_with_ocr.pdf\r\u001b[3C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[0.679571, "o", "\u001b[38;2;255;0;0mc\u001b[38;2;85;85;85mrmypdf multipage.pdf multipage_with_ocr.pdf\r\u001b[4C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[0.767271, "o", "\u001b[38;2;255;0;0mr\u001b[38;2;85;85;85mmypdf multipage.pdf multipage_with_ocr.pdf\r\u001b[5C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[0.814505, "o", "\u001b[38;2;255;0;0mm\u001b[38;2;85;85;85mypdf multipage.pdf multipage_with_ocr.pdf\r\u001b[6C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[0.938919, "o", "\u001b[38;2;255;0;0my\u001b[38;2;85;85;85mpdf multipage.pdf multipage_with_ocr.pdf\r\u001b[7C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[0.967347, "o", "\u001b[38;2;255;0;0mp\u001b[38;2;85;85;85mdf multipage.pdf multipage_with_ocr.pdf\r\u001b[8C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[1.009954, "o", "\u001b[38;2;255;0;0md\u001b[38;2;85;85;85mf multipage.pdf multipage_with_ocr.pdf\r\u001b[9C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[1.034488, "o", "\u001b[38;2;255;0;0mf\u001b[38;2;85;85;85m multipage.pdf multipage_with_ocr.pdf\r\u001b[10C\u001b[30m\u001b(B\u001b[m\b\b\b\b\b\b\b\b\u001b[38;2;0;95;215mocrmypdf\u001b[38;2;85;85;85m multipage.pdf multipage_with_ocr.pdf\r\u001b[10C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[1.069226, "o", "\u001b[38;2;0;95;215m \u001b[38;2;85;85;85mmultipage.pdf multipage_with_ocr.pdf\r\u001b[11C\u001b[30m\u001b(B\u001b[m\b \u001b[38;2;85;85;85mmultipage.pdf multipage_with_ocr.pdf\r\u001b[11C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[1.569682, "o", "-\u001b[K\r\u001b[12C\u001b[38;2;85;85;85m-version\r\u001b[12C\u001b[30m\u001b(B\u001b[m\b\u001b[38;2;0;175;255m-\u001b[38;2;85;85;85m-version\r\u001b[12C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[1.642096, "o", "\u001b[38;2;0;175;255m-\u001b[38;2;85;85;85mversion\r\u001b[13C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[1.71793, "o", "\u001b[38;2;0;175;255ms\u001b[30m\u001b(B\u001b[m\u001b[K\r\u001b[14C"]
|
||||||
|
[1.771483, "o", "\u001b[38;2;0;175;255mk\r\u001b[15C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[1.864664, "o", "\u001b[38;2;0;175;255mi\r\u001b[16C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[1.876085, "o", "\u001b[38;2;0;175;255mp\r\u001b[17C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[2.092979, "o", "\u001b[38;2;0;175;255m-\r\u001b[18C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[2.138821, "o", "\u001b[38;2;0;175;255mt\r\u001b[19C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[2.18017, "o", "\u001b[38;2;0;175;255me\r\u001b[20C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[2.268222, "o", "\u001b[38;2;0;175;255mx\r\u001b[21C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[2.277031, "o", "\u001b[38;2;0;175;255mt\r\u001b[22C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[2.322469, "o", "\u001b[38;2;0;175;255m \r\u001b[23C\u001b[30m\u001b(B\u001b[m\b \r\u001b[23C"]
|
||||||
|
[2.824696, "o", "m\r\u001b[24C\b\u001b[38;2;0;175;255m\u001b[4mm\r\u001b[24C\u001b[30m\u001b(B\u001b[m\u001b[38;2;85;85;85masks.pdf \r\u001b[24C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[2.923234, "o", "\u001b[38;2;0;175;255m\u001b[4mu\u001b[30m\u001b(B\u001b[m\u001b[K\r\u001b[25C\u001b[38;2;85;85;85mltipage.pdf \r\u001b[25C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[2.960685, "o", "\u001b[38;2;0;175;255m\u001b[4ml\u001b[38;2;85;85;85m\u001b[24mtipage.pdf \r\u001b[26C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[3.03365, "o", "\u001b[38;2;0;175;255m\u001b[4mt\u001b[38;2;85;85;85m\u001b[24mipage.pdf \r\u001b[27C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[3.479338, "o", "\u001b[38;2;0;175;255m\u001b[4mipage.pdf \r\u001b[37C\u001b[30m\u001b(B\u001b[m\b \r\u001b[37C"]
|
||||||
|
[3.754818, "o", "m\r\u001b[38C\b\u001b[38;2;0;175;255m\u001b[4mm\r\u001b[38C\u001b[30m\u001b(B\u001b[m\u001b[38;2;85;85;85masks.pdf \r\u001b[38C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[3.873318, "o", "\u001b[38;2;0;175;255m\u001b[4mu\u001b[30m\u001b(B\u001b[m\u001b[K\r\u001b[39C\u001b[38;2;85;85;85mltipage.pdf \r\u001b[39C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[3.926829, "o", "\u001b[38;2;0;175;255m\u001b[4ml\u001b[38;2;85;85;85m\u001b[24mtipage.pdf \r\u001b[40C\u001b[30m\u001b(B\u001b[m"]
|
||||||
|
[4.272251, "o", "\u001b[38;2;0;175;255m\u001b[4mtipage.pdf \r\u001b[51C\u001b[30m\u001b(B\u001b[m\b \r\u001b[51C"]
|
||||||
|
[4.343464, "o", "\r\u001b[50C"]
|
||||||
|
[4.416286, "o", "\r\u001b[49C"]
|
||||||
|
[4.490574, "o", "\r\u001b[48C"]
|
||||||
|
[4.564115, "o", "\r\u001b[47C"]
|
||||||
|
[4.630398, "o", "\r\u001b[46C"]
|
||||||
|
[4.76825, "o", "\u001b[38;2;0;175;255m\u001b[4m_.pd\u001b[30m\u001b(B\u001b[mf \r\u001b[47C\u001b[10D\u001b[38;2;0;175;255mmultipage_.pdf\u001b[30m\u001b(B\u001b[m \r\u001b[47C"]
|
||||||
|
[5.012506, "o", "\u001b[38;2;0;175;255mo.pd\u001b[30m\u001b(B\u001b[mf \r\u001b[48C\u001b[3C\u001b[38;2;0;175;255mf\u001b[30m\u001b(B\u001b[m \r\u001b[48C"]
|
||||||
|
[5.053615, "o", "\u001b[38;2;0;175;255mc.pd\u001b[30m\u001b(B\u001b[mf \r\u001b[49C\u001b[3C\u001b[38;2;0;175;255mf\u001b[30m\u001b(B\u001b[m \r\u001b[49C"]
|
||||||
|
[5.103957, "o", "\u001b[38;2;0;175;255mr.pd\u001b[30m\u001b(B\u001b[mf \r\u001b[50C\u001b[3C\u001b[38;2;0;175;255mf\u001b[30m\u001b(B\u001b[m \r\u001b[50C"]
|
||||||
|
[5.226183, "o", "\r\u001b[55C"]
|
||||||
|
[5.728321, "o", "\r\n\u001b[30m\u001b(B\u001b[m\u001b[?2004l\u001b]0;ocrmypdf --skip-text multipage.pdf multipage_ocr.pdf /home/jb/src/ocrmypdf/tests/resources\u0007\u001b[30m\u001b(B\u001b[m\r"]
|
||||||
|
[5.801032, "o", "\rScanning contents: 0%| | 0/6 [00:00<?, ?page/s]"]
|
||||||
|
[5.802664, "o", "\rScanning contents: 100%|█████████████████████████████████████████████████████████████████████████| 6/6 [00:00<00:00, 1270.68page/s]\r\n"]
|
||||||
|
[5.802747, "o", "Start processing 6 pages concurrently\r\n"]
|
||||||
|
[5.803488, "o", "\rOCR: 0%| | 0.0/6.0 [00:00<?, ?page/s]"]
|
||||||
|
[5.804896, "o", "\r \r 4 skipping all processing on this page\r\n\rOCR: 0%| | 0.0/6.0 [00:00<?, ?page/s]"]
|
||||||
|
[5.896969, "o", "\rOCR: 25%|█████████████████████▎ | 1.5/6.0 [00:00<00:00, 8.12page/s]"]
|
||||||
|
[6.170021, "o", "\rOCR: 42%|███████████████████████████████████▍ | 2.5/6.0 [00:00<00:01, 3.05page/s]"]
|
||||||
|
[6.292338, "o", "\rOCR: 58%|█████████████████████████████████████████████████▌ | 3.5/6.0 [00:00<00:00, 3.39page/s]"]
|
||||||
|
[6.586017, "o", "\rOCR: 75%|███████████████████████████████████████████████████████████████▊ | 4.5/6.0 [00:01<00:00, 2.49page/s]"]
|
||||||
|
[7.087058, "o", "\rOCR: 92%|█████████████████████████████████████████████████████████████████████████████▉ | 5.5/6.0 [00:06<00:00, 1.98s/page]\rOCR: 100%|█████████████████████████████████████████████████████████████████████████████████████| 6.0/6.0 [00:06<00:00, 1.09s/page]\r\nPostprocessing...\r\n"]
|
||||||
|
[7.104927, "o", "\rPDF/A conversion: 0%| | 0/6 [00:00<?, ?page/s]"]
|
||||||
|
[7.607392, "o", "\rPDF/A conversion: 50%|██████████████████████████████████████ | 3/6 [00:01<00:01, 1.61page/s]"]
|
||||||
|
[7.653781, "o", "\rPDF/A conversion: 83%|███████████████████████████████████████████████████████████████▎ | 5/6 [00:01<00:00, 2.90page/s]"]
|
||||||
|
[7.774532, "o", "\rPDF/A conversion: 100%|████████████████████████████████████████████████████████████████████████████| 6/6 [00:02<00:00, 2.71page/s]\r\n"]
|
||||||
|
[7.778252, "o", "\u001b[33mSome input metadata could not be copied because it is not permitted in PDF/A. You may wish to examine the output PDF's XMP metadata.\u001b[0m\r\n"]
|
||||||
|
[8.280789, "o", "\rRecompressing JPEGs: 0image [00:00, ?image/s]\rRecompressing JPEGs: 0image [00:00, ?image/s]\r\n\rDeflating JPEGs: 0%| | 0/4 [00:00<?, ?image/s]\rDeflating JPEGs: 100%|███████████████████████████████████████████████████████████████████████████| 4/4 [00:00<00:00, 238.28image/s]\r\n"]
|
||||||
|
[8.28149, "o", "\rJBIG2: 0item [00:00, ?item/s]\rJBIG2: 0item [00:00, ?item/s]\r\n"]
|
||||||
|
[8.289998, "o", "Image optimization ratio: 1.01 savings: 1.3%\r\nTotal file size ratio: 1.02 savings: 1.6%\r\n"]
|
||||||
|
[8.291209, "o", "Output file is a PDF/A-2B (as expected)\r\n"]
|
||||||
|
[8.361316, "o", "\u001b[2m⏎\u001b(B\u001b[m \r⏎ \r\u001b[K\u001b[?2004h\u001b]0;fish /home/jb/src/ocrmypdf/tests/resources\u0007\u001b[30m\u001b(B\u001b[m> \u001b[K\r\u001b[C\u001b[C"]
|
||||||
|
[8.862206, "o", "\r\n\u001b[30m\u001b(B\u001b[m\u001b[30m\u001b(B\u001b[m\u001b[?2004l"]
|
||||||
File diff suppressed because one or more lines are too long
|
After Width: | Height: | Size: 29 KiB |
+3
-4
@@ -2,6 +2,8 @@
|
|||||||
# SPDX-FileCopyrightText: 2017 Enantiomerie
|
# SPDX-FileCopyrightText: 2017 Enantiomerie
|
||||||
# SPDX-License-Identifier: MIT
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
"""Example OCRmyPDF for Synology NAS."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
# This script must be edited to meet your needs.
|
# This script must be edited to meet your needs.
|
||||||
@@ -25,10 +27,7 @@ logging.basicConfig(
|
|||||||
filemode='w',
|
filemode='w',
|
||||||
)
|
)
|
||||||
|
|
||||||
if len(sys.argv) > 1:
|
start_dir = sys.argv[1] if len(sys.argv) > 1 else '.'
|
||||||
start_dir = sys.argv[1]
|
|
||||||
else:
|
|
||||||
start_dir = '.'
|
|
||||||
|
|
||||||
for dir_name, _subdirs, file_list in os.walk(start_dir):
|
for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||||
logging.info(dir_name)
|
logging.info(dir_name)
|
||||||
|
|||||||
+7
-1
@@ -3,6 +3,8 @@
|
|||||||
# SPDX-FileCopyrightText: 2020 James R Barlow <https://github.com/jbarlow83>
|
# SPDX-FileCopyrightText: 2020 James R Barlow <https://github.com/jbarlow83>
|
||||||
# SPDX-License-Identifier: MIT
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
"""Watch a directory for new PDFs and OCR them."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
@@ -38,6 +40,7 @@ DESKEW = getenv_bool('OCR_DESKEW')
|
|||||||
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
||||||
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
||||||
USE_POLLING = getenv_bool('OCR_USE_POLLING')
|
USE_POLLING = getenv_bool('OCR_USE_POLLING')
|
||||||
|
RETRIES_LOADING_FILE = int(os.getenv('OCR_RETRIES_LOADING_FILE', '5'))
|
||||||
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
|
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
|
||||||
PATTERNS = ['*.pdf', '*.PDF']
|
PATTERNS = ['*.pdf', '*.PDF']
|
||||||
|
|
||||||
@@ -64,7 +67,7 @@ def wait_for_file_ready(file_path):
|
|||||||
# watchdog event before the file is actually fully on disk, causing
|
# watchdog event before the file is actually fully on disk, causing
|
||||||
# pikepdf to fail.
|
# pikepdf to fail.
|
||||||
|
|
||||||
retries = 5
|
retries = RETRIES_LOADING_FILE
|
||||||
while retries:
|
while retries:
|
||||||
try:
|
try:
|
||||||
pdf = pikepdf.open(file_path)
|
pdf = pikepdf.open(file_path)
|
||||||
@@ -103,6 +106,8 @@ def execute_ocrmypdf(file_path):
|
|||||||
elif ON_SUCCESS_ARCHIVE:
|
elif ON_SUCCESS_ARCHIVE:
|
||||||
log.info(f'OCR is done. Archiving {file_path.name} to {ARCHIVE_DIRECTORY}')
|
log.info(f'OCR is done. Archiving {file_path.name} to {ARCHIVE_DIRECTORY}')
|
||||||
shutil.move(file_path, f'{ARCHIVE_DIRECTORY}/{file_path.name}')
|
shutil.move(file_path, f'{ARCHIVE_DIRECTORY}/{file_path.name}')
|
||||||
|
else:
|
||||||
|
log.info('OCR is done')
|
||||||
else:
|
else:
|
||||||
log.info('OCR is done')
|
log.info('OCR is done')
|
||||||
|
|
||||||
@@ -140,6 +145,7 @@ def main():
|
|||||||
f"DESKEW: {DESKEW}\n"
|
f"DESKEW: {DESKEW}\n"
|
||||||
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
||||||
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
||||||
|
f"RETRIES_LOADING_FILE: {RETRIES_LOADING_FILE}\n"
|
||||||
f"USE_POLLING: {USE_POLLING}\n"
|
f"USE_POLLING: {USE_POLLING}\n"
|
||||||
f"LOGLEVEL: {LOGLEVEL}"
|
f"LOGLEVEL: {LOGLEVEL}"
|
||||||
)
|
)
|
||||||
|
|||||||
+3
-3
@@ -2,7 +2,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2019 James R. Barlow
|
# SPDX-FileCopyrightText: 2019 James R. Barlow
|
||||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||||
|
|
||||||
"""This is a simple web service/HTTP wrapper for OCRmyPDF
|
"""This is a simple web service/HTTP wrapper for OCRmyPDF.
|
||||||
|
|
||||||
This may be more convenient than the command line tool for some Docker users.
|
This may be more convenient than the command line tool for some Docker users.
|
||||||
Note that OCRmyPDF uses Ghostscript, which is licensed under AGPLv3+. While
|
Note that OCRmyPDF uses Ghostscript, which is licensed under AGPLv3+. While
|
||||||
@@ -15,7 +15,7 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import os
|
import os
|
||||||
import shlex
|
import shlex
|
||||||
from subprocess import PIPE, run
|
from subprocess import run
|
||||||
from tempfile import TemporaryDirectory
|
from tempfile import TemporaryDirectory
|
||||||
|
|
||||||
from flask import Flask, Response, request, send_from_directory
|
from flask import Flask, Response, request, send_from_directory
|
||||||
@@ -48,7 +48,7 @@ def do_ocrmypdf(file):
|
|||||||
return Response("--sidecar not supported", 501, mimetype='text/plain')
|
return Response("--sidecar not supported", 501, mimetype='text/plain')
|
||||||
|
|
||||||
ocrmypdf_args = ["ocrmypdf", *cmd_args, up_file, down_file]
|
ocrmypdf_args = ["ocrmypdf", *cmd_args, up_file, down_file]
|
||||||
proc = run(ocrmypdf_args, capture_output=True, encoding="utf-8")
|
proc = run(ocrmypdf_args, capture_output=True, encoding="utf-8", check=False)
|
||||||
if proc.returncode != 0:
|
if proc.returncode != 0:
|
||||||
stderr = proc.stderr
|
stderr = proc.stderr
|
||||||
return Response(stderr, 400, mimetype='text/plain')
|
return Response(stderr, 400, mimetype='text/plain')
|
||||||
|
|||||||
+101
-33
@@ -1,16 +1,87 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
[build-system]
|
[build-system]
|
||||||
requires = [
|
requires = ["setuptools >= 61", "setuptools_scm[toml] >= 7.0.5", "wheel"]
|
||||||
"setuptools >= 52",
|
|
||||||
"setuptools_scm[toml] >= 7.0.5",
|
|
||||||
"wheel"
|
|
||||||
]
|
|
||||||
build-backend = "setuptools.build_meta"
|
build-backend = "setuptools.build_meta"
|
||||||
|
|
||||||
|
[project]
|
||||||
|
name = "ocrmypdf"
|
||||||
|
dynamic = ["version"]
|
||||||
|
description = "OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched"
|
||||||
|
readme = "README.md"
|
||||||
|
license = { text = "MPL-2.0" }
|
||||||
|
requires-python = ">=3.9"
|
||||||
|
dependencies = [
|
||||||
|
"Pillow>=9.0.1",
|
||||||
|
"deprecation>=2.1.0",
|
||||||
|
"img2pdf>=0.4.4",
|
||||||
|
"packaging>=20",
|
||||||
|
"pdfminer.six>=20220319",
|
||||||
|
"pikepdf>=8",
|
||||||
|
"pluggy>=0.13.0",
|
||||||
|
"reportlab>=3.6.8",
|
||||||
|
"rich>=13",
|
||||||
|
"typing-extensions>=4;python_version<'3.10'",
|
||||||
|
]
|
||||||
|
authors = [{ name = "James R. Barlow", email = "james@purplerock.ca" }]
|
||||||
|
classifiers = [
|
||||||
|
"Development Status :: 5 - Production/Stable",
|
||||||
|
"Environment :: Console",
|
||||||
|
"Intended Audience :: End Users/Desktop",
|
||||||
|
"Intended Audience :: Science/Research",
|
||||||
|
"Intended Audience :: System Administrators",
|
||||||
|
"License :: OSI Approved :: Mozilla Public License 2.0 (MPL 2.0)",
|
||||||
|
"Operating System :: MacOS",
|
||||||
|
"Operating System :: Microsoft :: Windows",
|
||||||
|
"Operating System :: POSIX",
|
||||||
|
"Operating System :: POSIX :: BSD",
|
||||||
|
"Operating System :: POSIX :: Linux",
|
||||||
|
"Programming Language :: Python :: 3",
|
||||||
|
"Topic :: Scientific/Engineering :: Image Recognition",
|
||||||
|
"Topic :: Text Processing :: Indexing",
|
||||||
|
"Topic :: Text Processing :: Linguistic",
|
||||||
|
]
|
||||||
|
keywords = ["PDF", "OCR", "optical character recognition", "PDF/A", "scanning"]
|
||||||
|
|
||||||
|
[project.urls]
|
||||||
|
Documentation = "https://ocrmypdf.readthedocs.io/"
|
||||||
|
Source = "https://github.com/ocrmypdf/OCRmyPDF"
|
||||||
|
Tracker = "https://github.com/ocrmypdf/OCRmyPDF/issues"
|
||||||
|
|
||||||
|
[project.optional-dependencies]
|
||||||
|
docs = ["sphinx", "sphinx-issues", "sphinx-rtd-theme"]
|
||||||
|
extended_test = ["PyMuPDF==1.19.1"]
|
||||||
|
test = [
|
||||||
|
"coverage[toml]>=6.2",
|
||||||
|
"hypothesis>=6.36.0",
|
||||||
|
"pytest>=6.2.5",
|
||||||
|
"pytest-cov>=3.0.0",
|
||||||
|
"pytest-xdist>=2.5.0",
|
||||||
|
"python-xmp-toolkit==2.0.1", # also requires apt-get install libexempi3
|
||||||
|
"types-Pillow",
|
||||||
|
"types-humanfriendly",
|
||||||
|
]
|
||||||
|
watcher = ["watchdog>=1.0.2"]
|
||||||
|
webservice = ["Flask>=2.0.1"]
|
||||||
|
|
||||||
|
[project.scripts]
|
||||||
|
ocrmypdf = "ocrmypdf.__main__:run"
|
||||||
|
|
||||||
|
[tool.setuptools.package-data]
|
||||||
|
ocrmypdf = ["data/sRGB.icc", "py.typed"]
|
||||||
|
|
||||||
|
[tool.setuptools.packages.find]
|
||||||
|
where = ["src"]
|
||||||
|
namespaces = false
|
||||||
|
|
||||||
[tool.setuptools_scm]
|
[tool.setuptools_scm]
|
||||||
|
|
||||||
|
[tool.distutils.bdist_wheel]
|
||||||
|
python-tag = "py38"
|
||||||
|
|
||||||
[tool.black]
|
[tool.black]
|
||||||
line-length = 88
|
line-length = 88
|
||||||
target-version = ["py37", "py38"]
|
target-version = ["py38", "py39", "py310", "py311"]
|
||||||
skip-string-normalization = true
|
skip-string-normalization = true
|
||||||
include = '\.pyi?$'
|
include = '\.pyi?$'
|
||||||
exclude = '''
|
exclude = '''
|
||||||
@@ -51,28 +122,7 @@ exclude_lines = [
|
|||||||
"if 0:",
|
"if 0:",
|
||||||
"if False:",
|
"if False:",
|
||||||
"if __name__ == .__main__.:",
|
"if __name__ == .__main__.:",
|
||||||
"if TYPE_CHECKING:"
|
"if TYPE_CHECKING:",
|
||||||
]
|
|
||||||
|
|
||||||
[tool.isort]
|
|
||||||
profile = "black"
|
|
||||||
known_first_party = "ocrmypdf"
|
|
||||||
known_third_party = [
|
|
||||||
"PIL",
|
|
||||||
"flask",
|
|
||||||
"img2pdf",
|
|
||||||
"ocrmypdf",
|
|
||||||
"pdfminer",
|
|
||||||
"pikepdf",
|
|
||||||
"pkg_resources",
|
|
||||||
"pluggy",
|
|
||||||
"pytest",
|
|
||||||
"reportlab",
|
|
||||||
"setuptools",
|
|
||||||
"sphinx_rtd_theme",
|
|
||||||
"tqdm",
|
|
||||||
"watchdog",
|
|
||||||
"werkzeug"
|
|
||||||
]
|
]
|
||||||
|
|
||||||
[tool.pytest.ini_options]
|
[tool.pytest.ini_options]
|
||||||
@@ -95,11 +145,29 @@ module = [
|
|||||||
'reportlab.*',
|
'reportlab.*',
|
||||||
'fitz',
|
'fitz',
|
||||||
'libxmp.utils',
|
'libxmp.utils',
|
||||||
'importlib_metadata'
|
|
||||||
]
|
]
|
||||||
ignore_missing_imports = true
|
ignore_missing_imports = true
|
||||||
|
|
||||||
[tool.pylint.basic]
|
[tool.ruff]
|
||||||
good-names = ["i", "j", "k", "ex", "Run", "_", "e", "p", "im", "w", "h", "m", "x", "y", "a", "b", "fp", "n", "f", "s", "v", "q", "dx", "dy"]
|
select = [
|
||||||
logging-format-style = "old"
|
"D", # pydocstyle
|
||||||
disable = ["raw-checker-failed", "bad-inline-option", "locally-disabled", "file-ignored", "suppressed-message", "useless-suppression", "deprecated-pragma", "use-symbolic-message-instead", "logging-fstring-interpolation", "missing-function-docstring", "too-few-public-methods"]
|
"E", # pycodestyle
|
||||||
|
"W", # pycodestyle
|
||||||
|
"F", # pyflakes
|
||||||
|
"I001", # isort
|
||||||
|
"UP", # pyupgrade
|
||||||
|
]
|
||||||
|
target-version = "py38"
|
||||||
|
|
||||||
|
[tool.ruff.isort]
|
||||||
|
known-first-party = ["ocrmypdf"]
|
||||||
|
required-imports = ["from __future__ import annotations"]
|
||||||
|
|
||||||
|
[tool.ruff.pydocstyle]
|
||||||
|
convention = "google"
|
||||||
|
|
||||||
|
[tool.ruff.per-file-ignores]
|
||||||
|
"docs/conf.py" = ["D100", "D101", "D105"]
|
||||||
|
"tests/*.py" = ["D100", "D101", "D102", "D103", "D105"]
|
||||||
|
"misc/*.py" = ["D103", "D101", "D102"]
|
||||||
|
"src/ocrmypdf/builtin_plugins/*.py" = ["D103", "D102", "D105"]
|
||||||
|
|||||||
@@ -1,116 +0,0 @@
|
|||||||
[metadata]
|
|
||||||
name = ocrmypdf
|
|
||||||
description = OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched
|
|
||||||
long_description = file: README.md
|
|
||||||
long_description_content_type = text/markdown
|
|
||||||
url = https://github.com/ocrmypdf/OCRmyPDF
|
|
||||||
author = James R. Barlow
|
|
||||||
author_email = james@purplerock.ca
|
|
||||||
license = MPL-2.0
|
|
||||||
license_file = LICENSE
|
|
||||||
license_files =
|
|
||||||
LICENSE
|
|
||||||
classifiers =
|
|
||||||
Development Status :: 5 - Production/Stable
|
|
||||||
Environment :: Console
|
|
||||||
Intended Audience :: End Users/Desktop
|
|
||||||
Intended Audience :: Science/Research
|
|
||||||
Intended Audience :: System Administrators
|
|
||||||
License :: OSI Approved :: Mozilla Public License 2.0 (MPL 2.0)
|
|
||||||
Operating System :: MacOS :: MacOS X
|
|
||||||
Operating System :: Microsoft :: Windows :: Windows 10
|
|
||||||
Operating System :: POSIX
|
|
||||||
Operating System :: POSIX :: BSD
|
|
||||||
Operating System :: POSIX :: Linux
|
|
||||||
Programming Language :: Python :: 3
|
|
||||||
Programming Language :: Python :: 3 :: Only
|
|
||||||
Programming Language :: Python :: 3.7
|
|
||||||
Programming Language :: Python :: 3.8
|
|
||||||
Programming Language :: Python :: 3.9
|
|
||||||
Programming Language :: Python :: 3.10
|
|
||||||
Topic :: Scientific/Engineering :: Image Recognition
|
|
||||||
Topic :: Text Processing :: Indexing
|
|
||||||
Topic :: Text Processing :: Linguistic
|
|
||||||
keywords =
|
|
||||||
PDF
|
|
||||||
OCR
|
|
||||||
optical character recognition
|
|
||||||
PDF/A
|
|
||||||
scanning
|
|
||||||
project_urls =
|
|
||||||
Documentation = https://ocrmypdf.readthedocs.io/
|
|
||||||
Source = https://github.com/ocrmypdf/OCRmyPDF
|
|
||||||
Tracker = https://github.com/ocrmypdf/OCRmyPDF/issues
|
|
||||||
|
|
||||||
[options]
|
|
||||||
packages = find:
|
|
||||||
install_requires =
|
|
||||||
Pillow>=8.2.0
|
|
||||||
coloredlogs>=14.0 # strictly optional
|
|
||||||
img2pdf>=0.3.0 # pure Python
|
|
||||||
packaging>=20
|
|
||||||
pdfminer.six!=20200720,>=20191110
|
|
||||||
pikepdf!=5.0.0,>=4.0.0
|
|
||||||
pluggy>=0.13.0
|
|
||||||
reportlab>=3.5.66
|
|
||||||
tqdm>=4
|
|
||||||
importlib-metadata>=4;python_version<'3.8' # until Python 3.8
|
|
||||||
importlib-resources>=5;python_version<'3.9' # until Python 3.9
|
|
||||||
typing-extensions>=4;python_version<'3.10'
|
|
||||||
python_requires = >=3.7
|
|
||||||
include_package_data = True
|
|
||||||
package_dir =
|
|
||||||
=src
|
|
||||||
platforms = any
|
|
||||||
setup_requires =
|
|
||||||
setuptools-scm
|
|
||||||
setuptools-scm-git-archive
|
|
||||||
zip_safe = False
|
|
||||||
|
|
||||||
[options.packages.find]
|
|
||||||
where = src
|
|
||||||
|
|
||||||
[options.entry_points]
|
|
||||||
console_scripts =
|
|
||||||
ocrmypdf = ocrmypdf.__main__:run
|
|
||||||
|
|
||||||
[options.extras_require]
|
|
||||||
docs =
|
|
||||||
sphinx
|
|
||||||
sphinx-issues
|
|
||||||
sphinx-rtd-theme
|
|
||||||
extended_test =
|
|
||||||
PyMuPDF==1.19.1
|
|
||||||
test =
|
|
||||||
coverage[toml]>=5
|
|
||||||
pytest>=6.0.0
|
|
||||||
pytest-cov>=2.11.1
|
|
||||||
pytest-xdist>=2.2.0
|
|
||||||
python-xmp-toolkit==2.0.1 # also requires apt-get install libexempi3
|
|
||||||
types-Pillow
|
|
||||||
types-humanfriendly
|
|
||||||
watcher =
|
|
||||||
watchdog>=1.0.2
|
|
||||||
webservice =
|
|
||||||
Flask>=1
|
|
||||||
|
|
||||||
[options.package_data]
|
|
||||||
ocrmypdf =
|
|
||||||
data/sRGB.icc
|
|
||||||
py.typed
|
|
||||||
|
|
||||||
[bdist_wheel]
|
|
||||||
python-tag = py37
|
|
||||||
|
|
||||||
[aliases]
|
|
||||||
test = pytest
|
|
||||||
|
|
||||||
[check-manifest]
|
|
||||||
ignore =
|
|
||||||
.github
|
|
||||||
|
|
||||||
[flake8]
|
|
||||||
ignore = D203,F401,W503,E501,E203,F841
|
|
||||||
exclude = .git,__pycache__,docs/conf.py,build,dist,.venv,.venvpp,.eggs,tmp,src/ocrmypdf/lib/
|
|
||||||
max-complexity = 10
|
|
||||||
max-line-length = 100
|
|
||||||
@@ -1,10 +0,0 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
|
||||||
|
|
||||||
"""setup.py to support older setuptools and pip."""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
from setuptools import setup
|
|
||||||
|
|
||||||
setup()
|
|
||||||
+27
-5
@@ -1,8 +1,13 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2022 Alexander Langanke
|
||||||
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
|
# SPDX-FileCopyrightText: 2023 林博仁(Buo-ren, Lin) <Buo.Ren.Lin@gmail.com>
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
name: ocrmypdf
|
name: ocrmypdf
|
||||||
title: OCRmyPDF
|
title: OCRmyPDF
|
||||||
base: core20
|
base: core22
|
||||||
version: git
|
version: git
|
||||||
summary: OCRmyPDF adds optical character recognition (OCR) to PDFs
|
summary: OCRmyPDF adds a searchable text layer to scanned PDF files
|
||||||
description: OCRmyPDF packaged for snap
|
description: OCRmyPDF packaged for snap
|
||||||
grade: stable
|
grade: stable
|
||||||
confinement: strict
|
confinement: strict
|
||||||
@@ -13,8 +18,8 @@ architectures: [amd64]
|
|||||||
|
|
||||||
environment:
|
environment:
|
||||||
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
||||||
GS_LIB: $SNAP/usr/share/ghostscript/9.50/Resource/Init
|
GS_LIB: $SNAP/usr/share/ghostscript/9.55/Resource/Init
|
||||||
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.50/Resource/Font
|
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.55/Resource/Font
|
||||||
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
||||||
|
|
||||||
apps:
|
apps:
|
||||||
@@ -41,9 +46,19 @@ parts:
|
|||||||
stage-packages:
|
stage-packages:
|
||||||
- lib32stdc++6
|
- lib32stdc++6
|
||||||
|
|
||||||
|
jbig2enc:
|
||||||
|
plugin: autotools
|
||||||
|
source: https://github.com/agl/jbig2enc.git
|
||||||
|
source-tag: "0.29"
|
||||||
|
build-packages:
|
||||||
|
- libleptonica-dev
|
||||||
|
|
||||||
ocrmypdf:
|
ocrmypdf:
|
||||||
plugin: python
|
plugin: python
|
||||||
source: https://github.com/ocrmypdf/OCRmyPDF.git
|
source: .
|
||||||
|
|
||||||
|
build-packages:
|
||||||
|
- python3-pip
|
||||||
|
|
||||||
stage-packages:
|
stage-packages:
|
||||||
- ghostscript
|
- ghostscript
|
||||||
@@ -66,7 +81,14 @@ parts:
|
|||||||
- setuptools
|
- setuptools
|
||||||
- tqdm
|
- tqdm
|
||||||
- pipe
|
- pipe
|
||||||
|
- wheel
|
||||||
|
|
||||||
override-build: |
|
override-build: |
|
||||||
|
pip3 install --user dephell[full]
|
||||||
|
$HOME/.local/bin/dephell deps convert \
|
||||||
|
--from-path pyproject.toml \
|
||||||
|
--from-format pyproject \
|
||||||
|
--to-path setup.py \
|
||||||
|
--to-format setuppy
|
||||||
snapcraftctl build
|
snapcraftctl build
|
||||||
ln -sf ../usr/lib/libsnapcraft-preload.so $SNAPCRAFT_PART_INSTALL/lib/libsnapcraft-preload.so
|
ln -sf ../usr/lib/libsnapcraft-preload.so $SNAPCRAFT_PART_INSTALL/lib/libsnapcraft-preload.so
|
||||||
|
|||||||
@@ -1,3 +1,6 @@
|
|||||||
|
<!-- SPDX-FileCopyrightText: 2022 James R. Barlow -->
|
||||||
|
<!-- SPDX-License-Identifier: CC-BY-SA-4.0 -->
|
||||||
|
|
||||||
# Release checklist
|
# Release checklist
|
||||||
|
|
||||||
## Patch release
|
## Patch release
|
||||||
@@ -14,11 +17,11 @@
|
|||||||
|
|
||||||
- Check README.md
|
- Check README.md
|
||||||
|
|
||||||
- Check setup.py
|
- Check pyproject.toml
|
||||||
|
|
||||||
- Are classifiers up to date?
|
- Are classifiers up to date?
|
||||||
- Is `python_requires` correct?
|
- Is `python_requires` correct?
|
||||||
- Python 3.6 is EOL on December 2021-12. Could drop support then.
|
- Is it to drop support for older Pythons?
|
||||||
- Can we tighten any `install_requires` dependencies?
|
- Can we tighten any `install_requires` dependencies?
|
||||||
|
|
||||||
- Search for old version shims we can remove
|
- Search for old version shims we can remove
|
||||||
|
|||||||
@@ -21,7 +21,6 @@ from ocrmypdf.exceptions import (
|
|||||||
InputFileError,
|
InputFileError,
|
||||||
MissingDependencyError,
|
MissingDependencyError,
|
||||||
OutputFileAccessError,
|
OutputFileAccessError,
|
||||||
PdfMergeFailedError,
|
|
||||||
PriorOcrFoundError,
|
PriorOcrFoundError,
|
||||||
SubprocessOutputError,
|
SubprocessOutputError,
|
||||||
TesseractConfigError,
|
TesseractConfigError,
|
||||||
@@ -30,3 +29,33 @@ from ocrmypdf.exceptions import (
|
|||||||
from ocrmypdf.pluginspec import OcrEngine, OrientationConfidence
|
from ocrmypdf.pluginspec import OcrEngine, OrientationConfidence
|
||||||
|
|
||||||
hookimpl = _HookimplMarker('ocrmypdf')
|
hookimpl = _HookimplMarker('ocrmypdf')
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
'__version__',
|
||||||
|
'BadArgsError',
|
||||||
|
'configure_logging',
|
||||||
|
'DpiError',
|
||||||
|
'EncryptedPdfError',
|
||||||
|
'Executor',
|
||||||
|
'ExitCode',
|
||||||
|
'ExitCodeException',
|
||||||
|
'helpers',
|
||||||
|
'hocrtransform',
|
||||||
|
'hookimpl',
|
||||||
|
'InputFileError',
|
||||||
|
'MissingDependencyError',
|
||||||
|
'ocr',
|
||||||
|
'OcrEngine',
|
||||||
|
'OrientationConfidence',
|
||||||
|
'OutputFileAccessError',
|
||||||
|
'PageContext',
|
||||||
|
'pdfa',
|
||||||
|
'PdfContext',
|
||||||
|
'pdfinfo',
|
||||||
|
'PriorOcrFoundError',
|
||||||
|
'PROGRAM_NAME',
|
||||||
|
'SubprocessOutputError',
|
||||||
|
'TesseractConfigError',
|
||||||
|
'UnsupportedImageFormatError',
|
||||||
|
'Verbosity',
|
||||||
|
]
|
||||||
|
|||||||
@@ -11,7 +11,6 @@ import os
|
|||||||
import signal
|
import signal
|
||||||
import sys
|
import sys
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from multiprocessing import set_start_method
|
|
||||||
|
|
||||||
from ocrmypdf import __version__
|
from ocrmypdf import __version__
|
||||||
from ocrmypdf._plugin_manager import get_parser_options_plugins
|
from ocrmypdf._plugin_manager import get_parser_options_plugins
|
||||||
@@ -29,10 +28,16 @@ log = logging.getLogger('ocrmypdf')
|
|||||||
|
|
||||||
|
|
||||||
def sigbus(*args):
|
def sigbus(*args):
|
||||||
|
"""Handle SIGBUS signals.
|
||||||
|
|
||||||
|
pikepdf, depending on configuration, may use mmap so SIGBUS is a
|
||||||
|
possibility.
|
||||||
|
"""
|
||||||
raise InputFileError("Lost access to the input file")
|
raise InputFileError("Lost access to the input file")
|
||||||
|
|
||||||
|
|
||||||
def run(args=None):
|
def run(args=None):
|
||||||
|
"""Run the ocrmypdf command line interface."""
|
||||||
_parser, options, plugin_manager = get_parser_options_plugins(args=args)
|
_parser, options, plugin_manager = get_parser_options_plugins(args=args)
|
||||||
|
|
||||||
with suppress(AttributeError, PermissionError):
|
with suppress(AttributeError, PermissionError):
|
||||||
@@ -71,6 +76,4 @@ def run(args=None):
|
|||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
if sys.platform == 'darwin' and sys.version_info < (3, 8):
|
|
||||||
set_start_method('spawn') # see python bpo-33725
|
|
||||||
sys.exit(run())
|
sys.exit(run())
|
||||||
|
|||||||
@@ -51,8 +51,7 @@ class Executor(ABC):
|
|||||||
task_arguments: Iterable | None = None,
|
task_arguments: Iterable | None = None,
|
||||||
task_finished: Callable | None = None,
|
task_finished: Callable | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""
|
"""Set up parallel execution and progress reporting.
|
||||||
Set up parallel execution and progress reporting.
|
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
use_threads: If ``False``, the workload is the sort that will benefit from
|
use_threads: If ``False``, the workload is the sort that will benefit from
|
||||||
@@ -60,7 +59,7 @@ class Executor(ABC):
|
|||||||
heavily, and parallelizing it with threads is not expected to be
|
heavily, and parallelizing it with threads is not expected to be
|
||||||
performant).
|
performant).
|
||||||
max_workers: The maximum number of workers that should be run.
|
max_workers: The maximum number of workers that should be run.
|
||||||
tdqm_kwargs: Arguments to set up the progress bar.
|
tqdm_kwargs: Arguments to set up the progress bar.
|
||||||
worker_initializer: Called when a worker is initialized, in the worker's
|
worker_initializer: Called when a worker is initialized, in the worker's
|
||||||
execution context. If the child workers are processes, it must be
|
execution context. If the child workers are processes, it must be
|
||||||
possible to marshall/pickle the worker initializer.
|
possible to marshall/pickle the worker initializer.
|
||||||
@@ -73,7 +72,6 @@ class Executor(ABC):
|
|||||||
task. This runs in the parent's context, but the parameters must be
|
task. This runs in the parent's context, but the parameters must be
|
||||||
marshallable to the worker.
|
marshallable to the worker.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
if not task_arguments:
|
if not task_arguments:
|
||||||
return # Nothing to do!
|
return # Nothing to do!
|
||||||
if not worker_initializer:
|
if not worker_initializer:
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""Manage third party executables"""
|
"""Manage third party executables."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""Interface to Ghostscript executable"""
|
"""Interface to Ghostscript executable."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -14,6 +14,7 @@ from os import fspath
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, CalledProcessError
|
from subprocess import PIPE, CalledProcessError
|
||||||
|
|
||||||
|
from packaging.version import Version
|
||||||
from PIL import Image, UnidentifiedImageError
|
from PIL import Image, UnidentifiedImageError
|
||||||
|
|
||||||
from ocrmypdf.exceptions import SubprocessOutputError
|
from ocrmypdf.exceptions import SubprocessOutputError
|
||||||
@@ -27,39 +28,50 @@ except AttributeError:
|
|||||||
# Pillow 9 shim
|
# Pillow 9 shim
|
||||||
Transpose = Image # type: ignore
|
Transpose = Image # type: ignore
|
||||||
|
|
||||||
|
|
||||||
|
COLOR_CONVERSION_STRATEGIES = frozenset(
|
||||||
|
[
|
||||||
|
'CMYK',
|
||||||
|
'Gray',
|
||||||
|
'LeaveColorUnchanged',
|
||||||
|
'RGB',
|
||||||
|
'UseDeviceIndependentColor',
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
# Most reliable what to get the bitness of Python interpreter, according to Python docs
|
|
||||||
_IS_64BIT = sys.maxsize > 2**32
|
|
||||||
|
|
||||||
_GSWIN = None
|
class DuplicateFilter(logging.Filter):
|
||||||
if os.name == 'nt':
|
"""Filter out duplicate log messages."""
|
||||||
if _IS_64BIT:
|
|
||||||
_GSWIN = 'gswin64c'
|
|
||||||
else:
|
|
||||||
_GSWIN = 'gswin32c'
|
|
||||||
|
|
||||||
GS = _GSWIN if _GSWIN else 'gs'
|
def __init__(self, logger: logging.Logger):
|
||||||
del _GSWIN
|
self.last: logging.LogRecord | None = None
|
||||||
|
self.count = 0
|
||||||
|
self.logger = logger
|
||||||
|
|
||||||
|
def filter(self, record):
|
||||||
|
if self.last and record.msg == self.last.msg:
|
||||||
|
self.count += 1
|
||||||
|
return False
|
||||||
|
else:
|
||||||
|
if self.count >= 1:
|
||||||
|
rep_msg = f"(previous message repeated {self.count} times)"
|
||||||
|
self.count = 0 # Avoid infinite recursion
|
||||||
|
self.logger.log(self.last.levelno, rep_msg)
|
||||||
|
self.last = record
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
def version():
|
log.addFilter(DuplicateFilter(log))
|
||||||
return get_version(GS)
|
|
||||||
|
|
||||||
|
|
||||||
def jpeg_passthrough_available() -> bool:
|
# Ghostscript executable - gswin32c is not supported
|
||||||
"""Returns True if the installed version of Ghostscript supports JPEG passthru
|
GS = 'gswin64c' if os.name == 'nt' else 'gs'
|
||||||
|
|
||||||
Prior to 9.23, Ghostscript decoded and re-encoded JPEGs internally. In 9.23
|
|
||||||
it gained the ability to keep JPEGs unmodified. However, the 9.23
|
|
||||||
implementation was buggy and would deletes the last two bytes of images in
|
|
||||||
some cases, as reported here.
|
|
||||||
https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
|
||||||
|
|
||||||
The issue was fixed for 9.24, hence that is the first version we consider
|
def version() -> Version:
|
||||||
the feature available. (Ghostscript 9.24 has its own problems is blacklisted.)
|
return Version(get_version(GS))
|
||||||
"""
|
|
||||||
return version() >= '9.24'
|
|
||||||
|
|
||||||
|
|
||||||
def _gs_error_reported(stream) -> bool:
|
def _gs_error_reported(stream) -> bool:
|
||||||
@@ -77,6 +89,7 @@ def rasterize_pdf(
|
|||||||
page_dpi: Resolution | None = None,
|
page_dpi: Resolution | None = None,
|
||||||
rotation: int | None = None,
|
rotation: int | None = None,
|
||||||
filter_vector: bool = False,
|
filter_vector: bool = False,
|
||||||
|
stop_on_error: bool = False,
|
||||||
):
|
):
|
||||||
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units."""
|
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units."""
|
||||||
raster_dpi = raster_dpi.round(6)
|
raster_dpi = raster_dpi.round(6)
|
||||||
@@ -97,6 +110,7 @@ def rasterize_pdf(
|
|||||||
f'-r{raster_dpi.x:f}x{raster_dpi.y:f}',
|
f'-r{raster_dpi.x:f}x{raster_dpi.y:f}',
|
||||||
]
|
]
|
||||||
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
||||||
|
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||||
+ [
|
+ [
|
||||||
'-o',
|
'-o',
|
||||||
'-',
|
'-',
|
||||||
@@ -172,9 +186,11 @@ def generate_pdfa(
|
|||||||
output_file: os.PathLike,
|
output_file: os.PathLike,
|
||||||
*,
|
*,
|
||||||
compression: str,
|
compression: str,
|
||||||
|
color_conversion_strategy: str,
|
||||||
pdf_version: str = '1.5',
|
pdf_version: str = '1.5',
|
||||||
pdfa_part: str = '2',
|
pdfa_part: str = '2',
|
||||||
progressbar_class=None,
|
progressbar_class=None,
|
||||||
|
stop_on_error: bool = False,
|
||||||
):
|
):
|
||||||
# Ghostscript's compression is all or nothing. We can either force all images
|
# Ghostscript's compression is all or nothing. We can either force all images
|
||||||
# to JPEG, force all to Flate/PNG, or let it decide how to encode the images.
|
# to JPEG, force all to Flate/PNG, or let it decide how to encode the images.
|
||||||
@@ -200,23 +216,16 @@ def generate_pdfa(
|
|||||||
"-dAutoFilterGrayImages=true",
|
"-dAutoFilterGrayImages=true",
|
||||||
]
|
]
|
||||||
|
|
||||||
strategy = 'LeaveColorUnchanged'
|
|
||||||
# Older versions of Ghostscript expect a leading slash in
|
|
||||||
# sColorConversionStrategy, newer ones should not have it. See Ghostscript
|
|
||||||
# git commit fe1c025d.
|
|
||||||
gs_version = version()
|
gs_version = version()
|
||||||
strategy = ('/' + strategy) if gs_version < '9.19' else strategy
|
if gs_version == Version('9.56.0'):
|
||||||
|
# 9.56.0 breaks our OCR, should be fixed in 9.56.1
|
||||||
if gs_version == '9.23':
|
# https://bugs.ghostscript.com/show_bug.cgi?id=705187
|
||||||
# 9.23: added JPEG passthrough as a new feature, but with a bug that
|
|
||||||
# incorrectly formats some images. Fixed as of 9.24. So we disable this
|
|
||||||
# feature for 9.23.
|
|
||||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
|
||||||
compression_args.append('-dPassThroughJPEGImages=false')
|
|
||||||
elif gs_version == '9.56.0':
|
|
||||||
# 9.56.0 breaks our OCR...?
|
|
||||||
compression_args.append('-dNEWPDF=false')
|
compression_args.append('-dNEWPDF=false')
|
||||||
|
|
||||||
|
if os.name == 'nt':
|
||||||
|
# Windows has lots of fatal "permission denied" errors
|
||||||
|
stop_on_error = False
|
||||||
|
|
||||||
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
||||||
# is set; see:
|
# is set; see:
|
||||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699392
|
# https://bugs.ghostscript.com/show_bug.cgi?id=699392
|
||||||
@@ -226,15 +235,16 @@ def generate_pdfa(
|
|||||||
"-dBATCH",
|
"-dBATCH",
|
||||||
"-dNOPAUSE",
|
"-dNOPAUSE",
|
||||||
"-dSAFER",
|
"-dSAFER",
|
||||||
"-dCompatibilityLevel=" + str(pdf_version),
|
f"-dCompatibilityLevel={str(pdf_version)}",
|
||||||
"-sDEVICE=pdfwrite",
|
"-sDEVICE=pdfwrite",
|
||||||
"-dAutoRotatePages=/None",
|
"-dAutoRotatePages=/None",
|
||||||
"-sColorConversionStrategy=" + strategy,
|
f"-sColorConversionStrategy={color_conversion_strategy}",
|
||||||
]
|
]
|
||||||
|
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||||
+ compression_args
|
+ compression_args
|
||||||
+ [
|
+ [
|
||||||
"-dJPEGQ=95",
|
"-dJPEGQ=95",
|
||||||
"-dPDFA=" + pdfa_part,
|
f"-dPDFA={pdfa_part}",
|
||||||
"-dPDFACompatibilityPolicy=1",
|
"-dPDFACompatibilityPolicy=1",
|
||||||
"-o",
|
"-o",
|
||||||
"-",
|
"-",
|
||||||
@@ -265,14 +275,9 @@ def generate_pdfa(
|
|||||||
# If there is an error we log the whole stderr, except for filtering
|
# If there is an error we log the whole stderr, except for filtering
|
||||||
# duplicates.
|
# duplicates.
|
||||||
if _gs_error_reported(stderr):
|
if _gs_error_reported(stderr):
|
||||||
last_part = None
|
# Ghostscript outputs the pattern **** Error: .... frequently.
|
||||||
repcount = 0
|
# Occasionally the error message is spammed many times. We filter
|
||||||
|
# out duplicates of this message using the filter above. We use
|
||||||
|
# the **** pattern to split the stderr into parts.
|
||||||
for part in stderr.split('****'):
|
for part in stderr.split('****'):
|
||||||
if part != last_part:
|
log.error(part)
|
||||||
if repcount > 1:
|
|
||||||
log.error(f"(previous error message repeated {repcount} times)")
|
|
||||||
repcount = 0
|
|
||||||
log.error(part)
|
|
||||||
else:
|
|
||||||
repcount += 1
|
|
||||||
last_part = part
|
|
||||||
|
|||||||
@@ -1,18 +1,20 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""Interface to jbig2 executable"""
|
"""Interface to jbig2 executable."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from subprocess import PIPE
|
from subprocess import PIPE
|
||||||
|
|
||||||
|
from packaging.version import Version
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
from ocrmypdf.subprocess import get_version, run
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
|
|
||||||
def version():
|
def version() -> Version:
|
||||||
return get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
return Version(get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*'))
|
||||||
|
|
||||||
|
|
||||||
def available():
|
def available():
|
||||||
@@ -23,15 +25,17 @@ def available():
|
|||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
def convert_group(*, cwd, infiles, out_prefix):
|
def convert_group(*, cwd, infiles, out_prefix, threshold):
|
||||||
args = [
|
args = [
|
||||||
'jbig2',
|
'jbig2',
|
||||||
'-b',
|
'-b',
|
||||||
out_prefix,
|
out_prefix,
|
||||||
'-s', # symbol mode (lossy)
|
'--symbol-mode', # symbol mode (lossy)
|
||||||
|
'-t',
|
||||||
|
str(threshold), # threshold
|
||||||
# '-r', # refinement mode (lossless symbol mode, currently disabled in
|
# '-r', # refinement mode (lossless symbol mode, currently disabled in
|
||||||
# jbig2)
|
# jbig2)
|
||||||
'-p',
|
'--pdf',
|
||||||
]
|
]
|
||||||
args.extend(infiles)
|
args.extend(infiles)
|
||||||
proc = run(args, cwd=cwd, stdout=PIPE, stderr=PIPE)
|
proc = run(args, cwd=cwd, stdout=PIPE, stderr=PIPE)
|
||||||
@@ -40,11 +44,13 @@ def convert_group(*, cwd, infiles, out_prefix):
|
|||||||
|
|
||||||
|
|
||||||
def convert_group_mp(args):
|
def convert_group_mp(args):
|
||||||
return convert_group(cwd=args[0], infiles=args[1], out_prefix=args[2])
|
return convert_group(
|
||||||
|
cwd=args[0], infiles=args[1], out_prefix=args[2], threshold=args[3]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def convert_single(*, cwd, infile, outfile):
|
def convert_single(*, cwd, infile, outfile, threshold):
|
||||||
args = ['jbig2', '-p', infile]
|
args = ['jbig2', '--pdf', '-t', str(threshold), infile]
|
||||||
with open(outfile, 'wb') as fstdout:
|
with open(outfile, 'wb') as fstdout:
|
||||||
proc = run(args, cwd=cwd, stdout=fstdout, stderr=PIPE)
|
proc = run(args, cwd=cwd, stdout=fstdout, stderr=PIPE)
|
||||||
proc.check_returncode()
|
proc.check_returncode()
|
||||||
@@ -52,4 +58,6 @@ def convert_single(*, cwd, infile, outfile):
|
|||||||
|
|
||||||
|
|
||||||
def convert_single_mp(args):
|
def convert_single_mp(args):
|
||||||
return convert_single(cwd=args[0], infile=args[1], outfile=args[2])
|
return convert_single(
|
||||||
|
cwd=args[0], infile=args[1], outfile=args[2], threshold=args[3]
|
||||||
|
)
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""Interface to pngquant executable"""
|
"""Interface to pngquant executable."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -10,14 +10,15 @@ from io import BytesIO
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE
|
from subprocess import PIPE
|
||||||
|
|
||||||
|
from packaging.version import Version
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
from ocrmypdf.subprocess import get_version, run
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
|
|
||||||
def version():
|
def version() -> Version:
|
||||||
return get_version('pngquant', regex=r'(\d+(\.\d+)*).*')
|
return Version(get_version('pngquant', regex=r'(\d+(\.\d+)*).*'))
|
||||||
|
|
||||||
|
|
||||||
def available():
|
def available():
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""Interface to Tesseract executable"""
|
"""Interface to Tesseract executable."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -33,11 +33,13 @@ HOCR_TEMPLATE = """<?xml version="1.0" encoding="UTF-8"?>
|
|||||||
<head>
|
<head>
|
||||||
<title></title>
|
<title></title>
|
||||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8" />
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8" />
|
||||||
<meta name='ocr-system' content='tesseract 4.0.0' />
|
<meta name='ocr-system' content='tesseract 4.1.1' />
|
||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities'
|
||||||
|
content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "_blank.png"; bbox 0 0 {0} {1}; ppageno 0'>
|
<div class='ocr_page' id='page_1'
|
||||||
|
title='image "_blank.png"; bbox 0 0 {0} {1}; ppageno 0'>
|
||||||
</div>
|
</div>
|
||||||
</body>
|
</body>
|
||||||
</html>
|
</html>
|
||||||
@@ -52,7 +54,7 @@ TESSERACT_THRESHOLDING_METHODS: dict[str, int] = {
|
|||||||
|
|
||||||
|
|
||||||
class TesseractLoggerAdapter(logging.LoggerAdapter):
|
class TesseractLoggerAdapter(logging.LoggerAdapter):
|
||||||
"Prepend [tesseract] to messages emitted from tesseract"
|
"""Prepend [tesseract] to messages emitted from tesseract."""
|
||||||
|
|
||||||
def process(self, msg, kwargs):
|
def process(self, msg, kwargs):
|
||||||
kwargs['extra'] = self.extra
|
kwargs['extra'] = self.extra
|
||||||
@@ -104,28 +106,20 @@ TESSERACT_VERSION_PATTERN = r"""
|
|||||||
|
|
||||||
|
|
||||||
class TesseractVersion(Version):
|
class TesseractVersion(Version):
|
||||||
"Modify standard packaging.Version regex to support Tesseract idiosyncracies."
|
"""Modify standard packaging.Version regex to support Tesseract idiosyncrasies."""
|
||||||
|
|
||||||
_regex = re.compile(
|
_regex = re.compile(
|
||||||
r"^\s*" + TESSERACT_VERSION_PATTERN + r"\s*$", re.VERBOSE | re.IGNORECASE
|
r"^\s*" + TESSERACT_VERSION_PATTERN + r"\s*$", re.VERBOSE | re.IGNORECASE
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def version() -> str:
|
def version() -> Version:
|
||||||
return get_version('tesseract', regex=r'tesseract\s(.+)')
|
return TesseractVersion(get_version('tesseract', regex=r'tesseract\s(.+)'))
|
||||||
|
|
||||||
|
|
||||||
def has_user_words() -> bool:
|
|
||||||
"""Does Tesseract have --user-words capability?
|
|
||||||
|
|
||||||
Not available in 4.0, but available in 4.1. Also available in 3.x, but
|
|
||||||
we no longer support 3.x.
|
|
||||||
"""
|
|
||||||
return version() >= '4.1'
|
|
||||||
|
|
||||||
|
|
||||||
def has_thresholding() -> bool:
|
def has_thresholding() -> bool:
|
||||||
"""Does Tesseract have -c thresholding method capability?"""
|
"""Does Tesseract have -c thresholding method capability?"""
|
||||||
return version() >= '5.0'
|
return version() >= Version('5.0')
|
||||||
|
|
||||||
|
|
||||||
def get_languages() -> set[str]:
|
def get_languages() -> set[str]:
|
||||||
@@ -239,12 +233,13 @@ def get_deskew(
|
|||||||
parsed = _parse_tesseract_output(p.stdout)
|
parsed = _parse_tesseract_output(p.stdout)
|
||||||
deskew_radians = float(parsed.get('Deskew angle', 0))
|
deskew_radians = float(parsed.get('Deskew angle', 0))
|
||||||
deskew_degrees = 180 / pi * deskew_radians
|
deskew_degrees = 180 / pi * deskew_radians
|
||||||
|
log.debug(f"Deskew angle: {deskew_degrees:.3f}")
|
||||||
return deskew_degrees
|
return deskew_degrees
|
||||||
|
|
||||||
|
|
||||||
def tesseract_log_output(stream: bytes) -> None:
|
def tesseract_log_output(stream: bytes) -> None:
|
||||||
tlog = TesseractLoggerAdapter(
|
tlog = TesseractLoggerAdapter(
|
||||||
log, extra=log.extra if hasattr(log, 'extra') else None
|
log, extra=log.extra if hasattr(log, 'extra') else None # type: ignore
|
||||||
)
|
)
|
||||||
|
|
||||||
if not stream:
|
if not stream:
|
||||||
@@ -289,8 +284,10 @@ def page_timedout(timeout: float) -> None:
|
|||||||
|
|
||||||
|
|
||||||
def _generate_null_hocr(output_hocr: Path, output_text: Path, image: Path) -> None:
|
def _generate_null_hocr(output_hocr: Path, output_text: Path, image: Path) -> None:
|
||||||
"""Produce a .hocr file that reports no text detected on a page that is
|
"""Produce a .hocr file that reports no text detected.
|
||||||
the same size as the input image."""
|
|
||||||
|
Ensures page is the same size as the input image.
|
||||||
|
"""
|
||||||
with Image.open(image) as im:
|
with Image.open(image) as im:
|
||||||
w, h = im.size
|
w, h = im.size
|
||||||
|
|
||||||
|
|||||||
@@ -1,11 +1,9 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
from __future__ import annotations
|
"""Interface to unpaper executable."""
|
||||||
|
|
||||||
# unpaper documentation:
|
from __future__ import annotations
|
||||||
# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md
|
|
||||||
"""Interface to unpaper executable"""
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -17,11 +15,16 @@ from pathlib import Path
|
|||||||
from subprocess import PIPE, STDOUT
|
from subprocess import PIPE, STDOUT
|
||||||
from typing import Iterator, Union
|
from typing import Iterator, Union
|
||||||
|
|
||||||
|
from packaging.version import Version
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
||||||
from ocrmypdf.subprocess import get_version, run
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
|
# unpaper documentation:
|
||||||
|
# https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md
|
||||||
|
|
||||||
|
|
||||||
if sys.version_info >= (3, 10):
|
if sys.version_info >= (3, 10):
|
||||||
from tempfile import TemporaryDirectory
|
from tempfile import TemporaryDirectory
|
||||||
else:
|
else:
|
||||||
@@ -65,8 +68,8 @@ class UnpaperImageTooLargeError(Exception):
|
|||||||
super().__init__(self.message)
|
super().__init__(self.message)
|
||||||
|
|
||||||
|
|
||||||
def version() -> str:
|
def version() -> Version:
|
||||||
return get_version('unpaper')
|
return Version(get_version('unpaper'))
|
||||||
|
|
||||||
|
|
||||||
SUPPORTED_MODES = {'1', 'L', 'RGB'}
|
SUPPORTED_MODES = {'1', 'L', 'RGB'}
|
||||||
|
|||||||
@@ -37,7 +37,6 @@ def _update_resources(*, obj, font, font_key, procset):
|
|||||||
|
|
||||||
obj can be a page or Form XObject.
|
obj can be a page or Form XObject.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
resources = _ensure_dictionary(obj, Name.Resources)
|
resources = _ensure_dictionary(obj, Name.Resources)
|
||||||
fonts = _ensure_dictionary(resources, Name.Font)
|
fonts = _ensure_dictionary(resources, Name.Font)
|
||||||
if font_key is not None and font_key not in fonts:
|
if font_key is not None and font_key not in fonts:
|
||||||
@@ -131,7 +130,8 @@ class OcrGrafter:
|
|||||||
text_misaligned = (text_rotation - content_rotation) % 360
|
text_misaligned = (text_rotation - content_rotation) % 360
|
||||||
log.debug(
|
log.debug(
|
||||||
f"Text rotation: (text, autorotate, content) -> text misalignment = "
|
f"Text rotation: (text, autorotate, content) -> text misalignment = "
|
||||||
f"({text_rotation}, {autorotate_correction}, {content_rotation}) -> {text_misaligned}"
|
f"({text_rotation}, {autorotate_correction}, {content_rotation}) -> "
|
||||||
|
f"{text_misaligned}"
|
||||||
)
|
)
|
||||||
|
|
||||||
if textpdf and self.font:
|
if textpdf and self.font:
|
||||||
@@ -166,7 +166,6 @@ class OcrGrafter:
|
|||||||
the font to page 1 even if page 1 doesn't use it, so we have a way to get it
|
the font to page 1 even if page 1 doesn't use it, so we have a way to get it
|
||||||
back.
|
back.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
page0 = self.pdf_base.pages[0]
|
page0 = self.pdf_base.pages[0]
|
||||||
_update_resources(
|
_update_resources(
|
||||||
obj=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
obj=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
||||||
@@ -199,8 +198,7 @@ class OcrGrafter:
|
|||||||
return self.output_file
|
return self.output_file
|
||||||
|
|
||||||
def _find_font(self, text):
|
def _find_font(self, text):
|
||||||
"""Copy a font from the filename text into pdf_base"""
|
"""Copy a font from the filename text into pdf_base."""
|
||||||
|
|
||||||
font, font_key = None, None
|
font, font_key = None, None
|
||||||
possible_font_names = ('/f-0-0', '/F1')
|
possible_font_names = ('/f-0-0', '/F1')
|
||||||
try:
|
try:
|
||||||
@@ -233,8 +231,7 @@ class OcrGrafter:
|
|||||||
text_rotation: int,
|
text_rotation: int,
|
||||||
strip_old_text: bool,
|
strip_old_text: bool,
|
||||||
):
|
):
|
||||||
"""Insert the text layer from text page 0 on to pdf_base at page_num"""
|
"""Insert the text layer from text page 0 on to pdf_base at page_num."""
|
||||||
|
|
||||||
# pylint: disable=invalid-name
|
# pylint: disable=invalid-name
|
||||||
|
|
||||||
log.debug("Grafting")
|
log.debug("Grafting")
|
||||||
|
|||||||
@@ -59,7 +59,7 @@ class PdfContext:
|
|||||||
class PageContext:
|
class PageContext:
|
||||||
"""Holds our context for a page.
|
"""Holds our context for a page.
|
||||||
|
|
||||||
Must be pickable, so stores only intrinsic/simple data elements or those
|
Must be pickle-able, so stores only intrinsic/simple data elements or those
|
||||||
capable of their serializing themselves via ``__getstate__``.
|
capable of their serializing themselves via ``__getstate__``.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|||||||
+65
-16
@@ -6,9 +6,18 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
from contextlib import suppress
|
|
||||||
|
|
||||||
from tqdm import tqdm
|
from rich.console import Console
|
||||||
|
from rich.logging import RichHandler
|
||||||
|
from rich.progress import (
|
||||||
|
BarColumn,
|
||||||
|
MofNCompleteColumn,
|
||||||
|
Progress,
|
||||||
|
TaskProgressColumn,
|
||||||
|
TextColumn,
|
||||||
|
TimeRemainingColumn,
|
||||||
|
)
|
||||||
|
from rich.table import Column
|
||||||
|
|
||||||
|
|
||||||
class PageNumberFilter(logging.Filter):
|
class PageNumberFilter(logging.Filter):
|
||||||
@@ -23,21 +32,61 @@ class PageNumberFilter(logging.Filter):
|
|||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
class TqdmConsole:
|
class RichLoggingHandler(RichHandler):
|
||||||
"""Wrapper to log messages in a way that is compatible with tqdm progress bar
|
def __init__(self, console: Console, **kwargs):
|
||||||
|
super().__init__(
|
||||||
|
console=console, show_level=False, show_time=False, markup=True, **kwargs
|
||||||
|
)
|
||||||
|
|
||||||
This routes log messages through tqdm so that it can print them above the
|
|
||||||
progress bar, and then refresh the progress bar, rather than overwriting
|
|
||||||
it which looks messy.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, file):
|
class RichTqdmProgressAdapter:
|
||||||
self.file = file
|
"""Adapt tqdm API to rich progress bar."""
|
||||||
|
|
||||||
def write(self, msg):
|
def __init__(
|
||||||
# When no progress bar is active, tqdm.write() routes to print()
|
self,
|
||||||
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
*,
|
||||||
|
console: Console,
|
||||||
|
desc: str,
|
||||||
|
total: float | None = None,
|
||||||
|
unit: str | None = None,
|
||||||
|
unit_scale: float | None = 1.0,
|
||||||
|
disable: bool = False,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
self.progress = Progress(
|
||||||
|
TextColumn(
|
||||||
|
"[progress.description]{task.description}",
|
||||||
|
table_column=Column(min_width=20),
|
||||||
|
),
|
||||||
|
BarColumn(),
|
||||||
|
TaskProgressColumn(),
|
||||||
|
MofNCompleteColumn(),
|
||||||
|
TimeRemainingColumn(),
|
||||||
|
console=console,
|
||||||
|
auto_refresh=True,
|
||||||
|
redirect_stderr=True,
|
||||||
|
redirect_stdout=False,
|
||||||
|
disable=disable,
|
||||||
|
**kwargs,
|
||||||
|
)
|
||||||
|
self.unit_scale = unit_scale
|
||||||
|
self.progress_bar = self.progress.add_task(
|
||||||
|
desc,
|
||||||
|
total=total * self.unit_scale
|
||||||
|
if total is not None and self.unit_scale is not None
|
||||||
|
else None,
|
||||||
|
unit=unit,
|
||||||
|
)
|
||||||
|
|
||||||
def flush(self):
|
def __enter__(self):
|
||||||
with suppress(AttributeError):
|
self.progress.start()
|
||||||
self.file.flush()
|
return self
|
||||||
|
|
||||||
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
|
self.progress.refresh()
|
||||||
|
self.progress.stop()
|
||||||
|
return False
|
||||||
|
|
||||||
|
def update(self, value=None):
|
||||||
|
advance = self.unit_scale if value is None else value
|
||||||
|
self.progress.update(self.progress_bar, advance=advance)
|
||||||
|
|||||||
+313
-145
@@ -1,4 +1,5 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2018-2022 James R. Barlow
|
||||||
|
# SPDX-FileCopyrightText: 2019 Martin Wind
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""OCRmyPDF page processing pipeline functions."""
|
"""OCRmyPDF page processing pipeline functions."""
|
||||||
@@ -13,7 +14,7 @@ from contextlib import suppress
|
|||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj
|
from shutil import copyfileobj
|
||||||
from typing import Iterable
|
from typing import Any, BinaryIO, Iterable, Iterator, Sequence, cast
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
import pikepdf
|
import pikepdf
|
||||||
@@ -26,6 +27,7 @@ from ocrmypdf._jobcontext import PageContext, PdfContext
|
|||||||
from ocrmypdf._version import PROGRAM_NAME
|
from ocrmypdf._version import PROGRAM_NAME
|
||||||
from ocrmypdf._version import __version__ as VERSION
|
from ocrmypdf._version import __version__ as VERSION
|
||||||
from ocrmypdf.exceptions import (
|
from ocrmypdf.exceptions import (
|
||||||
|
DigitalSignatureError,
|
||||||
DpiError,
|
DpiError,
|
||||||
EncryptedPdfError,
|
EncryptedPdfError,
|
||||||
InputFileError,
|
InputFileError,
|
||||||
@@ -35,7 +37,8 @@ from ocrmypdf.exceptions import (
|
|||||||
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
|
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
|
||||||
from ocrmypdf.hocrtransform import HocrTransform
|
from ocrmypdf.hocrtransform import HocrTransform
|
||||||
from ocrmypdf.pdfa import generate_pdfa_ps
|
from ocrmypdf.pdfa import generate_pdfa_ps
|
||||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
from ocrmypdf.pdfinfo import Colorspace, Encoding, PageInfo, PdfInfo
|
||||||
|
from ocrmypdf.pluginspec import OrientationConfidence
|
||||||
|
|
||||||
# Remove this workaround when we require Pillow >= 10
|
# Remove this workaround when we require Pillow >= 10
|
||||||
try:
|
try:
|
||||||
@@ -49,7 +52,21 @@ log = logging.getLogger(__name__)
|
|||||||
VECTOR_PAGE_DPI = 400
|
VECTOR_PAGE_DPI = 400
|
||||||
|
|
||||||
|
|
||||||
def triage_image_file(input_file, output_file, options):
|
def triage_image_file(input_file: Path, output_file: Path, options) -> None:
|
||||||
|
"""Triage the input image file.
|
||||||
|
|
||||||
|
If the input file is an image, check its resolution and convert it to PDF.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
input_file: The path to the input file.
|
||||||
|
output_file: The path to the output file.
|
||||||
|
options: An object containing the options passed to the OCRmyPDF command.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
UnsupportedImageFormatError: If the input file is not a supported image format.
|
||||||
|
DpiError: If the input image has no resolution (DPI) in its metadata or if the
|
||||||
|
resolution is not credible.
|
||||||
|
"""
|
||||||
log.info("Input file is not a PDF, checking if it is an image...")
|
log.info("Input file is not a PDF, checking if it is an image...")
|
||||||
try:
|
try:
|
||||||
im = Image.open(input_file)
|
im = Image.open(input_file)
|
||||||
@@ -64,34 +81,32 @@ def triage_image_file(input_file, output_file, options):
|
|||||||
if im.info['dpi'] <= (96, 96) and not options.image_dpi:
|
if im.info['dpi'] <= (96, 96) and not options.image_dpi:
|
||||||
log.info("Image size: (%d, %d)", *im.size)
|
log.info("Image size: (%d, %d)", *im.size)
|
||||||
log.info("Image resolution: (%d, %d)", *im.info['dpi'])
|
log.info("Image resolution: (%d, %d)", *im.info['dpi'])
|
||||||
log.error(
|
raise DpiError(
|
||||||
"Input file is an image, but the resolution (DPI) is "
|
"Input file is an image, but the resolution (DPI) is "
|
||||||
"not credible. Estimate the resolution at which the "
|
"not credible. Estimate the resolution at which the "
|
||||||
"image was scanned and specify it using --image-dpi."
|
"image was scanned and specify it using --image-dpi."
|
||||||
)
|
)
|
||||||
raise DpiError()
|
|
||||||
elif not options.image_dpi:
|
elif not options.image_dpi:
|
||||||
log.info("Image size: (%d, %d)", *im.size)
|
log.info("Image size: (%d, %d)", *im.size)
|
||||||
log.error(
|
raise DpiError(
|
||||||
"Input file is an image, but has no resolution (DPI) "
|
"Input file is an image, but has no resolution (DPI) "
|
||||||
"in its metadata. Estimate the resolution at which "
|
"in its metadata. Estimate the resolution at which "
|
||||||
"image was scanned and specify it using --image-dpi."
|
"image was scanned and specify it using --image-dpi."
|
||||||
)
|
)
|
||||||
raise DpiError()
|
|
||||||
|
|
||||||
if im.mode in ('RGBA', 'LA'):
|
if im.mode in ('RGBA', 'LA'):
|
||||||
log.error(
|
raise UnsupportedImageFormatError(
|
||||||
"The input image has an alpha channel. Remove the alpha "
|
"The input image has an alpha channel. Remove the alpha "
|
||||||
"channel first."
|
"channel first."
|
||||||
)
|
)
|
||||||
raise UnsupportedImageFormatError()
|
|
||||||
|
|
||||||
if 'iccprofile' not in im.info:
|
if 'iccprofile' not in im.info:
|
||||||
if im.mode == 'RGB':
|
if im.mode == 'RGB':
|
||||||
log.info("Input image has no ICC profile, assuming sRGB")
|
log.info("Input image has no ICC profile, assuming sRGB")
|
||||||
elif im.mode == 'CMYK':
|
elif im.mode == 'CMYK':
|
||||||
log.error("Input CMYK image has no ICC profile, not usable")
|
raise UnsupportedImageFormatError(
|
||||||
raise UnsupportedImageFormatError()
|
"Input CMYK image has no ICC profile, not usable"
|
||||||
|
)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
log.info("Image seems valid. Try converting to PDF...")
|
log.info("Image seems valid. Try converting to PDF...")
|
||||||
@@ -109,27 +124,27 @@ def triage_image_file(input_file, output_file, options):
|
|||||||
)
|
)
|
||||||
log.info("Successfully converted to PDF, processing...")
|
log.info("Successfully converted to PDF, processing...")
|
||||||
except img2pdf.ImageOpenError as e:
|
except img2pdf.ImageOpenError as e:
|
||||||
log.error(e)
|
|
||||||
raise UnsupportedImageFormatError() from e
|
raise UnsupportedImageFormatError() from e
|
||||||
|
|
||||||
|
|
||||||
def _pdf_guess_version(input_file, search_window=1024):
|
def _pdf_guess_version(input_file: Path, search_window=1024) -> str:
|
||||||
"""Try to find version signature at start of file.
|
"""Try to find version signature at start of file.
|
||||||
|
|
||||||
Not robust enough to deal with appended files.
|
Not robust enough to deal with appended files.
|
||||||
|
|
||||||
Returns empty string if not found, indicating file is probably not PDF.
|
Returns empty string if not found, indicating file is probably not PDF.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
with open(input_file, 'rb') as f:
|
with open(input_file, 'rb') as f:
|
||||||
signature = f.read(search_window)
|
signature = f.read(search_window)
|
||||||
m = re.search(br'%PDF-(\d\.\d)', signature)
|
m = re.search(br'%PDF-(\d\.\d)', signature)
|
||||||
if m:
|
if m:
|
||||||
return m.group(1)
|
return m.group(1).decode('ascii')
|
||||||
return ''
|
return ''
|
||||||
|
|
||||||
|
|
||||||
def triage(original_filename, input_file, output_file, options):
|
def triage(
|
||||||
|
original_filename: str, input_file: Path, output_file: Path, options
|
||||||
|
) -> Path:
|
||||||
try:
|
try:
|
||||||
if _pdf_guess_version(input_file):
|
if _pdf_guess_version(input_file):
|
||||||
if options.image_dpi:
|
if options.image_dpi:
|
||||||
@@ -153,9 +168,9 @@ def get_pdfinfo(
|
|||||||
input_file,
|
input_file,
|
||||||
*,
|
*,
|
||||||
executor: Executor,
|
executor: Executor,
|
||||||
detailed_analysis=False,
|
detailed_analysis: bool = False,
|
||||||
progbar=False,
|
progbar: bool = False,
|
||||||
max_workers=None,
|
max_workers: int | None = None,
|
||||||
check_pages=None,
|
check_pages=None,
|
||||||
) -> PdfInfo:
|
) -> PdfInfo:
|
||||||
try:
|
try:
|
||||||
@@ -173,32 +188,26 @@ def get_pdfinfo(
|
|||||||
raise InputFileError() from e
|
raise InputFileError() from e
|
||||||
|
|
||||||
|
|
||||||
def validate_pdfinfo_options(context: PdfContext):
|
def validate_pdfinfo_options(context: PdfContext) -> None:
|
||||||
pdfinfo = context.pdfinfo
|
pdfinfo = context.pdfinfo
|
||||||
options = context.options
|
options = context.options
|
||||||
|
|
||||||
if pdfinfo.needs_rendering:
|
if pdfinfo.needs_rendering:
|
||||||
log.error(
|
raise InputFileError(
|
||||||
"This PDF contains dynamic XFA forms created by Adobe LiveCycle "
|
"This PDF contains dynamic XFA forms created by Adobe LiveCycle "
|
||||||
"Designer and can only be read by Adobe Acrobat or Adobe Reader."
|
"Designer and can only be read by Adobe Acrobat or Adobe Reader."
|
||||||
)
|
)
|
||||||
raise InputFileError()
|
if pdfinfo.has_signature:
|
||||||
if pdfinfo.has_userunit and options.output_type.startswith('pdfa'):
|
if options.invalidate_digital_signatures:
|
||||||
log.error(
|
log.warning("All digital signatures will be invalidated")
|
||||||
"This input file uses a PDF feature that is not supported "
|
else:
|
||||||
"by Ghostscript, so you cannot use --output-type=pdfa for this "
|
raise DigitalSignatureError()
|
||||||
"file. (Specifically, it uses the PDF-1.6 /UserUnit feature to "
|
|
||||||
"support very large or small page sizes, and Ghostscript cannot "
|
|
||||||
"output these files.) Use --output-type=pdf instead."
|
|
||||||
)
|
|
||||||
raise InputFileError()
|
|
||||||
if pdfinfo.has_acroform:
|
if pdfinfo.has_acroform:
|
||||||
if options.redo_ocr:
|
if options.redo_ocr:
|
||||||
log.error(
|
raise InputFileError(
|
||||||
"This PDF has a user fillable form. --redo-ocr is not "
|
"This PDF has a user fillable form. --redo-ocr is not "
|
||||||
"currently possible on such files."
|
"currently possible on such files."
|
||||||
)
|
)
|
||||||
raise InputFileError()
|
|
||||||
else:
|
else:
|
||||||
log.warning(
|
log.warning(
|
||||||
"This PDF has a fillable form. "
|
"This PDF has a fillable form. "
|
||||||
@@ -214,29 +223,24 @@ def validate_pdfinfo_options(context: PdfContext):
|
|||||||
context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options)
|
context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options)
|
||||||
|
|
||||||
|
|
||||||
def _vector_page_dpi(pageinfo):
|
def _vector_page_dpi(pageinfo: PageInfo) -> int:
|
||||||
return VECTOR_PAGE_DPI if pageinfo.has_vector or pageinfo.has_text else 0.0
|
"""Get a DPI to use for vector pages, if the page has vector content."""
|
||||||
|
return VECTOR_PAGE_DPI if pageinfo.has_vector or pageinfo.has_text else 0
|
||||||
|
|
||||||
|
|
||||||
def get_page_dpi(pageinfo, options):
|
def get_page_square_dpi(
|
||||||
"Get the DPI when nonsquare DPI is tolerable"
|
page_context: PageContext, image_dpi: Resolution | None = None
|
||||||
xres = max(
|
) -> Resolution:
|
||||||
pageinfo.dpi.x or VECTOR_PAGE_DPI,
|
"""Get the DPI when we require xres == yres, scaled to physical units.
|
||||||
options.oversample or 0.0,
|
|
||||||
_vector_page_dpi(pageinfo),
|
|
||||||
)
|
|
||||||
yres = max(
|
|
||||||
pageinfo.dpi.y or VECTOR_PAGE_DPI,
|
|
||||||
options.oversample or 0,
|
|
||||||
_vector_page_dpi(pageinfo),
|
|
||||||
)
|
|
||||||
return Resolution(float(xres), float(yres))
|
|
||||||
|
|
||||||
|
Page DPI includes UserUnit scaling.
|
||||||
def get_page_square_dpi(pageinfo, options) -> Resolution:
|
"""
|
||||||
"Get the DPI when we require xres == yres, scaled to physical units"
|
pageinfo = page_context.pageinfo
|
||||||
xres = pageinfo.dpi.x or 0.0
|
options = page_context.options
|
||||||
yres = pageinfo.dpi.y or 0.0
|
if not image_dpi:
|
||||||
|
image_dpi = pageinfo.dpi
|
||||||
|
xres = image_dpi.x or 0.0
|
||||||
|
yres = image_dpi.y or 0.0
|
||||||
userunit = float(pageinfo.userunit) or 1.0
|
userunit = float(pageinfo.userunit) or 1.0
|
||||||
units = float(
|
units = float(
|
||||||
max(
|
max(
|
||||||
@@ -249,12 +253,23 @@ def get_page_square_dpi(pageinfo, options) -> Resolution:
|
|||||||
return Resolution(units, units)
|
return Resolution(units, units)
|
||||||
|
|
||||||
|
|
||||||
def get_canvas_square_dpi(pageinfo, options) -> Resolution:
|
def get_canvas_square_dpi(
|
||||||
"""Get the DPI when we require xres == yres, in Postscript units"""
|
page_context: PageContext, image_dpi: Resolution | None = None
|
||||||
|
) -> Resolution:
|
||||||
|
"""Get the DPI when we require xres == yres, in Postscript units.
|
||||||
|
|
||||||
|
Canvas DPI is independent of PDF UserUnit scaling, which is
|
||||||
|
used to describe situations where the PDF user space is not 1:1 with
|
||||||
|
the physical units of the page.
|
||||||
|
"""
|
||||||
|
pageinfo = page_context.pageinfo
|
||||||
|
options = page_context.options
|
||||||
|
if not image_dpi:
|
||||||
|
image_dpi = pageinfo.dpi
|
||||||
units = float(
|
units = float(
|
||||||
max(
|
max(
|
||||||
(pageinfo.dpi.x) or VECTOR_PAGE_DPI,
|
image_dpi.x or VECTOR_PAGE_DPI,
|
||||||
(pageinfo.dpi.y) or VECTOR_PAGE_DPI,
|
image_dpi.y or VECTOR_PAGE_DPI,
|
||||||
_vector_page_dpi(pageinfo),
|
_vector_page_dpi(pageinfo),
|
||||||
options.oversample or 0.0,
|
options.oversample or 0.0,
|
||||||
)
|
)
|
||||||
@@ -262,7 +277,8 @@ def get_canvas_square_dpi(pageinfo, options) -> Resolution:
|
|||||||
return Resolution(units, units)
|
return Resolution(units, units)
|
||||||
|
|
||||||
|
|
||||||
def is_ocr_required(page_context: PageContext):
|
def is_ocr_required(page_context: PageContext) -> bool:
|
||||||
|
"""Check if the page needs to be OCR'd."""
|
||||||
pageinfo = page_context.pageinfo
|
pageinfo = page_context.pageinfo
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
|
|
||||||
@@ -312,8 +328,8 @@ def is_ocr_required(page_context: PageContext):
|
|||||||
log.warning(
|
log.warning(
|
||||||
"page has no images - "
|
"page has no images - "
|
||||||
"all vector content will be "
|
"all vector content will be "
|
||||||
f"rasterized at {VECTOR_PAGE_DPI} DPI, losing some resolution and likely "
|
f"rasterized at {VECTOR_PAGE_DPI} DPI, losing some resolution and "
|
||||||
"increasing file size. Use --oversample to adjust the "
|
"likely increasing file size. Use --oversample to adjust the "
|
||||||
"DPI."
|
"DPI."
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
@@ -337,10 +353,13 @@ def is_ocr_required(page_context: PageContext):
|
|||||||
return ocr_required
|
return ocr_required
|
||||||
|
|
||||||
|
|
||||||
def rasterize_preview(input_file: Path, page_context: PageContext):
|
def rasterize_preview(input_file: Path, page_context: PageContext) -> Path:
|
||||||
|
"""Generate a lower quality preview image."""
|
||||||
output_file = page_context.get_path('rasterize_preview.jpg')
|
output_file = page_context.get_path('rasterize_preview.jpg')
|
||||||
canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options)
|
canvas_dpi = Resolution(300.0, 300.0).take_min(
|
||||||
page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
[get_canvas_square_dpi(page_context)]
|
||||||
|
)
|
||||||
|
page_dpi = Resolution(300.0, 300.0).take_min([get_page_square_dpi(page_context)])
|
||||||
page_context.plugin_manager.hook.rasterize_pdf_page(
|
page_context.plugin_manager.hook.rasterize_pdf_page(
|
||||||
input_file=input_file,
|
input_file=input_file,
|
||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
@@ -350,14 +369,15 @@ def rasterize_preview(input_file: Path, page_context: PageContext):
|
|||||||
page_dpi=page_dpi,
|
page_dpi=page_dpi,
|
||||||
rotation=0,
|
rotation=0,
|
||||||
filter_vector=False,
|
filter_vector=False,
|
||||||
|
stop_on_soft_error=not page_context.options.continue_on_soft_render_error,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def describe_rotation(page_context: PageContext, orient_conf, correction: int):
|
def describe_rotation(
|
||||||
"""
|
page_context: PageContext, orient_conf: OrientationConfidence, correction: int
|
||||||
Describe the page rotation we are going to perform.
|
) -> str:
|
||||||
"""
|
"""Describe the page rotation we are going to perform (or not perform)."""
|
||||||
direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'}
|
direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'}
|
||||||
turns = {0: ' ', 90: '⬏', 180: '↻', 270: '⬑'}
|
turns = {0: ' ', 90: '⬏', 180: '↻', 270: '⬑'}
|
||||||
|
|
||||||
@@ -383,8 +403,8 @@ def describe_rotation(page_context: PageContext, orient_conf, correction: int):
|
|||||||
return f"{facing}, confidence {orient_conf.confidence:.2f} - {action}"
|
return f"{facing}, confidence {orient_conf.confidence:.2f} - {action}"
|
||||||
|
|
||||||
|
|
||||||
def get_orientation_correction(preview: Path, page_context: PageContext):
|
def get_orientation_correction(preview: Path, page_context: PageContext) -> int:
|
||||||
"""Work out orientation correct for each page.
|
"""Work out orientation correction for each page.
|
||||||
|
|
||||||
We ask Ghostscript to draw a preview page, which will rasterize with the
|
We ask Ghostscript to draw a preview page, which will rasterize with the
|
||||||
current /Rotate applied, and then ask OCR which way the page is
|
current /Rotate applied, and then ask OCR which way the page is
|
||||||
@@ -398,7 +418,6 @@ def get_orientation_correction(preview: Path, page_context: PageContext):
|
|||||||
which points it (hopefully) upright. _graft.py takes care of the orienting
|
which points it (hopefully) upright. _graft.py takes care of the orienting
|
||||||
the image and text layers.
|
the image and text layers.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
orient_conf = page_context.plugin_manager.hook.get_ocr_engine().get_orientation(
|
orient_conf = page_context.plugin_manager.hook.get_ocr_engine().get_orientation(
|
||||||
preview, page_context.options
|
preview, page_context.options
|
||||||
)
|
)
|
||||||
@@ -414,13 +433,58 @@ def get_orientation_correction(preview: Path, page_context: PageContext):
|
|||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|
||||||
|
def calculate_image_dpi(page_context: PageContext) -> Resolution:
|
||||||
|
pageinfo = page_context.pageinfo
|
||||||
|
dpi_profile = pageinfo.page_dpi_profile()
|
||||||
|
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
|
||||||
|
image_dpi = Resolution(dpi_profile.weighted_dpi, dpi_profile.weighted_dpi)
|
||||||
|
else:
|
||||||
|
image_dpi = pageinfo.dpi
|
||||||
|
return image_dpi
|
||||||
|
|
||||||
|
|
||||||
|
def calculate_raster_dpi(page_context: PageContext):
|
||||||
|
"""Calculate the DPI for rasterization."""
|
||||||
|
# Produce the page image with square resolution or else deskew and OCR
|
||||||
|
# will not work properly.
|
||||||
|
image_dpi = calculate_image_dpi(page_context)
|
||||||
|
dpi_profile = page_context.pageinfo.page_dpi_profile()
|
||||||
|
canvas_dpi = get_canvas_square_dpi(page_context, image_dpi)
|
||||||
|
page_dpi = get_page_square_dpi(page_context, image_dpi)
|
||||||
|
if dpi_profile and dpi_profile.average_to_max_dpi_ratio < 0.8:
|
||||||
|
log.warning(
|
||||||
|
"Weight average image DPI is %0.1f, max DPI is %0.1f. "
|
||||||
|
"The discrepancy may indicate a high detail region on this page, "
|
||||||
|
"but could also indicate a problem with the input PDF file. "
|
||||||
|
"Page image will be rendered at %0.1f DPI.",
|
||||||
|
dpi_profile.weighted_dpi,
|
||||||
|
dpi_profile.max_dpi,
|
||||||
|
canvas_dpi.to_scalar(),
|
||||||
|
)
|
||||||
|
return canvas_dpi, page_dpi
|
||||||
|
|
||||||
|
|
||||||
def rasterize(
|
def rasterize(
|
||||||
input_file: Path,
|
input_file: Path,
|
||||||
page_context: PageContext,
|
page_context: PageContext,
|
||||||
correction: int = 0,
|
correction: int = 0,
|
||||||
output_tag: str = '',
|
output_tag: str = '',
|
||||||
remove_vectors=None,
|
remove_vectors: bool | None = None,
|
||||||
):
|
) -> Path:
|
||||||
|
"""Rasterize a PDF page to a PNG image.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
input_file: The input PDF file path.
|
||||||
|
page_context: The page context object.
|
||||||
|
correction: The orientation correction angle. Defaults to 0.
|
||||||
|
output_tag: The output tag. Defaults to ''.
|
||||||
|
remove_vectors: Whether to remove vectors. Defaults to None, which means
|
||||||
|
the value from the page context options will be used. If the value
|
||||||
|
is True or False, it will override the page context options.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Path: The output PNG file path.
|
||||||
|
"""
|
||||||
colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m']
|
colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m']
|
||||||
device_idx = 0
|
device_idx = 0
|
||||||
|
|
||||||
@@ -451,10 +515,7 @@ def rasterize(
|
|||||||
|
|
||||||
log.debug(f"Rasterize with {device}, rotation {correction}")
|
log.debug(f"Rasterize with {device}, rotation {correction}")
|
||||||
|
|
||||||
# Produce the page image with square resolution or else deskew and OCR
|
canvas_dpi, page_dpi = calculate_raster_dpi(page_context)
|
||||||
# will not work properly.
|
|
||||||
canvas_dpi = get_canvas_square_dpi(pageinfo, page_context.options)
|
|
||||||
page_dpi = get_page_square_dpi(pageinfo, page_context.options)
|
|
||||||
|
|
||||||
page_context.plugin_manager.hook.rasterize_pdf_page(
|
page_context.plugin_manager.hook.rasterize_pdf_page(
|
||||||
input_file=input_file,
|
input_file=input_file,
|
||||||
@@ -465,24 +526,33 @@ def rasterize(
|
|||||||
pageno=pageinfo.pageno + 1,
|
pageno=pageinfo.pageno + 1,
|
||||||
rotation=correction,
|
rotation=correction,
|
||||||
filter_vector=remove_vectors,
|
filter_vector=remove_vectors,
|
||||||
|
stop_on_soft_error=not page_context.options.continue_on_soft_render_error,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def preprocess_remove_background(input_file: Path, page_context: PageContext):
|
def preprocess_remove_background(input_file: Path, page_context: PageContext) -> Path:
|
||||||
if any(image.bpc > 1 for image in page_context.pageinfo.images):
|
if any(image.bpc > 1 for image in page_context.pageinfo.images):
|
||||||
raise NotImplementedError("--remove-background is temporarily not implemented")
|
raise NotImplementedError("--remove-background is temporarily not implemented")
|
||||||
# output_file = page_context.get_path('pp_rm_bg.png')
|
# output_file = page_context.get_path('pp_rm_bg.png')
|
||||||
# leptonica.remove_background(input_file, output_file)
|
# leptonica.remove_background(input_file, output_file)
|
||||||
# return output_file
|
# return output_file
|
||||||
else:
|
log.info("background removal skipped on mono page")
|
||||||
log.info("background removal skipped on mono page")
|
return input_file
|
||||||
return input_file
|
|
||||||
|
|
||||||
|
|
||||||
def preprocess_deskew(input_file: Path, page_context: PageContext):
|
def preprocess_deskew(input_file: Path, page_context: PageContext) -> Path:
|
||||||
|
"""Deskews the input image using the OCR engine and saves the output to a file.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
input_file: The input image file to deskew.
|
||||||
|
page_context: The context of the page being processed.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Path: The path to the deskewed image file.
|
||||||
|
"""
|
||||||
output_file = page_context.get_path('pp_deskew.png')
|
output_file = page_context.get_path('pp_deskew.png')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context, calculate_image_dpi(page_context))
|
||||||
|
|
||||||
ocr_engine = page_context.plugin_manager.hook.get_ocr_engine()
|
ocr_engine = page_context.plugin_manager.hook.get_ocr_engine()
|
||||||
deskew_angle_degrees = ocr_engine.get_deskew(input_file, page_context.options)
|
deskew_angle_degrees = ocr_engine.get_deskew(input_file, page_context.options)
|
||||||
@@ -500,22 +570,23 @@ def preprocess_deskew(input_file: Path, page_context: PageContext):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def preprocess_clean(input_file: Path, page_context: PageContext):
|
def preprocess_clean(input_file: Path, page_context: PageContext) -> Path:
|
||||||
output_file = page_context.get_path('pp_clean.png')
|
output_file = page_context.get_path('pp_clean.png')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context, calculate_image_dpi(page_context))
|
||||||
return unpaper.clean(
|
return unpaper.clean(
|
||||||
input_file,
|
input_file,
|
||||||
output_file,
|
output_file,
|
||||||
dpi=dpi.x,
|
dpi=dpi.to_scalar(),
|
||||||
unpaper_args=page_context.options.unpaper_args,
|
unpaper_args=page_context.options.unpaper_args,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def create_ocr_image(image: Path, page_context: PageContext):
|
def create_ocr_image(image: Path, page_context: PageContext) -> Path:
|
||||||
"""Create the image we send for OCR. May not be the same as the display
|
"""Create the image we send for OCR.
|
||||||
image depending on preprocessing. This image will never be shown to the
|
|
||||||
user."""
|
|
||||||
|
|
||||||
|
Might not be the same as the display image depending on preprocessing.
|
||||||
|
This image will never be shown to the user.
|
||||||
|
"""
|
||||||
output_file = page_context.get_path('ocr.png')
|
output_file = page_context.get_path('ocr.png')
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
with Image.open(image) as im:
|
with Image.open(image) as im:
|
||||||
@@ -559,7 +630,7 @@ def create_ocr_image(image: Path, page_context: PageContext):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def ocr_engine_hocr(input_file: Path, page_context: PageContext):
|
def ocr_engine_hocr(input_file: Path, page_context: PageContext) -> tuple[Path, Path]:
|
||||||
hocr_out = page_context.get_path('ocr_hocr.hocr')
|
hocr_out = page_context.get_path('ocr_hocr.hocr')
|
||||||
hocr_text_out = page_context.get_path('ocr_hocr.txt')
|
hocr_text_out = page_context.get_path('ocr_hocr.txt')
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
@@ -574,9 +645,20 @@ def ocr_engine_hocr(input_file: Path, page_context: PageContext):
|
|||||||
return (hocr_out, hocr_text_out)
|
return (hocr_out, hocr_text_out)
|
||||||
|
|
||||||
|
|
||||||
def should_visible_page_image_use_jpg(pageinfo):
|
def should_visible_page_image_use_jpg(pageinfo: PageInfo) -> bool:
|
||||||
# If all images were JPEGs originally, produce a JPEG as output
|
"""Determines whether the visible page image should be saved as a JPEG.
|
||||||
return pageinfo.images and all(im.enc == Encoding.jpeg for im in pageinfo.images)
|
|
||||||
|
If all images were JPEGs originally, permit a JPEG as output.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
pageinfo: The PageInfo object containing information about the page.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
A boolean indicating whether the visible page image should be saved as a JPEG.
|
||||||
|
"""
|
||||||
|
return bool(pageinfo.images) and all(
|
||||||
|
im.enc == Encoding.jpeg for im in pageinfo.images
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def create_visible_page_jpg(image: Path, page_context: PageContext) -> Path:
|
def create_visible_page_jpg(image: Path, page_context: PageContext) -> Path:
|
||||||
@@ -591,7 +673,7 @@ def create_visible_page_jpg(image: Path, page_context: PageContext) -> Path:
|
|||||||
dpi = Resolution(*im.info['dpi'])
|
dpi = Resolution(*im.info['dpi'])
|
||||||
else:
|
else:
|
||||||
# Fallback to page-implied DPI
|
# Fallback to page-implied DPI
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context, calculate_image_dpi(page_context))
|
||||||
|
|
||||||
# Pillow requires integer DPI
|
# Pillow requires integer DPI
|
||||||
im.save(output_file, format='JPEG', dpi=dpi.to_int())
|
im.save(output_file, format='JPEG', dpi=dpi.to_int())
|
||||||
@@ -599,8 +681,8 @@ def create_visible_page_jpg(image: Path, page_context: PageContext) -> Path:
|
|||||||
|
|
||||||
|
|
||||||
def create_pdf_page_from_image(
|
def create_pdf_page_from_image(
|
||||||
image: Path, page_context: PageContext, orientation_correction
|
image: Path, page_context: PageContext, orientation_correction: int
|
||||||
):
|
) -> Path:
|
||||||
# We rasterize a square DPI version of each page because most image
|
# We rasterize a square DPI version of each page because most image
|
||||||
# processing tools don't support rectangular DPI. Use the square DPI as it
|
# processing tools don't support rectangular DPI. Use the square DPI as it
|
||||||
# accurately describes the image. It would be possible to resample the image
|
# accurately describes the image. It would be possible to resample the image
|
||||||
@@ -628,14 +710,13 @@ def create_pdf_page_from_image(
|
|||||||
output_file = page_context.plugin_manager.hook.filter_pdf_page(
|
output_file = page_context.plugin_manager.hook.filter_pdf_page(
|
||||||
page=page_context, image_filename=image, output_pdf=output_file
|
page=page_context, image_filename=image, output_pdf=output_file
|
||||||
)
|
)
|
||||||
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def render_hocr_page(hocr: Path, page_context: PageContext):
|
def render_hocr_page(hocr: Path, page_context: PageContext) -> Path:
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
output_file = page_context.get_path('ocr_hocr.pdf')
|
output_file = page_context.get_path('ocr_hocr.pdf')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, options)
|
dpi = get_page_square_dpi(page_context, calculate_image_dpi(page_context))
|
||||||
debug_mode = options.pdf_renderer == 'hocrdebug'
|
debug_mode = options.pdf_renderer == 'hocrdebug'
|
||||||
|
|
||||||
hocrtransform = HocrTransform(hocr_filename=hocr, dpi=dpi.x) # square
|
hocrtransform = HocrTransform(hocr_filename=hocr, dpi=dpi.x) # square
|
||||||
@@ -649,7 +730,9 @@ def render_hocr_page(hocr: Path, page_context: PageContext):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def ocr_engine_textonly_pdf(input_image: Path, page_context: PageContext):
|
def ocr_engine_textonly_pdf(
|
||||||
|
input_image: Path, page_context: PageContext
|
||||||
|
) -> tuple[Path, Path]:
|
||||||
output_pdf = page_context.get_path('ocr_tess.pdf')
|
output_pdf = page_context.get_path('ocr_tess.pdf')
|
||||||
output_text = page_context.get_path('ocr_tess.txt')
|
output_text = page_context.get_path('ocr_tess.txt')
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
@@ -695,13 +778,21 @@ def get_docinfo(base_pdf: pikepdf.Pdf, context: PdfContext) -> dict[str, str]:
|
|||||||
return pdfmark
|
return pdfmark
|
||||||
|
|
||||||
|
|
||||||
def generate_postscript_stub(context: PdfContext):
|
def generate_postscript_stub(context: PdfContext) -> Path:
|
||||||
|
"""Generates a PostScript file stub for the given PDF context.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
context: The PDF context to generate the PostScript file stub for.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Path: The path to the generated PostScript file stub.
|
||||||
|
"""
|
||||||
output_file = context.get_path('pdfa.ps')
|
output_file = context.get_path('pdfa.ps')
|
||||||
generate_pdfa_ps(output_file)
|
generate_pdfa_ps(output_file)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext):
|
def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext) -> Path:
|
||||||
options = context.options
|
options = context.options
|
||||||
input_pdfinfo = context.pdfinfo
|
input_pdfinfo = context.pdfinfo
|
||||||
fix_docinfo_file = context.get_path('fix_docinfo.pdf')
|
fix_docinfo_file = context.get_path('fix_docinfo.pdf')
|
||||||
@@ -712,21 +803,8 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext):
|
|||||||
# NULs in DocumentInfo seem to be common since older Acrobats included them.
|
# NULs in DocumentInfo seem to be common since older Acrobats included them.
|
||||||
# pikepdf can deal with this, but we make the world a better place by
|
# pikepdf can deal with this, but we make the world a better place by
|
||||||
# stamping them out as soon as possible.
|
# stamping them out as soon as possible.
|
||||||
modified = False
|
|
||||||
with pikepdf.open(input_pdf) as pdf_file:
|
with pikepdf.open(input_pdf) as pdf_file:
|
||||||
try:
|
if _repair_docinfo_nuls(pdf_file):
|
||||||
len(pdf_file.docinfo)
|
|
||||||
except TypeError:
|
|
||||||
log.error(
|
|
||||||
"File contains a malformed DocumentInfo block - continuing anyway"
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
if pdf_file.docinfo:
|
|
||||||
for k, v in pdf_file.docinfo.items():
|
|
||||||
if b'\x00' in bytes(v):
|
|
||||||
pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'')
|
|
||||||
modified = True
|
|
||||||
if modified:
|
|
||||||
pdf_file.save(fix_docinfo_file)
|
pdf_file.save(fix_docinfo_file)
|
||||||
else:
|
else:
|
||||||
safe_symlink(input_pdf, fix_docinfo_file)
|
safe_symlink(input_pdf, fix_docinfo_file)
|
||||||
@@ -736,26 +814,46 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext):
|
|||||||
pdf_pages=[fix_docinfo_file],
|
pdf_pages=[fix_docinfo_file],
|
||||||
pdfmark=input_ps_stub,
|
pdfmark=input_ps_stub,
|
||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
compression=options.pdfa_image_compression,
|
context=context,
|
||||||
pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3
|
pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3
|
||||||
progressbar_class=(
|
progressbar_class=(
|
||||||
context.plugin_manager.hook.get_progressbar_class()
|
context.plugin_manager.hook.get_progressbar_class()
|
||||||
if options.progress_bar
|
if options.progress_bar
|
||||||
else None
|
else None
|
||||||
),
|
),
|
||||||
|
stop_on_soft_error=not options.continue_on_soft_render_error,
|
||||||
)
|
)
|
||||||
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def should_linearize(working_file: Path, context: PdfContext):
|
def _repair_docinfo_nuls(pdf):
|
||||||
|
"""If the DocumentInfo block contains NUL characters, remove them.
|
||||||
|
|
||||||
|
If the DocumentInfo block is malformed, log an error and continue.
|
||||||
|
"""
|
||||||
|
modified = False
|
||||||
|
try:
|
||||||
|
if not isinstance(pdf.docinfo, pikepdf.Dictionary):
|
||||||
|
raise TypeError("DocumentInfo is not a dictionary")
|
||||||
|
for k, v in pdf.docinfo.items():
|
||||||
|
if isinstance(v, str) and b'\x00' in bytes(v):
|
||||||
|
pdf.docinfo[k] = bytes(v).replace(b'\x00', b'')
|
||||||
|
modified = True
|
||||||
|
except TypeError:
|
||||||
|
# TypeError can also be raised if dictionary items are unexpected types
|
||||||
|
log.error("File contains a malformed DocumentInfo block - continuing anyway.")
|
||||||
|
return modified
|
||||||
|
|
||||||
|
|
||||||
|
def should_linearize(working_file: Path, context: PdfContext) -> bool:
|
||||||
filesize = os.stat(working_file).st_size
|
filesize = os.stat(working_file).st_size
|
||||||
if filesize > (context.options.fast_web_view * 1_000_000):
|
if filesize > (context.options.fast_web_view * 1_000_000):
|
||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
def get_pdf_save_settings(output_type: str):
|
def get_pdf_save_settings(output_type: str) -> dict[str, Any]:
|
||||||
if output_type == 'pdfa-1':
|
if output_type == 'pdfa-1':
|
||||||
# Trigger recompression to ensure object streams are removed, because
|
# Trigger recompression to ensure object streams are removed, because
|
||||||
# Acrobat complains about them in PDF/A-1b validation.
|
# Acrobat complains about them in PDF/A-1b validation.
|
||||||
@@ -773,7 +871,7 @@ def get_pdf_save_settings(output_type: str):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def metadata_fixup(working_file: Path, context: PdfContext):
|
def metadata_fixup(working_file: Path, context: PdfContext) -> Path:
|
||||||
output_file = context.get_path('metafix.pdf')
|
output_file = context.get_path('metafix.pdf')
|
||||||
options = context.options
|
options = context.options
|
||||||
|
|
||||||
@@ -796,24 +894,45 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
|||||||
|
|
||||||
with pikepdf.open(context.origin) as original, pikepdf.open(working_file) as pdf:
|
with pikepdf.open(context.origin) as original, pikepdf.open(working_file) as pdf:
|
||||||
docinfo = get_docinfo(original, context)
|
docinfo = get_docinfo(original, context)
|
||||||
with pdf.open_metadata() as meta:
|
with pdf.open_metadata() as meta_pdf:
|
||||||
meta.load_from_docinfo(docinfo, delete_missing=False, raise_failure=False)
|
meta_pdf.load_from_docinfo(
|
||||||
|
docinfo, delete_missing=False, raise_failure=False
|
||||||
|
)
|
||||||
# If xmp:CreateDate is missing, set it to the modify date to
|
# If xmp:CreateDate is missing, set it to the modify date to
|
||||||
# match Ghostscript, for consistency
|
# ensure consistency with Ghostscript.
|
||||||
if 'xmp:CreateDate' not in meta:
|
if 'xmp:CreateDate' not in meta_pdf:
|
||||||
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
meta_pdf['xmp:CreateDate'] = meta_pdf.get('xmp:ModifyDate', '')
|
||||||
|
|
||||||
with original.open_metadata(
|
with original.open_metadata(
|
||||||
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
||||||
) as meta_original:
|
) as meta_original:
|
||||||
if meta.get('dc:title') == 'Untitled':
|
if meta_pdf.get('dc:title') == 'Untitled':
|
||||||
# Ghostscript likes to set title to Untitled if omitted from input.
|
# Ghostscript likes to set title to Untitled if omitted from input.
|
||||||
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
||||||
# and the XMP Spec do not make this recommendation.
|
# and the XMP Spec do not make this recommendation.
|
||||||
if 'dc:title' not in meta_original:
|
if 'dc:title' not in meta_original:
|
||||||
del meta['dc:title']
|
del meta_pdf['dc:title']
|
||||||
missing = set(meta_original.keys()) - set(meta.keys())
|
# If the user explicitly specified an empty string for any of the
|
||||||
report_on_metadata(missing)
|
# following, they should be unset and not reported as missing in
|
||||||
|
# the output pdf. Note that some metadata fields use differing names
|
||||||
|
# between PDF-A and PDF.
|
||||||
|
for meta in [meta_pdf, meta_original]:
|
||||||
|
if options.title == '' and 'dc:title' in meta:
|
||||||
|
del meta['dc:title'] # PDF-A and PDF
|
||||||
|
if options.author == '':
|
||||||
|
if 'dc:creator' in meta:
|
||||||
|
del meta['dc:creator'] # PDF-A (Not xmp:CreatorTool)
|
||||||
|
if 'pdf:Author' in meta:
|
||||||
|
del meta['pdf:Author'] # PDF
|
||||||
|
if options.subject == '':
|
||||||
|
if 'dc:description' in meta:
|
||||||
|
del meta['dc:description'] # PDF-A
|
||||||
|
if 'dc:subject' in meta:
|
||||||
|
del meta['dc:subject'] # PDF
|
||||||
|
if options.keywords == '' and 'pdf:Keywords' in meta:
|
||||||
|
del meta['pdf:Keywords'] # PDF-A and PDF
|
||||||
|
meta_missing = set(meta_original.keys()) - set(meta_pdf.keys())
|
||||||
|
report_on_metadata(meta_missing)
|
||||||
|
|
||||||
optimizing = context.plugin_manager.hook.is_optimization_enabled(
|
optimizing = context.plugin_manager.hook.is_optimization_enabled(
|
||||||
context=context
|
context=context
|
||||||
@@ -829,7 +948,32 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def optimize_pdf(input_file: Path, context: PdfContext, executor: Executor):
|
def _file_size_ratio(
|
||||||
|
input_file: Path, output_file: Path
|
||||||
|
) -> tuple[float | None, float | None]:
|
||||||
|
"""Calculate ratio of input to output file sizes and percentage savings.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
input_file (Path): The path to the input file.
|
||||||
|
output_file (Path): The path to the output file.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple[float | None, float | None]: A tuple containing the file size
|
||||||
|
ratio and the percentage savings achieved by the output file size
|
||||||
|
compared to the input file size.
|
||||||
|
"""
|
||||||
|
input_size = input_file.stat().st_size
|
||||||
|
output_size = output_file.stat().st_size
|
||||||
|
if output_size == 0:
|
||||||
|
return None, None
|
||||||
|
ratio = input_size / output_size
|
||||||
|
savings = 1 - output_size / input_size
|
||||||
|
return ratio, savings
|
||||||
|
|
||||||
|
|
||||||
|
def optimize_pdf(
|
||||||
|
input_file: Path, context: PdfContext, executor: Executor
|
||||||
|
) -> tuple[Path, Sequence[str]]:
|
||||||
output_file = context.get_path('optimize.pdf')
|
output_file = context.get_path('optimize.pdf')
|
||||||
output_pdf, messages = context.plugin_manager.hook.optimize_pdf(
|
output_pdf, messages = context.plugin_manager.hook.optimize_pdf(
|
||||||
input_pdf=input_file,
|
input_pdf=input_file,
|
||||||
@@ -839,17 +983,29 @@ def optimize_pdf(input_file: Path, context: PdfContext, executor: Executor):
|
|||||||
linearize=should_linearize(input_file, context),
|
linearize=should_linearize(input_file, context),
|
||||||
)
|
)
|
||||||
|
|
||||||
input_size = input_file.stat().st_size
|
ratio, savings = _file_size_ratio(input_file, output_file)
|
||||||
output_size = output_file.stat().st_size
|
if ratio:
|
||||||
if output_size > 0:
|
log.info(f"Image optimization ratio: {ratio:.2f} savings: {(savings):.1%}")
|
||||||
ratio = input_size / output_size
|
ratio, savings = _file_size_ratio(context.origin, output_file)
|
||||||
savings = 1 - output_size / input_size
|
if ratio:
|
||||||
log.info(f"Optimize ratio: {ratio:.2f} savings: {(savings):.1%}")
|
log.info(f"Total file size ratio: {ratio:.2f} savings: {(savings):.1%}")
|
||||||
|
|
||||||
return output_pdf, messages
|
return output_pdf, messages
|
||||||
|
|
||||||
|
|
||||||
def enumerate_compress_ranges(iterable):
|
def enumerate_compress_ranges(
|
||||||
|
iterable: Iterable,
|
||||||
|
) -> Iterator[tuple[tuple[int, int], Any]]:
|
||||||
|
"""Enumerate the ranges of non-empty elements in an iterable.
|
||||||
|
|
||||||
|
Compresses consecutive ranges of length 1 into single elements.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
iterable: An iterable of elements to enumerate.
|
||||||
|
|
||||||
|
Yields:
|
||||||
|
A tuple containing a range of indices and the corresponding element.
|
||||||
|
If the element is None, the range represents a skipped range of indices.
|
||||||
|
"""
|
||||||
skipped_from, index = None, None
|
skipped_from, index = None, None
|
||||||
for index, txt_file in enumerate(iterable):
|
for index, txt_file in enumerate(iterable):
|
||||||
index += 1
|
index += 1
|
||||||
@@ -865,7 +1021,7 @@ def enumerate_compress_ranges(iterable):
|
|||||||
yield (skipped_from, index), None
|
yield (skipped_from, index), None
|
||||||
|
|
||||||
|
|
||||||
def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext):
|
def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext) -> Path:
|
||||||
output_file = context.get_path('sidecar.txt')
|
output_file = context.get_path('sidecar.txt')
|
||||||
with open(output_file, 'w', encoding="utf-8") as stream:
|
with open(output_file, 'w', encoding="utf-8") as stream:
|
||||||
for (from_, to_), txt_file in enumerate_compress_ranges(txt_files):
|
for (from_, to_), txt_file in enumerate_compress_ranges(txt_files):
|
||||||
@@ -890,15 +1046,27 @@ def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def copy_final(input_file, output_file, _context: PdfContext):
|
def copy_final(
|
||||||
|
input_file: Path, output_file: str | Path | BinaryIO, _context: PdfContext
|
||||||
|
) -> None:
|
||||||
|
"""Copy the final temporary file to the output destination.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
input_file (Path): The input file to copy.
|
||||||
|
output_file (str | Path | BinaryIO): The output file to copy to.
|
||||||
|
_context (PdfContext): The PDF context.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
None
|
||||||
|
"""
|
||||||
log.debug('%s -> %s', input_file, output_file)
|
log.debug('%s -> %s', input_file, output_file)
|
||||||
with open(input_file, 'rb') as input_stream:
|
with input_file.open('rb') as input_stream:
|
||||||
if output_file == '-':
|
if output_file == '-':
|
||||||
copyfileobj(input_stream, sys.stdout.buffer)
|
copyfileobj(input_stream, sys.stdout.buffer) # type: ignore[misc]
|
||||||
sys.stdout.flush()
|
sys.stdout.flush()
|
||||||
elif hasattr(output_file, 'writable'):
|
elif hasattr(output_file, 'writable'):
|
||||||
output_stream = output_file
|
output_stream = cast(BinaryIO, output_file)
|
||||||
copyfileobj(input_stream, output_stream)
|
copyfileobj(input_stream, output_stream) # type: ignore[misc]
|
||||||
with suppress(AttributeError):
|
with suppress(AttributeError):
|
||||||
output_stream.flush()
|
output_stream.flush()
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -100,7 +100,9 @@ class OcrmypdfPluginManager(pluggy.PluginManager):
|
|||||||
self.register(module)
|
self.register(module)
|
||||||
|
|
||||||
|
|
||||||
def get_plugin_manager(plugins: list[str | Path], builtins=True):
|
def get_plugin_manager(
|
||||||
|
plugins: list[str | Path], builtins=True
|
||||||
|
) -> OcrmypdfPluginManager:
|
||||||
return OcrmypdfPluginManager(
|
return OcrmypdfPluginManager(
|
||||||
project_name='ocrmypdf',
|
project_name='ocrmypdf',
|
||||||
plugins=plugins,
|
plugins=plugins,
|
||||||
|
|||||||
+9
-10
@@ -1,4 +1,5 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2019-2022 James R. Barlow
|
||||||
|
# SPDX-FileCopyrightText: 2019 Martin Wind
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
||||||
@@ -206,7 +207,7 @@ def exec_page_sync(page_context: PageContext) -> PageResult:
|
|||||||
filtered_image = page_context.plugin_manager.hook.filter_page_image(
|
filtered_image = page_context.plugin_manager.hook.filter_page_image(
|
||||||
page=page_context, image_filename=visible_image_out
|
page=page_context, image_filename=visible_image_out
|
||||||
)
|
)
|
||||||
if filtered_image:
|
if filtered_image is not None: # None if no hook is present
|
||||||
visible_image_out = filtered_image
|
visible_image_out = filtered_image
|
||||||
pdf_page_from_image_out = create_pdf_page_from_image(
|
pdf_page_from_image_out = create_pdf_page_from_image(
|
||||||
visible_image_out, page_context, orientation_correction
|
visible_image_out, page_context, orientation_correction
|
||||||
@@ -250,8 +251,7 @@ def worker_init(max_pixels: int) -> None:
|
|||||||
|
|
||||||
|
|
||||||
def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
||||||
"""Execute the pipeline concurrently"""
|
"""Execute the pipeline concurrently."""
|
||||||
|
|
||||||
# Run exec_page_sync on every page context
|
# Run exec_page_sync on every page context
|
||||||
options = context.options
|
options = context.options
|
||||||
max_workers = min(len(context.pdfinfo), options.jobs)
|
max_workers = min(len(context.pdfinfo), options.jobs)
|
||||||
@@ -315,8 +315,7 @@ def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
|||||||
def configure_debug_logging(
|
def configure_debug_logging(
|
||||||
log_filename: Path, prefix: str = ''
|
log_filename: Path, prefix: str = ''
|
||||||
) -> logging.FileHandler:
|
) -> logging.FileHandler:
|
||||||
"""
|
"""Create a debug log file at a specified location.
|
||||||
Create a debug log file at a specified location.
|
|
||||||
|
|
||||||
Arguments:
|
Arguments:
|
||||||
log_filename: Where to the put the log file.
|
log_filename: Where to the put the log file.
|
||||||
@@ -419,13 +418,13 @@ def run_pipeline(
|
|||||||
options, start_input_file, options.output_file, optimize_messages
|
options, start_input_file, options.output_file, optimize_messages
|
||||||
)
|
)
|
||||||
|
|
||||||
except (KeyboardInterrupt if not api else NeverRaise):
|
except KeyboardInterrupt if not api else NeverRaise:
|
||||||
if options.verbose >= 1:
|
if options.verbose >= 1:
|
||||||
log.exception("KeyboardInterrupt")
|
log.exception("KeyboardInterrupt")
|
||||||
else:
|
else:
|
||||||
log.error("KeyboardInterrupt")
|
log.error("KeyboardInterrupt")
|
||||||
return ExitCode.ctrl_c
|
return ExitCode.ctrl_c
|
||||||
except (ExitCodeException if not api else NeverRaise) as e:
|
except ExitCodeException if not api else NeverRaise as e:
|
||||||
e = cast(ExitCodeException, e)
|
e = cast(ExitCodeException, e)
|
||||||
if options.verbose >= 1:
|
if options.verbose >= 1:
|
||||||
log.exception("ExitCodeException")
|
log.exception("ExitCodeException")
|
||||||
@@ -434,7 +433,7 @@ def run_pipeline(
|
|||||||
else:
|
else:
|
||||||
log.error(type(e).__name__)
|
log.error(type(e).__name__)
|
||||||
return e.exit_code
|
return e.exit_code
|
||||||
except (PIL.Image.DecompressionBombError if not api else NeverRaise):
|
except PIL.Image.DecompressionBombError if not api else NeverRaise:
|
||||||
log.exception(
|
log.exception(
|
||||||
"A decompression bomb error was encountered while executing the "
|
"A decompression bomb error was encountered while executing the "
|
||||||
"pipeline. Use the argument --max-image-mpixels to raise the maximum "
|
"pipeline. Use the argument --max-image-mpixels to raise the maximum "
|
||||||
@@ -452,7 +451,7 @@ def run_pipeline(
|
|||||||
"argument."
|
"argument."
|
||||||
)
|
)
|
||||||
return ExitCode.child_process_error
|
return ExitCode.child_process_error
|
||||||
except (Exception if not api else NeverRaise): # pylint: disable=broad-except
|
except Exception if not api else NeverRaise: # pylint: disable=broad-except
|
||||||
log.exception("An exception occurred while executing the pipeline")
|
log.exception("An exception occurred while executing the pipeline")
|
||||||
return ExitCode.other_error
|
return ExitCode.other_error
|
||||||
finally:
|
finally:
|
||||||
|
|||||||
+28
-28
@@ -43,35 +43,47 @@ log = logging.getLogger(__name__)
|
|||||||
|
|
||||||
|
|
||||||
def check_platform() -> None:
|
def check_platform() -> None:
|
||||||
if os.name == 'nt' and sys.maxsize <= 2**32: # pragma: no cover
|
if sys.maxsize <= 2**32: # pragma: no cover
|
||||||
# 32-bit interpreter on Windows
|
|
||||||
log.error(
|
log.error(
|
||||||
"You are running OCRmyPDF in a 32-bit (x86) Python interpreter."
|
"You are running OCRmyPDF in a 32-bit (x86) Python interpreter. "
|
||||||
|
"This is not supported. 32-bit does not have enough address space "
|
||||||
|
"to process large files. "
|
||||||
"Please use a 64-bit (x86-64) version of Python."
|
"Please use a 64-bit (x86-64) version of Python."
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def check_options_languages(options: Namespace, ocr_engine_languages: set[str]) -> None:
|
def check_options_languages(
|
||||||
|
options: Namespace, ocr_engine_languages: list[str]
|
||||||
|
) -> None:
|
||||||
if not options.languages:
|
if not options.languages:
|
||||||
options.languages = {DEFAULT_LANGUAGE}
|
options.languages = [DEFAULT_LANGUAGE]
|
||||||
system_lang = locale.getlocale()[0]
|
system_lang = locale.getlocale()[0]
|
||||||
if system_lang and not system_lang.startswith('en'):
|
if system_lang and not system_lang.startswith('en'):
|
||||||
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
||||||
if not ocr_engine_languages:
|
if not ocr_engine_languages:
|
||||||
return
|
return
|
||||||
missing_languages = options.languages - ocr_engine_languages
|
missing_languages = set(options.languages) - set(ocr_engine_languages)
|
||||||
if missing_languages:
|
if missing_languages:
|
||||||
|
lang_text = '\n'.join(lang for lang in missing_languages)
|
||||||
msg = (
|
msg = (
|
||||||
"OCR engine does not have language data for the following "
|
"OCR engine does not have language data for the following "
|
||||||
"requested languages: \n"
|
"requested languages: \n"
|
||||||
|
f"{lang_text}\n"
|
||||||
|
"Please install the appropriate language data for your OCR engine.\n"
|
||||||
|
"\n"
|
||||||
|
"See the online documentation for instructions:\n"
|
||||||
|
" https://ocrmypdf.readthedocs.io/en/latest/languages.html\n"
|
||||||
|
"\n"
|
||||||
|
"Note: most languages are identified by a 3-letter ISO 639-2 Code.\n"
|
||||||
|
"For example, English is 'eng', German is 'deu', and Spanish is 'spa'.\n"
|
||||||
|
"Simplified Chinese is 'chi_sim' and Traditional Chinese is 'chi_tra'."
|
||||||
|
"\n"
|
||||||
)
|
)
|
||||||
msg += '\n'.join(lang for lang in missing_languages)
|
|
||||||
msg += '\nNote: most languages are identified by a 3-digit ISO 639-2 Code'
|
|
||||||
raise MissingDependencyError(msg)
|
raise MissingDependencyError(msg)
|
||||||
|
|
||||||
|
|
||||||
def check_options_output(options: Namespace) -> None:
|
def check_options_output(options: Namespace) -> None:
|
||||||
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
is_latin = set(options.languages).issubset(HOCR_OK_LANGS)
|
||||||
|
|
||||||
if options.pdf_renderer.startswith('hocr') and not is_latin:
|
if options.pdf_renderer.startswith('hocr') and not is_latin:
|
||||||
log.warning(
|
log.warning(
|
||||||
@@ -109,12 +121,10 @@ def check_options_output(options: Namespace) -> None:
|
|||||||
def check_options_sidecar(options: Namespace) -> None:
|
def check_options_sidecar(options: Namespace) -> None:
|
||||||
if options.sidecar == '\0':
|
if options.sidecar == '\0':
|
||||||
if options.output_file == '-':
|
if options.output_file == '-':
|
||||||
raise BadArgsError(
|
raise BadArgsError("--sidecar filename needed when output file is stdout.")
|
||||||
"--sidecar filename must be specified when output file is stdout."
|
|
||||||
)
|
|
||||||
elif options.output_file == os.devnull:
|
elif options.output_file == os.devnull:
|
||||||
raise BadArgsError(
|
raise BadArgsError(
|
||||||
"--sidecar filename must be specified when output file is /dev/null or NUL."
|
"--sidecar filename needed when output file is /dev/null or NUL."
|
||||||
)
|
)
|
||||||
options.sidecar = options.output_file + '.txt'
|
options.sidecar = options.output_file + '.txt'
|
||||||
if options.sidecar == options.input_file or options.sidecar == options.output_file:
|
if options.sidecar == options.input_file or options.sidecar == options.output_file:
|
||||||
@@ -134,7 +144,7 @@ def check_options_preprocessing(options: Namespace) -> None:
|
|||||||
package='unpaper',
|
package='unpaper',
|
||||||
version_checker=unpaper.version,
|
version_checker=unpaper.version,
|
||||||
need_version='6.1',
|
need_version='6.1',
|
||||||
required_for=['--clean, --clean-final'],
|
required_for="--clean, --clean-final", # Problem arguments
|
||||||
)
|
)
|
||||||
try:
|
try:
|
||||||
if options.unpaper_args:
|
if options.unpaper_args:
|
||||||
@@ -195,16 +205,6 @@ def check_options_ocr_behavior(options: Namespace) -> None:
|
|||||||
options.pages = _pages_from_ranges(options.pages)
|
options.pages = _pages_from_ranges(options.pages)
|
||||||
|
|
||||||
|
|
||||||
def check_options_advanced(options: Namespace) -> None:
|
|
||||||
if options.pdfa_image_compression != 'auto' and not options.output_type.startswith(
|
|
||||||
'pdfa'
|
|
||||||
):
|
|
||||||
log.warning(
|
|
||||||
"--pdfa-image-compression argument only applies when "
|
|
||||||
"--output-type is one of 'pdfa', 'pdfa-1', or 'pdfa-2'"
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def check_options_metadata(options: Namespace) -> None:
|
def check_options_metadata(options: Namespace) -> None:
|
||||||
docinfo = [options.title, options.author, options.keywords, options.subject]
|
docinfo = [options.title, options.author, options.keywords, options.subject]
|
||||||
for s in (m for m in docinfo if m):
|
for s in (m for m in docinfo if m):
|
||||||
@@ -221,7 +221,7 @@ def check_options_metadata(options: Namespace) -> None:
|
|||||||
def check_options_pillow(options: Namespace) -> None:
|
def check_options_pillow(options: Namespace) -> None:
|
||||||
PIL.Image.MAX_IMAGE_PIXELS = int(options.max_image_mpixels * 1_000_000)
|
PIL.Image.MAX_IMAGE_PIXELS = int(options.max_image_mpixels * 1_000_000)
|
||||||
if PIL.Image.MAX_IMAGE_PIXELS == 0:
|
if PIL.Image.MAX_IMAGE_PIXELS == 0:
|
||||||
PIL.Image.MAX_IMAGE_PIXELS = None
|
PIL.Image.MAX_IMAGE_PIXELS = None # type: ignore
|
||||||
|
|
||||||
|
|
||||||
def _check_plugin_invariant_options(options: Namespace) -> None:
|
def _check_plugin_invariant_options(options: Namespace) -> None:
|
||||||
@@ -231,7 +231,6 @@ def _check_plugin_invariant_options(options: Namespace) -> None:
|
|||||||
check_options_sidecar(options)
|
check_options_sidecar(options)
|
||||||
check_options_preprocessing(options)
|
check_options_preprocessing(options)
|
||||||
check_options_ocr_behavior(options)
|
check_options_ocr_behavior(options)
|
||||||
check_options_advanced(options)
|
|
||||||
check_options_pillow(options)
|
check_options_pillow(options)
|
||||||
|
|
||||||
|
|
||||||
@@ -276,7 +275,8 @@ def create_input_file(options: Namespace, work_folder: Path) -> tuple[Path, str]
|
|||||||
"permissions correctly.\n"
|
"permissions correctly.\n"
|
||||||
"You may find it easier to use stdin/stdout:"
|
"You may find it easier to use stdin/stdout:"
|
||||||
"\n"
|
"\n"
|
||||||
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf\n"
|
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf"
|
||||||
|
"\n"
|
||||||
)
|
)
|
||||||
raise InputFileError(msg) from e
|
raise InputFileError(msg) from e
|
||||||
|
|
||||||
@@ -333,7 +333,7 @@ def report_output_file_size(
|
|||||||
for arg in image_preproc:
|
for arg in image_preproc:
|
||||||
if getattr(options, arg, False):
|
if getattr(options, arg, False):
|
||||||
reasons.append(
|
reasons.append(
|
||||||
f"The argument --{arg.replace('_', '-')} was issued, causing transcoding."
|
f"--{arg.replace('_', '-')} was issued, causing transcoding."
|
||||||
)
|
)
|
||||||
|
|
||||||
reasons.extend(optimize_messages)
|
reasons.extend(optimize_messages)
|
||||||
|
|||||||
@@ -8,10 +8,7 @@ OCRmyPDF uses setuptools_scm to derive version from git tags.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
try:
|
from importlib.metadata import version as _package_version
|
||||||
from importlib.metadata import version as _package_version
|
|
||||||
except ImportError:
|
|
||||||
from importlib_metadata import version as _package_version # type: ignore
|
|
||||||
|
|
||||||
PROGRAM_NAME = 'ocrmypdf'
|
PROGRAM_NAME = 'ocrmypdf'
|
||||||
|
|
||||||
|
|||||||
+78
-66
@@ -9,28 +9,22 @@ import logging
|
|||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
|
from argparse import Namespace
|
||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
from io import IOBase
|
from io import IOBase
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import AnyStr, BinaryIO, Iterable, Union
|
from typing import AnyStr, BinaryIO, Iterable, Union
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
from ocrmypdf._logging import PageNumberFilter, TqdmConsole
|
import pluggy
|
||||||
|
|
||||||
|
from ocrmypdf._logging import PageNumberFilter
|
||||||
from ocrmypdf._plugin_manager import get_plugin_manager
|
from ocrmypdf._plugin_manager import get_plugin_manager
|
||||||
from ocrmypdf._sync import run_pipeline
|
from ocrmypdf._sync import run_pipeline
|
||||||
from ocrmypdf._validation import check_options
|
from ocrmypdf._validation import check_options
|
||||||
from ocrmypdf.cli import ArgumentParser, get_parser
|
from ocrmypdf.cli import ArgumentParser, get_parser
|
||||||
from ocrmypdf.helpers import is_iterable_notstr
|
from ocrmypdf.helpers import is_iterable_notstr
|
||||||
|
|
||||||
try:
|
|
||||||
import coloredlogs
|
|
||||||
except ModuleNotFoundError:
|
|
||||||
coloredlogs = None # pylint: disable=invalid-name
|
|
||||||
|
|
||||||
if coloredlogs:
|
|
||||||
from humanfriendly.terminal import enable_ansi_support
|
|
||||||
|
|
||||||
|
|
||||||
StrPath = Union[Path, AnyStr]
|
StrPath = Union[Path, AnyStr]
|
||||||
PathOrIO = Union[BinaryIO, StrPath]
|
PathOrIO = Union[BinaryIO, StrPath]
|
||||||
|
|
||||||
@@ -52,7 +46,7 @@ def configure_logging(
|
|||||||
*,
|
*,
|
||||||
progress_bar_friendly: bool = True,
|
progress_bar_friendly: bool = True,
|
||||||
manage_root_logger: bool = False,
|
manage_root_logger: bool = False,
|
||||||
plugin_manager=None,
|
plugin_manager: pluggy.PluginManager | None = None,
|
||||||
):
|
):
|
||||||
"""Set up logging.
|
"""Set up logging.
|
||||||
|
|
||||||
@@ -92,7 +86,6 @@ def configure_logging(
|
|||||||
Returns:
|
Returns:
|
||||||
The toplevel logger for ocrmypdf (or the root logger, if we are managing it).
|
The toplevel logger for ocrmypdf (or the root logger, if we are managing it).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
prefix = '' if manage_root_logger else 'ocrmypdf'
|
prefix = '' if manage_root_logger else 'ocrmypdf'
|
||||||
|
|
||||||
log = logging.getLogger(prefix)
|
log = logging.getLogger(prefix)
|
||||||
@@ -119,14 +112,7 @@ def configure_logging(
|
|||||||
else:
|
else:
|
||||||
fmt = '%(pageno)s%(message)s'
|
fmt = '%(pageno)s%(message)s'
|
||||||
|
|
||||||
use_colors = progress_bar_friendly
|
|
||||||
formatter = None
|
formatter = None
|
||||||
if coloredlogs and use_colors:
|
|
||||||
use_colors = enable_ansi_support()
|
|
||||||
if use_colors:
|
|
||||||
use_colors = coloredlogs.terminal_supports_colors()
|
|
||||||
if use_colors:
|
|
||||||
formatter = coloredlogs.ColoredFormatter(fmt=fmt)
|
|
||||||
|
|
||||||
if not formatter:
|
if not formatter:
|
||||||
formatter = logging.Formatter(fmt=fmt)
|
formatter = logging.Formatter(fmt=fmt)
|
||||||
@@ -148,7 +134,21 @@ def configure_logging(
|
|||||||
|
|
||||||
def create_options(
|
def create_options(
|
||||||
*, input_file: PathOrIO, output_file: PathOrIO, parser: ArgumentParser, **kwargs
|
*, input_file: PathOrIO, output_file: PathOrIO, parser: ArgumentParser, **kwargs
|
||||||
):
|
) -> Namespace:
|
||||||
|
"""Construct an options object from the input/output files and keyword arguments.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
input_file: Input file path or file object.
|
||||||
|
output_file: Output file path or file object.
|
||||||
|
parser: ArgumentParser object.
|
||||||
|
**kwargs: Keyword arguments.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
argparse.Namespace: A Namespace object containing the parsed arguments.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
TypeError: If the type of a keyword argument is not supported.
|
||||||
|
"""
|
||||||
cmdline = []
|
cmdline = []
|
||||||
deferred = []
|
deferred = []
|
||||||
|
|
||||||
@@ -209,59 +209,73 @@ def create_options(
|
|||||||
return options
|
return options
|
||||||
|
|
||||||
|
|
||||||
def ocr( # pylint: disable=unused-argument
|
def ocr( # noqa: ruff: disable=D417
|
||||||
input_file: PathOrIO,
|
input_file: PathOrIO,
|
||||||
output_file: PathOrIO,
|
output_file: PathOrIO,
|
||||||
*,
|
*,
|
||||||
language: Iterable[str] = None,
|
language: Iterable[str] | None = None,
|
||||||
image_dpi: int = None,
|
image_dpi: int | None = None,
|
||||||
output_type=None,
|
output_type: str | None = None,
|
||||||
sidecar: StrPath | None = None,
|
sidecar: StrPath | None = None,
|
||||||
jobs: int = None,
|
jobs: int | None = None,
|
||||||
use_threads: bool = None,
|
use_threads: bool | None = None,
|
||||||
title: str = None,
|
title: str | None = None,
|
||||||
author: str = None,
|
author: str | None = None,
|
||||||
subject: str = None,
|
subject: str | None = None,
|
||||||
keywords: str = None,
|
keywords: str | None = None,
|
||||||
rotate_pages: bool = None,
|
rotate_pages: bool | None = None,
|
||||||
remove_background: bool = None,
|
remove_background: bool | None = None,
|
||||||
deskew: bool = None,
|
deskew: bool | None = None,
|
||||||
clean: bool = None,
|
clean: bool | None = None,
|
||||||
clean_final: bool = None,
|
clean_final: bool | None = None,
|
||||||
unpaper_args: str = None,
|
unpaper_args: str | None = None,
|
||||||
oversample: int = None,
|
oversample: int | None = None,
|
||||||
remove_vectors: bool = None,
|
remove_vectors: bool | None = None,
|
||||||
force_ocr: bool = None,
|
force_ocr: bool | None = None,
|
||||||
skip_text: bool = None,
|
skip_text: bool | None = None,
|
||||||
redo_ocr: bool = None,
|
redo_ocr: bool | None = None,
|
||||||
skip_big: float = None,
|
skip_big: float | None = None,
|
||||||
optimize: int = None,
|
optimize: int | None = None,
|
||||||
jpg_quality: int = None,
|
jpg_quality: int | None = None,
|
||||||
png_quality: int = None,
|
png_quality: int | None = None,
|
||||||
jbig2_lossy: bool = None,
|
jbig2_lossy: bool | None = None,
|
||||||
jbig2_page_group_size: int = None,
|
jbig2_page_group_size: int | None = None,
|
||||||
pages: str = None,
|
pages: str | None = None,
|
||||||
max_image_mpixels: float = None,
|
max_image_mpixels: float | None = None,
|
||||||
tesseract_config: Iterable[str] = None,
|
tesseract_config: Iterable[str] | None = None,
|
||||||
tesseract_pagesegmode: int = None,
|
tesseract_pagesegmode: int | None = None,
|
||||||
tesseract_oem: int = None,
|
tesseract_oem: int | None = None,
|
||||||
tesseract_thresholding: int = None,
|
tesseract_thresholding: int | None = None,
|
||||||
pdf_renderer=None,
|
pdf_renderer: str | None = None,
|
||||||
tesseract_timeout: float = None,
|
tesseract_timeout: float | None = None,
|
||||||
rotate_pages_threshold: float = None,
|
tesseract_non_ocr_timeout: float | None = None,
|
||||||
pdfa_image_compression=None,
|
rotate_pages_threshold: float | None = None,
|
||||||
user_words: os.PathLike = None,
|
pdfa_image_compression: str | None = None,
|
||||||
user_patterns: os.PathLike = None,
|
user_words: os.PathLike | None = None,
|
||||||
fast_web_view: float = None,
|
user_patterns: os.PathLike | None = None,
|
||||||
plugins: Iterable[StrPath] = None,
|
fast_web_view: float | None = None,
|
||||||
|
continue_on_soft_render_error: bool | None = None,
|
||||||
|
plugins: Iterable[StrPath] | None = None,
|
||||||
plugin_manager=None,
|
plugin_manager=None,
|
||||||
keep_temporary_files: bool = None,
|
keep_temporary_files: bool | None = None,
|
||||||
progress_bar: bool = None,
|
progress_bar: bool | None = None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
"""Run OCRmyPDF on one PDF or image.
|
"""Run OCRmyPDF on one PDF or image.
|
||||||
|
|
||||||
For most arguments, see documentation for the equivalent command line parameter.
|
For most arguments, see documentation for the equivalent command line parameter.
|
||||||
|
|
||||||
|
This API takes a threading lock, because OCRmyPDF uses global state in particular
|
||||||
|
for the plugin system. The jobs parameter will be used to create a pool of
|
||||||
|
worker threads or processes at different times, subject to change. A Python
|
||||||
|
process can only run one OCRmyPDF task at a time.
|
||||||
|
|
||||||
|
To run parallelize instances OCRmyPDF, use separate Python processes to scale
|
||||||
|
horizontally. Generally speaking you should set jobs=sqrt(cpu_count) and run
|
||||||
|
sqrt(cpu_count) processes as a starting point. If you have files with a high page
|
||||||
|
count, run fewer processes and more jobs per process. If you have a lot of short
|
||||||
|
files, run more processes and fewer jobs per process.
|
||||||
|
|
||||||
A few specific arguments are discussed here:
|
A few specific arguments are discussed here:
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
@@ -283,9 +297,8 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
When a stream is used as output, whether via a writable object or
|
When a stream is used as output, whether via a writable object or
|
||||||
``"-"``, some final validation steps are not performed (we do not read
|
``"-"``, some final validation steps are not performed (we do not read
|
||||||
back the stream after it is written).
|
back the stream after it is written).
|
||||||
|
|
||||||
Raises:
|
Raises:
|
||||||
ocrmypdf.PdfMergeFailedError: If the input PDF is malformed, preventing merging
|
|
||||||
with the OCR layer.
|
|
||||||
ocrmypdf.MissingDependencyError: If a required dependency program is missing or
|
ocrmypdf.MissingDependencyError: If a required dependency program is missing or
|
||||||
was not found on PATH.
|
was not found on PATH.
|
||||||
ocrmypdf.UnsupportedImageFormatError: If the input file type was an image that
|
ocrmypdf.UnsupportedImageFormatError: If the input file type was an image that
|
||||||
@@ -342,7 +355,6 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
|
|
||||||
__all__ = [
|
__all__ = [
|
||||||
'PageNumberFilter',
|
'PageNumberFilter',
|
||||||
'TqdmConsole',
|
|
||||||
'Verbosity',
|
'Verbosity',
|
||||||
'check_options',
|
'check_options',
|
||||||
'configure_logging',
|
'configure_logging',
|
||||||
|
|||||||
@@ -1,8 +1,6 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
from __future__ import annotations
|
"""Plugins in this package are automatically loaded by ocrmypdf."""
|
||||||
|
|
||||||
# This file exists only mark builtin_plugins as a package.
|
from __future__ import annotations
|
||||||
# The plugin manager will not load it, so anything defined here may not be
|
|
||||||
# processed as a module.
|
|
||||||
|
|||||||
@@ -1,9 +1,9 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
"""OCRmyPDF's multiprocessing/multithreading abstraction layer."""
|
"""OCRmyPDF's multiprocessing/multithreading abstraction layer."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import logging.handlers
|
import logging.handlers
|
||||||
import multiprocessing
|
import multiprocessing
|
||||||
@@ -16,10 +16,10 @@ from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_compl
|
|||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from typing import Callable, Iterable, Type, Union
|
from typing import Callable, Iterable, Type, Union
|
||||||
|
|
||||||
from tqdm import tqdm
|
from rich.console import Console as RichConsole
|
||||||
|
|
||||||
from ocrmypdf import Executor, hookimpl
|
from ocrmypdf import Executor, hookimpl
|
||||||
from ocrmypdf._logging import TqdmConsole
|
from ocrmypdf._logging import RichLoggingHandler, RichTqdmProgressAdapter
|
||||||
from ocrmypdf.exceptions import InputFileError
|
from ocrmypdf.exceptions import InputFileError
|
||||||
from ocrmypdf.helpers import remove_all_log_handlers
|
from ocrmypdf.helpers import remove_all_log_handlers
|
||||||
|
|
||||||
@@ -30,7 +30,7 @@ WorkerInit = Callable[[Queue, UserInit, int], None]
|
|||||||
|
|
||||||
|
|
||||||
def log_listener(q: Queue):
|
def log_listener(q: Queue):
|
||||||
"""Listen to the worker processes and forward the messages to logging
|
"""Listen to the worker processes and forward the messages to logging.
|
||||||
|
|
||||||
For simplicity this is a thread rather than a process. Only one process
|
For simplicity this is a thread rather than a process. Only one process
|
||||||
should actually write to sys.stderr or whatever we're using, so if this is
|
should actually write to sys.stderr or whatever we're using, so if this is
|
||||||
@@ -39,7 +39,6 @@ def log_listener(q: Queue):
|
|||||||
See:
|
See:
|
||||||
https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes
|
https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes
|
||||||
"""
|
"""
|
||||||
|
|
||||||
while True:
|
while True:
|
||||||
try:
|
try:
|
||||||
record = q.get()
|
record = q.get()
|
||||||
@@ -55,12 +54,12 @@ def log_listener(q: Queue):
|
|||||||
|
|
||||||
|
|
||||||
def process_sigbus(*args):
|
def process_sigbus(*args):
|
||||||
|
"""Handle SIGBUS signal at the worker level."""
|
||||||
raise InputFileError("A worker process lost access to an input file")
|
raise InputFileError("A worker process lost access to an input file")
|
||||||
|
|
||||||
|
|
||||||
def process_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
def process_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
||||||
"""Initialize a process pool worker"""
|
"""Initialize a process pool worker."""
|
||||||
|
|
||||||
# Ignore SIGINT (our parent process will kill us gracefully)
|
# Ignore SIGINT (our parent process will kill us gracefully)
|
||||||
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
||||||
|
|
||||||
@@ -69,7 +68,7 @@ def process_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
|||||||
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
|
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
|
||||||
signal.signal(signal.SIGBUS, process_sigbus)
|
signal.signal(signal.SIGBUS, process_sigbus)
|
||||||
|
|
||||||
# Remove any log handlers that belong to the parent process
|
# Remove any log handlers inherited from the parent process
|
||||||
root = logging.getLogger()
|
root = logging.getLogger()
|
||||||
remove_all_log_handlers(root)
|
remove_all_log_handlers(root)
|
||||||
|
|
||||||
@@ -82,6 +81,7 @@ def process_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
|||||||
|
|
||||||
|
|
||||||
def thread_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
def thread_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
||||||
|
"""Begin a thread pool worker."""
|
||||||
del q # unused but required argument
|
del q # unused but required argument
|
||||||
del loglevel # unused but required argument
|
del loglevel # unused but required argument
|
||||||
# As a thread, block SIGBUS so the main thread deals with it...
|
# As a thread, block SIGBUS so the main thread deals with it...
|
||||||
@@ -95,15 +95,6 @@ def thread_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
|||||||
class StandardExecutor(Executor):
|
class StandardExecutor(Executor):
|
||||||
"""Standard OCRmyPDF concurrent task executor."""
|
"""Standard OCRmyPDF concurrent task executor."""
|
||||||
|
|
||||||
def _cancel_futures_kwargs(self):
|
|
||||||
"""Shim older Pythons that do not have Executor.shutdown(...cancel_futures=).
|
|
||||||
|
|
||||||
Remove this code when support for Python 3.8 is dropped.
|
|
||||||
"""
|
|
||||||
if sys.version_info[:2] < (3, 9):
|
|
||||||
return {}
|
|
||||||
return dict(cancel_futures=True)
|
|
||||||
|
|
||||||
def _execute(
|
def _execute(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
@@ -142,7 +133,7 @@ class StandardExecutor(Executor):
|
|||||||
task_finished(result, pbar)
|
task_finished(result, pbar)
|
||||||
except KeyboardInterrupt:
|
except KeyboardInterrupt:
|
||||||
# Terminate pool so we exit instantly
|
# Terminate pool so we exit instantly
|
||||||
executor.shutdown(wait=False, **self._cancel_futures_kwargs())
|
executor.shutdown(wait=False, cancel_futures=True)
|
||||||
raise
|
raise
|
||||||
except Exception:
|
except Exception:
|
||||||
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
||||||
@@ -151,7 +142,7 @@ class StandardExecutor(Executor):
|
|||||||
# results will be discard. But if the condition above is True,
|
# results will be discard. But if the condition above is True,
|
||||||
# then we are running in pytest, and we want everything to exit
|
# then we are running in pytest, and we want everything to exit
|
||||||
# as cleanly as possible so that we get good error messages.
|
# as cleanly as possible so that we get good error messages.
|
||||||
executor.shutdown(wait=False, **self._cancel_futures_kwargs())
|
executor.shutdown(wait=False, cancel_futures=True)
|
||||||
raise
|
raise
|
||||||
finally:
|
finally:
|
||||||
# Terminate log listener
|
# Terminate log listener
|
||||||
@@ -164,14 +155,24 @@ class StandardExecutor(Executor):
|
|||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def get_executor(progressbar_class):
|
def get_executor(progressbar_class):
|
||||||
|
"""Return the default executor."""
|
||||||
return StandardExecutor(pbar_class=progressbar_class)
|
return StandardExecutor(pbar_class=progressbar_class)
|
||||||
|
|
||||||
|
|
||||||
|
RICH_CONSOLE = RichConsole(stderr=True)
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def get_progressbar_class():
|
def get_progressbar_class():
|
||||||
return tqdm
|
"""Return the default progress bar class."""
|
||||||
|
|
||||||
|
def partial_RichTqdmProgressAdapter(*args, **kwargs):
|
||||||
|
return RichTqdmProgressAdapter(*args, **kwargs, console=RICH_CONSOLE)
|
||||||
|
|
||||||
|
return partial_RichTqdmProgressAdapter
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def get_logging_console():
|
def get_logging_console():
|
||||||
return logging.StreamHandler(stream=TqdmConsole(sys.stderr))
|
"""Return the default logging console handler."""
|
||||||
|
return RichLoggingHandler(console=RICH_CONSOLE)
|
||||||
|
|||||||
@@ -8,48 +8,72 @@ import logging
|
|||||||
|
|
||||||
from ocrmypdf import hookimpl
|
from ocrmypdf import hookimpl
|
||||||
from ocrmypdf._exec import ghostscript
|
from ocrmypdf._exec import ghostscript
|
||||||
from ocrmypdf._validation import HOCR_OK_LANGS
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
from ocrmypdf.subprocess import check_external_program
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
# Currently all blacklisted versions are lower than 9.55, so none need to
|
||||||
|
# be added here. If a future version is blacklisted, add it here.
|
||||||
|
BLACKLISTED_GS_VERSIONS: frozenset[str] = frozenset()
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def add_options(parser):
|
||||||
|
gs = parser.add_argument_group("Ghostscript", "Advanced control of Ghostscript")
|
||||||
|
gs.add_argument(
|
||||||
|
'--color-conversion-strategy',
|
||||||
|
action='store',
|
||||||
|
type=str,
|
||||||
|
metavar='STRATEGY',
|
||||||
|
choices=ghostscript.COLOR_CONVERSION_STRATEGIES,
|
||||||
|
default='LeaveColorUnchanged',
|
||||||
|
help="Set Ghostscript color conversion strategy",
|
||||||
|
)
|
||||||
|
gs.add_argument(
|
||||||
|
'--pdfa-image-compression',
|
||||||
|
choices=['auto', 'jpeg', 'lossless'],
|
||||||
|
default='auto',
|
||||||
|
help="Specify how to compress images in the output PDF/A. 'auto' lets "
|
||||||
|
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
|
||||||
|
"JPEG compression. 'lossless' uses PNG-style lossless compression "
|
||||||
|
"for all images. Monochrome images are always compressed using a "
|
||||||
|
"lossless codec. Compression settings "
|
||||||
|
"are applied to all pages, including those for which OCR was "
|
||||||
|
"skipped. Not supported for --output-type=pdf ; that setting "
|
||||||
|
"preserves the original compression of all images.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def check_options(options):
|
def check_options(options):
|
||||||
|
"""Check that the options are valid for this plugin."""
|
||||||
check_external_program(
|
check_external_program(
|
||||||
program='gs',
|
program='gs',
|
||||||
package='ghostscript',
|
package='ghostscript',
|
||||||
version_checker=ghostscript.version,
|
version_checker=ghostscript.version,
|
||||||
need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports
|
need_version='9.55', # Ubuntu 22.04's version
|
||||||
)
|
)
|
||||||
gs_version = ghostscript.version()
|
gs_version = ghostscript.version()
|
||||||
if gs_version in ('9.24', '9.51'):
|
if gs_version in BLACKLISTED_GS_VERSIONS:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
f"Ghostscript {gs_version} contains serious regressions and is not "
|
f"Ghostscript {gs_version} contains serious regressions and is not "
|
||||||
"supported. Please upgrade to a newer version, or downgrade to the "
|
"supported. Please upgrade to a newer version, or downgrade to the "
|
||||||
"previous version."
|
"previous version."
|
||||||
)
|
)
|
||||||
|
|
||||||
# We have these constraints to check for.
|
|
||||||
# 1. Ghostscript < 9.20 mangles multibyte Unicode
|
|
||||||
# 2. hocr doesn't work on non-Latin languages (so don't select it)
|
|
||||||
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
|
||||||
if gs_version < '9.20' and options.output_type != 'pdf' and not is_latin:
|
|
||||||
# https://bugs.ghostscript.com/show_bug.cgi?id=696874
|
|
||||||
# Ghostscript < 9.20 fails to encode multibyte characters properly
|
|
||||||
log.warning(
|
|
||||||
f"The installed version of Ghostscript ({gs_version}) does not work "
|
|
||||||
"correctly with the OCR languages you specified. Use --output-type pdf or "
|
|
||||||
"upgrade to Ghostscript 9.20 or later to avoid this issue."
|
|
||||||
)
|
|
||||||
|
|
||||||
if options.output_type == 'pdfa':
|
if options.output_type == 'pdfa':
|
||||||
options.output_type = 'pdfa-2'
|
options.output_type = 'pdfa-2'
|
||||||
|
if options.color_conversion_strategy not in ghostscript.COLOR_CONVERSION_STRATEGIES:
|
||||||
if options.output_type == 'pdfa-3' and ghostscript.version() < '9.19':
|
raise ValueError(
|
||||||
raise MissingDependencyError(
|
f"Invalid color conversion strategy: {options.color_conversion_strategy}"
|
||||||
"--output-type pdfa-3 requires Ghostscript 9.19 or later"
|
)
|
||||||
|
if options.pdfa_image_compression != 'auto' and not options.output_type.startswith(
|
||||||
|
'pdfa'
|
||||||
|
):
|
||||||
|
log.warning(
|
||||||
|
"--pdfa-image-compression argument only applies when "
|
||||||
|
"--output-type is one of 'pdfa', 'pdfa-1', or 'pdfa-2'"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -63,7 +87,9 @@ def rasterize_pdf_page(
|
|||||||
page_dpi,
|
page_dpi,
|
||||||
rotation,
|
rotation,
|
||||||
filter_vector,
|
filter_vector,
|
||||||
|
stop_on_soft_error,
|
||||||
):
|
):
|
||||||
|
"""Rasterize a single page of a PDF file using Ghostscript."""
|
||||||
ghostscript.rasterize_pdf(
|
ghostscript.rasterize_pdf(
|
||||||
input_file,
|
input_file,
|
||||||
output_file,
|
output_file,
|
||||||
@@ -73,6 +99,7 @@ def rasterize_pdf_page(
|
|||||||
page_dpi=page_dpi,
|
page_dpi=page_dpi,
|
||||||
rotation=rotation,
|
rotation=rotation,
|
||||||
filter_vector=filter_vector,
|
filter_vector=filter_vector,
|
||||||
|
stop_on_error=stop_on_soft_error,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
@@ -82,17 +109,21 @@ def generate_pdfa(
|
|||||||
pdf_pages,
|
pdf_pages,
|
||||||
pdfmark,
|
pdfmark,
|
||||||
output_file,
|
output_file,
|
||||||
compression,
|
context,
|
||||||
pdf_version,
|
pdf_version,
|
||||||
pdfa_part,
|
pdfa_part,
|
||||||
progressbar_class,
|
progressbar_class,
|
||||||
|
stop_on_soft_error,
|
||||||
):
|
):
|
||||||
|
"""Generate a PDF/A from the list of PDF pages and PDF/A metadata."""
|
||||||
ghostscript.generate_pdfa(
|
ghostscript.generate_pdfa(
|
||||||
pdf_pages=[*pdf_pages, pdfmark],
|
pdf_pages=[*pdf_pages, pdfmark],
|
||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
compression=compression,
|
compression=context.options.pdfa_image_compression,
|
||||||
|
color_conversion_strategy=context.options.color_conversion_strategy,
|
||||||
pdf_version=pdf_version,
|
pdf_version=pdf_version,
|
||||||
pdfa_part=pdfa_part,
|
pdfa_part=pdfa_part,
|
||||||
progressbar_class=progressbar_class,
|
progressbar_class=progressbar_class,
|
||||||
|
stop_on_error=stop_on_soft_error,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|||||||
@@ -86,6 +86,16 @@ def add_options(parser):
|
|||||||
# Adjust number of pages to consider at once for JBIG2 compression
|
# Adjust number of pages to consider at once for JBIG2 compression
|
||||||
help=argparse.SUPPRESS,
|
help=argparse.SUPPRESS,
|
||||||
)
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jbig2-threshold',
|
||||||
|
type=numeric(float, 0.4, 0.9),
|
||||||
|
default=0.85,
|
||||||
|
metavar='T',
|
||||||
|
help=(
|
||||||
|
"Adjust JBIG2 symbol code classification threshold "
|
||||||
|
"(default 0.85), range 0.4 to 0.9."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
@@ -95,7 +105,7 @@ def check_options(options):
|
|||||||
program='pngquant',
|
program='pngquant',
|
||||||
package='pngquant',
|
package='pngquant',
|
||||||
version_checker=pngquant.version,
|
version_checker=pngquant.version,
|
||||||
need_version='2.0.1',
|
need_version='2.12.2',
|
||||||
required_for='--optimize {2,3}',
|
required_for='--optimize {2,3}',
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -8,10 +8,14 @@ from __future__ import annotations
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf import hookimpl
|
from ocrmypdf import hookimpl
|
||||||
from ocrmypdf._exec import tesseract
|
from ocrmypdf._exec import tesseract
|
||||||
|
from ocrmypdf._jobcontext import PageContext
|
||||||
from ocrmypdf.cli import numeric, str_to_int
|
from ocrmypdf.cli import numeric, str_to_int
|
||||||
from ocrmypdf.helpers import clamp
|
from ocrmypdf.helpers import clamp
|
||||||
|
from ocrmypdf.imageops import calculate_downsample, downsample_image
|
||||||
from ocrmypdf.pluginspec import OcrEngine
|
from ocrmypdf.pluginspec import OcrEngine
|
||||||
from ocrmypdf.subprocess import check_external_program
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
@@ -26,7 +30,7 @@ def add_options(parser):
|
|||||||
action='append',
|
action='append',
|
||||||
metavar='CFG',
|
metavar='CFG',
|
||||||
default=[],
|
default=[],
|
||||||
help="Additional Tesseract configuration files -- see documentation",
|
help="Additional Tesseract configuration files -- see documentation.",
|
||||||
)
|
)
|
||||||
tess.add_argument(
|
tess.add_argument(
|
||||||
'--tesseract-pagesegmode',
|
'--tesseract-pagesegmode',
|
||||||
@@ -34,7 +38,7 @@ def add_options(parser):
|
|||||||
type=int,
|
type=int,
|
||||||
metavar='PSM',
|
metavar='PSM',
|
||||||
choices=range(0, 14),
|
choices=range(0, 14),
|
||||||
help="Set Tesseract page segmentation mode (see tesseract --help)",
|
help="Set Tesseract page segmentation mode (see tesseract --help).",
|
||||||
)
|
)
|
||||||
tess.add_argument(
|
tess.add_argument(
|
||||||
'--tesseract-oem',
|
'--tesseract-oem',
|
||||||
@@ -43,7 +47,7 @@ def add_options(parser):
|
|||||||
metavar='MODE',
|
metavar='MODE',
|
||||||
choices=range(0, 4),
|
choices=range(0, 4),
|
||||||
help=(
|
help=(
|
||||||
"Set Tesseract 4.0+ OCR engine mode: "
|
"Set Tesseract 4+ OCR engine mode: "
|
||||||
"0 - original Tesseract only; "
|
"0 - original Tesseract only; "
|
||||||
"1 - neural nets LSTM only; "
|
"1 - neural nets LSTM only; "
|
||||||
"2 - Tesseract + LSTM; "
|
"2 - Tesseract + LSTM; "
|
||||||
@@ -58,7 +62,7 @@ def add_options(parser):
|
|||||||
metavar='METHOD',
|
metavar='METHOD',
|
||||||
help=(
|
help=(
|
||||||
"Set Tesseract 5.0+ input image thresholding mode. This may improve OCR "
|
"Set Tesseract 5.0+ input image thresholding mode. This may improve OCR "
|
||||||
"results on low quality images or those that contain high constrast color. "
|
"results on low quality images or those that contain high contrast color. "
|
||||||
"legacy-otsu is the Tesseract default; adaptive-otsu is an improved Otsu "
|
"legacy-otsu is the Tesseract default; adaptive-otsu is an improved Otsu "
|
||||||
"algorithm with improved sort for background color changes; sauvola is "
|
"algorithm with improved sort for background color changes; sauvola is "
|
||||||
"based on local standard deviation."
|
"based on local standard deviation."
|
||||||
@@ -69,8 +73,51 @@ def add_options(parser):
|
|||||||
default=180.0,
|
default=180.0,
|
||||||
type=numeric(float, 0),
|
type=numeric(float, 0),
|
||||||
metavar='SECONDS',
|
metavar='SECONDS',
|
||||||
help='Give up on OCR after the timeout, but copy the preprocessed page '
|
help=(
|
||||||
'into the final output',
|
"Give up on OCR after the timeout, but copy the preprocessed page "
|
||||||
|
"into the final output. This timeout is only used when using Tesseract "
|
||||||
|
"for OCR. When Tesseract is used for other operations such as "
|
||||||
|
"deskewing and orientation, the timeout is controlled by "
|
||||||
|
"--tesseract-non-ocr-timeout."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--tesseract-non-ocr-timeout',
|
||||||
|
default=180.0,
|
||||||
|
type=numeric(float, 0),
|
||||||
|
metavar='SECONDS',
|
||||||
|
help=(
|
||||||
|
"Give up on non-OCR operations such as deskewing and orientation "
|
||||||
|
"after timeout. This is a separate timeout from --tesseract-timeout "
|
||||||
|
"because these operations are not as expensive as OCR."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--tesseract-downsample-large-images',
|
||||||
|
action='store_true',
|
||||||
|
help=(
|
||||||
|
"Downsample large images before OCR. Tesseract has an upper limit on the "
|
||||||
|
"size images it will support. If this argument is given, OCRmyPDF will "
|
||||||
|
"downsample large images to fit Tesseract. This may reduce OCR quality, "
|
||||||
|
"on large images the most desirable text is usually larger. If this "
|
||||||
|
"parameter is not supplied, Tesseract will error out and produce no OCR "
|
||||||
|
"on the page in question. This argument should be used with a high value "
|
||||||
|
"of --tesseract-timeout to ensure Tesseract has enough to time."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--tesseract-downsample-above',
|
||||||
|
action='store',
|
||||||
|
type=numeric(int, 100, 32767),
|
||||||
|
default=32767,
|
||||||
|
help=(
|
||||||
|
"Downsample images larger than this size pixel size in either dimension "
|
||||||
|
"before OCR. --tesseract-downsample-large-images downsamples only when "
|
||||||
|
"an image exceeds Tesseract's internal limits. This argument causes "
|
||||||
|
"downsampling to occur when an image exceeds the given size. This may "
|
||||||
|
"reduce OCR quality, but on large images the most desirable text is "
|
||||||
|
"usually larger."
|
||||||
|
),
|
||||||
)
|
)
|
||||||
tess.add_argument(
|
tess.add_argument(
|
||||||
'--user-words',
|
'--user-words',
|
||||||
@@ -93,7 +140,7 @@ def check_options(options):
|
|||||||
program='tesseract',
|
program='tesseract',
|
||||||
package={'linux': 'tesseract-ocr'},
|
package={'linux': 'tesseract-ocr'},
|
||||||
version_checker=tesseract.version,
|
version_checker=tesseract.version,
|
||||||
need_version='4.0.0-beta.1', # using backport for Travis CI
|
need_version='4.1.1', # Ubuntu 22.04 version (also 20.04)
|
||||||
version_parser=tesseract.TesseractVersion,
|
version_parser=tesseract.TesseractVersion,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -101,11 +148,6 @@ def check_options(options):
|
|||||||
if options.pdf_renderer == 'auto':
|
if options.pdf_renderer == 'auto':
|
||||||
options.pdf_renderer = 'sandwich'
|
options.pdf_renderer = 'sandwich'
|
||||||
|
|
||||||
if not tesseract.has_user_words() and (options.user_words or options.user_patterns):
|
|
||||||
log.warning(
|
|
||||||
"Tesseract 4.0 (which you have installed) ignores --user-words and "
|
|
||||||
"--user-patterns, so these arguments have no effect."
|
|
||||||
)
|
|
||||||
if not tesseract.has_thresholding() and options.tesseract_thresholding != 0:
|
if not tesseract.has_thresholding() and options.tesseract_thresholding != 0:
|
||||||
log.warning(
|
log.warning(
|
||||||
"The installed version of Tesseract does not support changes to its "
|
"The installed version of Tesseract does not support changes to its "
|
||||||
@@ -136,13 +178,41 @@ def validate(pdfinfo, options):
|
|||||||
tess_threads = int(os.environ['OMP_THREAD_LIMIT'])
|
tess_threads = int(os.environ['OMP_THREAD_LIMIT'])
|
||||||
log.debug("Using Tesseract OpenMP thread limit %d", tess_threads)
|
log.debug("Using Tesseract OpenMP thread limit %d", tess_threads)
|
||||||
|
|
||||||
|
if (
|
||||||
|
options.tesseract_downsample_above != 32767
|
||||||
|
and not options.tesseract_downsample_large_images
|
||||||
|
):
|
||||||
|
log.warning(
|
||||||
|
"The --tesseract-downsample-above argument will have no effect unless "
|
||||||
|
"--tesseract-downsample-large-images is also given."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def filter_ocr_image(page: PageContext, image: Image.Image) -> Image.Image:
|
||||||
|
"""Filter the image before OCR.
|
||||||
|
|
||||||
|
Tesseract cannot handle images with more than 32767 pixels in either axis,
|
||||||
|
or more than 2**31 bytes. This function resizes the image to fit within
|
||||||
|
those limits.
|
||||||
|
"""
|
||||||
|
threshold = min(page.options.tesseract_downsample_above, 32767)
|
||||||
|
|
||||||
|
options = page.options
|
||||||
|
if options.tesseract_downsample_large_images:
|
||||||
|
size = calculate_downsample(
|
||||||
|
image, max_size=(threshold, threshold), max_bytes=(2**31) - 1
|
||||||
|
)
|
||||||
|
image = downsample_image(image, size)
|
||||||
|
return image
|
||||||
|
|
||||||
|
|
||||||
class TesseractOcrEngine(OcrEngine):
|
class TesseractOcrEngine(OcrEngine):
|
||||||
"""Implements OCR with Tesseract."""
|
"""Implements OCR with Tesseract."""
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def version():
|
def version():
|
||||||
return tesseract.version()
|
return str(tesseract.version())
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def creator_tag(options):
|
def creator_tag(options):
|
||||||
@@ -161,7 +231,7 @@ class TesseractOcrEngine(OcrEngine):
|
|||||||
return tesseract.get_orientation(
|
return tesseract.get_orientation(
|
||||||
input_file,
|
input_file,
|
||||||
engine_mode=options.tesseract_oem,
|
engine_mode=options.tesseract_oem,
|
||||||
timeout=options.tesseract_timeout,
|
timeout=options.tesseract_non_ocr_timeout,
|
||||||
)
|
)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@@ -170,7 +240,7 @@ class TesseractOcrEngine(OcrEngine):
|
|||||||
input_file,
|
input_file,
|
||||||
languages=options.languages,
|
languages=options.languages,
|
||||||
engine_mode=options.tesseract_oem,
|
engine_mode=options.tesseract_oem,
|
||||||
timeout=options.tesseract_timeout,
|
timeout=options.tesseract_non_ocr_timeout,
|
||||||
)
|
)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
|
|||||||
+38
-19
@@ -15,7 +15,11 @@ T = TypeVar('T', int, float)
|
|||||||
|
|
||||||
|
|
||||||
def numeric(basetype: Callable[[Any], T], min_: T | None = None, max_: T | None = None):
|
def numeric(basetype: Callable[[Any], T], min_: T | None = None, max_: T | None = None):
|
||||||
"""Validator for numeric params"""
|
"""Validator for numeric command line parameters.
|
||||||
|
|
||||||
|
Stipulates that the value must be of type basetype (typically int or float), and
|
||||||
|
optionally, within the range [min_, max_].
|
||||||
|
"""
|
||||||
min_ = basetype(min_) if min_ is not None else None
|
min_ = basetype(min_) if min_ is not None else None
|
||||||
max_ = basetype(max_) if max_ is not None else None
|
max_ = basetype(max_) if max_ is not None else None
|
||||||
|
|
||||||
@@ -46,7 +50,7 @@ def str_to_int(mapping: Mapping[str, int]):
|
|||||||
|
|
||||||
|
|
||||||
class ArgumentParser(argparse.ArgumentParser):
|
class ArgumentParser(argparse.ArgumentParser):
|
||||||
"""Override parser's default behavior of calling sys.exit()
|
"""Override parser's default behavior of calling sys.exit().
|
||||||
|
|
||||||
https://stackoverflow.com/questions/5943249/python-argparse-and-controlling-overriding-the-exit-status-code
|
https://stackoverflow.com/questions/5943249/python-argparse-and-controlling-overriding-the-exit-status-code
|
||||||
|
|
||||||
@@ -57,13 +61,21 @@ class ArgumentParser(argparse.ArgumentParser):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, *args, **kwargs):
|
def __init__(self, *args, **kwargs):
|
||||||
|
"""Initialize the parser."""
|
||||||
super().__init__(*args, **kwargs)
|
super().__init__(*args, **kwargs)
|
||||||
self._api_mode = False
|
self._api_mode = False
|
||||||
|
|
||||||
def enable_api_mode(self):
|
def enable_api_mode(self):
|
||||||
|
"""Enable API mode.
|
||||||
|
|
||||||
|
When set, the parser will not call sys.exit() on error. OCRmyPDF was originally
|
||||||
|
a command line program, but now it has an API. The API works by synthesizing
|
||||||
|
command line arguments.
|
||||||
|
"""
|
||||||
self._api_mode = True
|
self._api_mode = True
|
||||||
|
|
||||||
def error(self, message):
|
def error(self, message):
|
||||||
|
"""Override the default argparse error behavior."""
|
||||||
if not self._api_mode:
|
if not self._api_mode:
|
||||||
super().error(message)
|
super().error(message)
|
||||||
return
|
return
|
||||||
@@ -74,19 +86,22 @@ class LanguageSetAction(argparse.Action):
|
|||||||
"""Manages a list of languages."""
|
"""Manages a list of languages."""
|
||||||
|
|
||||||
def __init__(self, option_strings, dest, default=None, **kwargs):
|
def __init__(self, option_strings, dest, default=None, **kwargs):
|
||||||
|
"""Initialize the action."""
|
||||||
if default is None:
|
if default is None:
|
||||||
default = set()
|
default = list()
|
||||||
super().__init__(option_strings, dest, default=default, **kwargs)
|
super().__init__(option_strings, dest, default=default, **kwargs)
|
||||||
|
|
||||||
def __call__(self, parser, namespace, values, option_string=None):
|
def __call__(self, parser, namespace, values, option_string=None):
|
||||||
|
"""Add a language to the set."""
|
||||||
dest = getattr(namespace, self.dest)
|
dest = getattr(namespace, self.dest)
|
||||||
if '+' in values:
|
if '+' in values:
|
||||||
dest.update(lang for lang in values.split('+'))
|
[dest.append(lang) for lang in values.split('+')]
|
||||||
else:
|
else:
|
||||||
dest.add(values)
|
dest.append(values)
|
||||||
|
|
||||||
|
|
||||||
def get_parser():
|
def get_parser():
|
||||||
|
"""Get the main CLI parser."""
|
||||||
parser = ArgumentParser(
|
parser = ArgumentParser(
|
||||||
prog=_PROGRAM_NAME,
|
prog=_PROGRAM_NAME,
|
||||||
allow_abbrev=True,
|
allow_abbrev=True,
|
||||||
@@ -166,7 +181,9 @@ Online documentation is located at:
|
|||||||
'--image-dpi',
|
'--image-dpi',
|
||||||
metavar='DPI',
|
metavar='DPI',
|
||||||
type=int,
|
type=int,
|
||||||
help="For input image instead of PDF, use this DPI instead of file's.",
|
help="When the input file is an image, not a PDF, use this DPI instead "
|
||||||
|
"of the DPI claimed by the input file. If the input does not claim a "
|
||||||
|
"sensible DPI, this option will be required.",
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
'--output-type',
|
'--output-type',
|
||||||
@@ -348,6 +365,13 @@ Online documentation is located at:
|
|||||||
help="Skip OCR on pages larger than the specified amount of megapixels, "
|
help="Skip OCR on pages larger than the specified amount of megapixels, "
|
||||||
"but include skipped pages in final output",
|
"but include skipped pages in final output",
|
||||||
)
|
)
|
||||||
|
ocrsettings.add_argument(
|
||||||
|
'--invalidate-digital-signatures',
|
||||||
|
action='store_true',
|
||||||
|
help="Normally, OCRmyPDF will refuse to OCR a PDF that has a digital "
|
||||||
|
"signature. This option allows OCR to proceed, but the digital signature "
|
||||||
|
"will be invalidated.",
|
||||||
|
)
|
||||||
|
|
||||||
advanced = parser.add_argument_group(
|
advanced = parser.add_argument_group(
|
||||||
"Advanced", "Advanced options to control OCRmyPDF"
|
"Advanced", "Advanced options to control OCRmyPDF"
|
||||||
@@ -384,19 +408,6 @@ Online documentation is located at:
|
|||||||
help="Only rotate pages when confidence is above this value (arbitrary "
|
help="Only rotate pages when confidence is above this value (arbitrary "
|
||||||
"units reported by tesseract)",
|
"units reported by tesseract)",
|
||||||
)
|
)
|
||||||
advanced.add_argument(
|
|
||||||
'--pdfa-image-compression',
|
|
||||||
choices=['auto', 'jpeg', 'lossless'],
|
|
||||||
default='auto',
|
|
||||||
help="Specify how to compress images in the output PDF/A. 'auto' lets "
|
|
||||||
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
|
|
||||||
"JPEG compression. 'lossless' uses PNG-style lossless compression "
|
|
||||||
"for all images. Monochrome images are always compressed using a "
|
|
||||||
"lossless codec. Compression settings "
|
|
||||||
"are applied to all pages, including those for which OCR was "
|
|
||||||
"skipped. Not supported for --output-type=pdf ; that setting "
|
|
||||||
"preserves the original compression of all images.",
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--fast-web-view',
|
'--fast-web-view',
|
||||||
type=numeric(float, 0),
|
type=numeric(float, 0),
|
||||||
@@ -409,6 +420,14 @@ Online documentation is located at:
|
|||||||
"which do not benefit. If the threshold is 0 it will be apply to all files. "
|
"which do not benefit. If the threshold is 0 it will be apply to all files. "
|
||||||
"Set the threshold very high to disable.",
|
"Set the threshold very high to disable.",
|
||||||
)
|
)
|
||||||
|
advanced.add_argument(
|
||||||
|
'--continue-on-soft-render-error',
|
||||||
|
action='store_true',
|
||||||
|
help="Continue processing pages after a recoverable PDF rendering error. "
|
||||||
|
"A recoverable error is one that does not prevent the page from being "
|
||||||
|
"rendered, but may result in visual differences compared to the input "
|
||||||
|
"file. Missing fonts are a typical source of these errors.",
|
||||||
|
)
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--plugin',
|
'--plugin',
|
||||||
dest='plugins',
|
dest='plugins',
|
||||||
|
|||||||
+15
-22
@@ -35,6 +35,7 @@ class ExitCodeException(Exception):
|
|||||||
message = ""
|
message = ""
|
||||||
|
|
||||||
def __str__(self):
|
def __str__(self):
|
||||||
|
"""Return a string representation of the exception."""
|
||||||
super_msg = super().__str__() # Don't do str(super())
|
super_msg = super().__str__() # Don't do str(super())
|
||||||
if self.message:
|
if self.message:
|
||||||
return self.message.format(super_msg)
|
return self.message.format(super_msg)
|
||||||
@@ -47,26 +48,6 @@ class BadArgsError(ExitCodeException):
|
|||||||
exit_code = ExitCode.bad_args
|
exit_code = ExitCode.bad_args
|
||||||
|
|
||||||
|
|
||||||
class PdfMergeFailedError(ExitCodeException): # deprecated
|
|
||||||
"""An intermediate PDF can't be merged.
|
|
||||||
|
|
||||||
No longer in use.
|
|
||||||
"""
|
|
||||||
|
|
||||||
exit_code = ExitCode.input_file
|
|
||||||
message = dedent(
|
|
||||||
'''\
|
|
||||||
Failed to merge PDF image layer with OCR layer
|
|
||||||
|
|
||||||
Usually this happens because the input PDF file is malformed and
|
|
||||||
ocrmypdf cannot correct the problem on its own.
|
|
||||||
|
|
||||||
Try using
|
|
||||||
ocrmypdf --pdf-renderer sandwich [..other args..]
|
|
||||||
'''
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
class MissingDependencyError(ExitCodeException):
|
class MissingDependencyError(ExitCodeException):
|
||||||
"""A third-party dependency is missing."""
|
"""A third-party dependency is missing."""
|
||||||
|
|
||||||
@@ -114,7 +95,7 @@ class EncryptedPdfError(ExitCodeException):
|
|||||||
|
|
||||||
exit_code = ExitCode.encrypted_pdf
|
exit_code = ExitCode.encrypted_pdf
|
||||||
message = dedent(
|
message = dedent(
|
||||||
'''\
|
"""\
|
||||||
Input PDF is encrypted. The encryption must be removed to
|
Input PDF is encrypted. The encryption must be removed to
|
||||||
perform OCR.
|
perform OCR.
|
||||||
|
|
||||||
@@ -123,7 +104,19 @@ class EncryptedPdfError(ExitCodeException):
|
|||||||
|
|
||||||
You can remove the encryption using
|
You can remove the encryption using
|
||||||
qpdf --decrypt [--password=[password]] infilename
|
qpdf --decrypt [--password=[password]] infilename
|
||||||
'''
|
"""
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class DigitalSignatureError(ExitCodeException):
|
||||||
|
"""PDF has a digital signature."""
|
||||||
|
|
||||||
|
exit_code = ExitCode.input_file
|
||||||
|
message = dedent(
|
||||||
|
"""\
|
||||||
|
Input PDF has a digital signature. OCR would alter the document,
|
||||||
|
invalidating the signature.
|
||||||
|
"""
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,3 +1,9 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
#
|
#
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
"""Extra plugins. These are not automatically inserted when ocrmypdf is run.
|
||||||
|
|
||||||
|
You can use these plugins by specifying them on the command line, e.g.:
|
||||||
|
ocrmypdf --plugin ocrmypdf.extra_plugins.semfree ...
|
||||||
|
"""
|
||||||
|
|||||||
@@ -12,7 +12,7 @@ worker communicates only with the main process.
|
|||||||
|
|
||||||
This is not without drawbacks. If the tasks are not "even" in size, which cannot
|
This is not without drawbacks. If the tasks are not "even" in size, which cannot
|
||||||
be guaranteed, some workers may end up with too much work while others are idle.
|
be guaranteed, some workers may end up with too much work while others are idle.
|
||||||
It is less efficient than the standard implementation, so not th edefault.
|
It is less efficient than the standard implementation, so not the default.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -54,6 +54,7 @@ def split_every(n: int, iterable: Iterable) -> Iterator:
|
|||||||
|
|
||||||
|
|
||||||
def process_sigbus(*args):
|
def process_sigbus(*args):
|
||||||
|
"""Handle SIGBUS signal at the worker level."""
|
||||||
raise InputFileError("A worker process lost access to an input file")
|
raise InputFileError("A worker process lost access to an input file")
|
||||||
|
|
||||||
|
|
||||||
@@ -61,20 +62,21 @@ class ConnectionLogHandler(logging.handlers.QueueHandler):
|
|||||||
"""Handler used by child processes to forward log messages to parent."""
|
"""Handler used by child processes to forward log messages to parent."""
|
||||||
|
|
||||||
def __init__(self, conn: Connection) -> None:
|
def __init__(self, conn: Connection) -> None:
|
||||||
|
"""Initialize the handler."""
|
||||||
# sets the parent's queue to None - parent only touches queue
|
# sets the parent's queue to None - parent only touches queue
|
||||||
# in enqueue() which we override
|
# in enqueue() which we override
|
||||||
super().__init__(None) # type: ignore
|
super().__init__(None) # type: ignore
|
||||||
self.conn = conn
|
self.conn = conn
|
||||||
|
|
||||||
def enqueue(self, record):
|
def enqueue(self, record):
|
||||||
|
"""Enqueue a log message."""
|
||||||
self.conn.send(('log', record))
|
self.conn.send(('log', record))
|
||||||
|
|
||||||
|
|
||||||
def process_loop(
|
def process_loop(
|
||||||
conn: Connection, user_init: Callable[[], None], loglevel, task, task_args
|
conn: Connection, user_init: Callable[[], None], loglevel, task, task_args
|
||||||
):
|
):
|
||||||
"""Initialize a process pool worker"""
|
"""Initialize a process pool worker."""
|
||||||
|
|
||||||
# Install SIGBUS handler (so our parent process can abort somewhat gracefully)
|
# Install SIGBUS handler (so our parent process can abort somewhat gracefully)
|
||||||
with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS
|
with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS
|
||||||
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
|
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
|
||||||
@@ -166,8 +168,7 @@ class LambdaExecutor(Executor):
|
|||||||
continue
|
continue
|
||||||
|
|
||||||
if msg_type == MessageType.result:
|
if msg_type == MessageType.result:
|
||||||
if task_finished:
|
task_finished(msg, pbar)
|
||||||
task_finished(msg, pbar)
|
|
||||||
elif msg_type == 'log':
|
elif msg_type == 'log':
|
||||||
record = msg
|
record = msg
|
||||||
logger = logging.getLogger(record.name)
|
logger = logging.getLogger(record.name)
|
||||||
@@ -185,14 +186,20 @@ class LambdaExecutor(Executor):
|
|||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def get_executor(progressbar_class):
|
def get_executor(progressbar_class):
|
||||||
|
"""Return a LambdaExecutor instance."""
|
||||||
return LambdaExecutor(pbar_class=progressbar_class)
|
return LambdaExecutor(pbar_class=progressbar_class)
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def get_logging_console():
|
def get_logging_console():
|
||||||
|
"""Return a logging.StreamHandler instance."""
|
||||||
return logging.StreamHandler()
|
return logging.StreamHandler()
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def get_progressbar_class():
|
def get_progressbar_class():
|
||||||
|
"""Return a NullProgressBar instance.
|
||||||
|
|
||||||
|
This executor cannot use a progress bar.
|
||||||
|
"""
|
||||||
return NullProgressBar
|
return NullProgressBar
|
||||||
|
|||||||
+101
-64
@@ -10,48 +10,61 @@ import multiprocessing
|
|||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import warnings
|
import warnings
|
||||||
from collections import namedtuple
|
|
||||||
from collections.abc import Iterable
|
from collections.abc import Iterable
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from functools import wraps
|
from decimal import Decimal
|
||||||
from io import StringIO
|
from io import StringIO
|
||||||
from math import isclose, isfinite
|
from math import isclose, isfinite
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Sequence
|
from statistics import harmonic_mean
|
||||||
|
from typing import (
|
||||||
|
Any,
|
||||||
|
Callable,
|
||||||
|
Generic,
|
||||||
|
Sequence,
|
||||||
|
SupportsFloat,
|
||||||
|
SupportsRound,
|
||||||
|
TypeVar,
|
||||||
|
)
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
import pikepdf
|
import pikepdf
|
||||||
from packaging.version import Version
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
if Version(img2pdf.__version__) < Version('0.4.0'):
|
IMG2PDF_KWARGS = dict(engine=img2pdf.Engine.pikepdf, rotation=img2pdf.Rotation.ifvalid)
|
||||||
IMG2PDF_KWARGS = dict(without_pdfw=True)
|
|
||||||
elif Version(img2pdf.__version__) < Version('0.4.3'):
|
|
||||||
IMG2PDF_KWARGS = dict(engine=img2pdf.Engine.pikepdf)
|
|
||||||
else:
|
|
||||||
IMG2PDF_KWARGS = dict(
|
|
||||||
engine=img2pdf.Engine.pikepdf, rotation=img2pdf.Rotation.ifvalid
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
class Resolution(namedtuple('Resolution', ('x', 'y'))):
|
T = TypeVar('T', float, int, Decimal)
|
||||||
|
|
||||||
|
|
||||||
|
class Resolution(Generic[T]):
|
||||||
"""The number of pixels per inch in each 2D direction.
|
"""The number of pixels per inch in each 2D direction.
|
||||||
|
|
||||||
Resolution objects are considered "equal" for == purposes if they are
|
Resolution objects are considered "equal" for == purposes if they are
|
||||||
equal to a reasonable tolerance.
|
equal to a reasonable tolerance.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
__slots__ = ()
|
x: T
|
||||||
|
y: T
|
||||||
|
|
||||||
|
__slots__ = ('x', 'y')
|
||||||
|
|
||||||
|
def __init__(self, x: T, y: T):
|
||||||
|
"""Construct a Resolution object."""
|
||||||
|
self.x = x
|
||||||
|
self.y = y
|
||||||
|
|
||||||
# rel_tol after converting from dpi to pixels per meter and saving
|
# rel_tol after converting from dpi to pixels per meter and saving
|
||||||
# as integer with rounding, as many file formats
|
# as integer with rounding, as many file formats
|
||||||
CONVERSION_ERROR = 0.002
|
CONVERSION_ERROR = 0.002
|
||||||
|
|
||||||
def round(self, ndigits: int):
|
def round(self, ndigits: int) -> Resolution:
|
||||||
|
"""Round to ndigits after the decimal point."""
|
||||||
return Resolution(round(self.x, ndigits), round(self.y, ndigits))
|
return Resolution(round(self.x, ndigits), round(self.y, ndigits))
|
||||||
|
|
||||||
def to_int(self):
|
def to_int(self) -> Resolution[int]:
|
||||||
|
"""Round to nearest integer."""
|
||||||
return Resolution(int(round(self.x)), int(round(self.y)))
|
return Resolution(int(round(self.x)), int(round(self.y)))
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -60,31 +73,65 @@ class Resolution(namedtuple('Resolution', ('x', 'y'))):
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def is_square(self) -> bool:
|
def is_square(self) -> bool:
|
||||||
|
"""True if the resolution is square (x == y)."""
|
||||||
return self._isclose(self.x, self.y)
|
return self._isclose(self.x, self.y)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def is_finite(self) -> bool:
|
def is_finite(self) -> bool:
|
||||||
|
"""True if both x and y are finite numbers."""
|
||||||
return isfinite(self.x) and isfinite(self.y)
|
return isfinite(self.x) and isfinite(self.y)
|
||||||
|
|
||||||
def take_max(self, vals, yvals=None):
|
def to_scalar(self) -> float:
|
||||||
if yvals is not None:
|
"""Return the harmonic mean of x and y as a 1D approximation.
|
||||||
return Resolution(max(self.x, *vals), max(self.y, *yvals))
|
|
||||||
max_x, max_y = self.x, self.y
|
|
||||||
for x, y in vals:
|
|
||||||
max_x = max(x, max_x)
|
|
||||||
max_y = max(y, max_y)
|
|
||||||
return Resolution(max_x, max_y)
|
|
||||||
|
|
||||||
def flip_axis(self):
|
In most cases, Resolution is 2D, but typically it is "square" (x == y) and
|
||||||
|
can be approximated as a single number. When not square, the harmonic mean
|
||||||
|
is used to approximate the 2D resolution as a single number.
|
||||||
|
"""
|
||||||
|
return harmonic_mean([float(self.x), float(self.y)])
|
||||||
|
|
||||||
|
def _take_minmax(
|
||||||
|
self, vals: Iterable[Any], yvals: Iterable[Any] | None, cmp: Callable
|
||||||
|
) -> Resolution:
|
||||||
|
"""Return a new Resolution object with the maximum resolution of inputs."""
|
||||||
|
if yvals is not None:
|
||||||
|
return Resolution(cmp(self.x, *vals), cmp(self.y, *yvals))
|
||||||
|
cmp_x, cmp_y = self.x, self.y
|
||||||
|
for x, y in vals:
|
||||||
|
cmp_x = cmp(x, cmp_x)
|
||||||
|
cmp_y = cmp(y, cmp_y)
|
||||||
|
return Resolution(cmp_x, cmp_y)
|
||||||
|
|
||||||
|
def take_max(
|
||||||
|
self, vals: Iterable[Any], yvals: Iterable[Any] | None = None
|
||||||
|
) -> Resolution:
|
||||||
|
"""Return a new Resolution object with the maximum resolution of inputs."""
|
||||||
|
return self._take_minmax(vals, yvals, max)
|
||||||
|
|
||||||
|
def take_min(
|
||||||
|
self, vals: Iterable[Any], yvals: Iterable[Any] | None = None
|
||||||
|
) -> Resolution:
|
||||||
|
"""Return a new Resolution object with the minimum resolution of inputs."""
|
||||||
|
return self._take_minmax(vals, yvals, min)
|
||||||
|
|
||||||
|
def flip_axis(self) -> Resolution[T]:
|
||||||
|
"""Return a new Resolution object with x and y swapped."""
|
||||||
return Resolution(self.y, self.x)
|
return Resolution(self.y, self.x)
|
||||||
|
|
||||||
|
def __getitem__(self, idx: int | slice) -> T:
|
||||||
|
"""Support [0] and [1] indexing."""
|
||||||
|
return (self.x, self.y)[idx]
|
||||||
|
|
||||||
def __str__(self):
|
def __str__(self):
|
||||||
return f"{self.x:f}x{self.y:f}"
|
"""Return a string representation of the resolution."""
|
||||||
|
return f"{self.x:f}×{self.y:f}"
|
||||||
|
|
||||||
def __repr__(self): # pragma: no cover
|
def __repr__(self): # pragma: no cover
|
||||||
return f"Resolution({self.x}x{self.y} dpi)"
|
"""Return a repr() of the resolution."""
|
||||||
|
return f"Resolution({self.x!r}, {self.y!r})"
|
||||||
|
|
||||||
def __eq__(self, other):
|
def __eq__(self, other):
|
||||||
|
"""Return True if the resolution is equal to another resolution."""
|
||||||
if isinstance(other, tuple) and len(other) == 2:
|
if isinstance(other, tuple) and len(other) == 2:
|
||||||
other = Resolution(*other)
|
other = Resolution(*other)
|
||||||
if not isinstance(other, Resolution):
|
if not isinstance(other, Resolution):
|
||||||
@@ -93,10 +140,10 @@ class Resolution(namedtuple('Resolution', ('x', 'y'))):
|
|||||||
|
|
||||||
|
|
||||||
class NeverRaise(Exception):
|
class NeverRaise(Exception):
|
||||||
"""An exception that is never raised"""
|
"""An exception that is never raised."""
|
||||||
|
|
||||||
|
|
||||||
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike) -> None:
|
||||||
"""Create a symbolic link at ``soft_link_name``, which references ``input_file``.
|
"""Create a symbolic link at ``soft_link_name``, which references ``input_file``.
|
||||||
|
|
||||||
Think of this as copying ``input_file`` to ``soft_link_name`` with less overhead.
|
Think of this as copying ``input_file`` to ``soft_link_name`` with less overhead.
|
||||||
@@ -111,7 +158,7 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
|||||||
# Guard against soft linking to oneself
|
# Guard against soft linking to oneself
|
||||||
if input_file == soft_link_name:
|
if input_file == soft_link_name:
|
||||||
log.warning(
|
log.warning(
|
||||||
"No symbolic link created. You are using the original data directory "
|
"No symbolic link created. You are using the original data directory "
|
||||||
"as the working directory."
|
"as the working directory."
|
||||||
)
|
)
|
||||||
return
|
return
|
||||||
@@ -137,7 +184,11 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
|||||||
os.symlink(os.path.abspath(input_file), soft_link_name)
|
os.symlink(os.path.abspath(input_file), soft_link_name)
|
||||||
|
|
||||||
|
|
||||||
def samefile(file1: os.PathLike, file2: os.PathLike):
|
def samefile(file1: os.PathLike, file2: os.PathLike) -> bool:
|
||||||
|
"""Return True if two files are the same file.
|
||||||
|
|
||||||
|
Attempts to account for different relative paths to the same file.
|
||||||
|
"""
|
||||||
if os.name == 'nt':
|
if os.name == 'nt':
|
||||||
return file1 == file2
|
return file1 == file2
|
||||||
else:
|
else:
|
||||||
@@ -155,7 +206,7 @@ def monotonic(seq: Sequence) -> bool:
|
|||||||
|
|
||||||
|
|
||||||
def page_number(input_file: os.PathLike) -> int:
|
def page_number(input_file: os.PathLike) -> int:
|
||||||
"""Get one-based page number implied by filename (000002.pdf -> 2)"""
|
"""Get one-based page number implied by filename (000002.pdf -> 2)."""
|
||||||
return int(os.path.basename(os.fspath(input_file))[0:6])
|
return int(os.path.basename(os.fspath(input_file))[0:6])
|
||||||
|
|
||||||
|
|
||||||
@@ -184,7 +235,7 @@ def is_file_writable(test_file: os.PathLike) -> bool:
|
|||||||
p = p.resolve(strict=False)
|
p = p.resolve(strict=False)
|
||||||
|
|
||||||
# p.is_file() throws an exception in some cases
|
# p.is_file() throws an exception in some cases
|
||||||
if p.exists() and p.is_file():
|
if p.exists() and (p.is_file() or p.samefile(os.devnull)):
|
||||||
return os.access(
|
return os.access(
|
||||||
os.fspath(p),
|
os.fspath(p),
|
||||||
os.W_OK,
|
os.W_OK,
|
||||||
@@ -252,43 +303,29 @@ def check_pdf(input_file: Path) -> bool:
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
def clamp(n, smallest, largest): # mypy doesn't understand types for this
|
def clamp(n: T, smallest: T, largest: T) -> T:
|
||||||
"""Clamps the value of ``n`` to between ``smallest`` and ``largest``."""
|
"""Clamps the value of ``n`` to between ``smallest`` and ``largest``."""
|
||||||
return max(smallest, min(n, largest))
|
return max(smallest, min(n, largest))
|
||||||
|
|
||||||
|
|
||||||
def remove_all_log_handlers(logger):
|
def remove_all_log_handlers(logger: logging.Logger) -> None:
|
||||||
"Remove all log handlers, usually used in a child process."
|
"""Remove all log handlers, usually used in a child process.
|
||||||
|
|
||||||
|
The child process inherits the log handlers from the parent process when
|
||||||
|
a fork occurs. Typically we want to remove all log handlers in the child
|
||||||
|
process so that the child process can set up a single queue handler to
|
||||||
|
forward log messages to the parent process.
|
||||||
|
"""
|
||||||
for handler in logger.handlers[:]:
|
for handler in logger.handlers[:]:
|
||||||
logger.removeHandler(handler)
|
logger.removeHandler(handler)
|
||||||
handler.close() # To ensure handlers with opened resources are released
|
handler.close() # To ensure handlers with opened resources are released
|
||||||
|
|
||||||
|
|
||||||
def pikepdf_enable_mmap():
|
def pikepdf_enable_mmap() -> None:
|
||||||
# try:
|
"""Enable pikepdf mmap."""
|
||||||
# if pikepdf._qpdf.set_access_default_mmap(True):
|
try:
|
||||||
# log.debug("pikepdf mmap enabled")
|
if pikepdf._core.set_access_default_mmap(True):
|
||||||
# except AttributeError:
|
log.debug("pikepdf mmap enabled")
|
||||||
# log.debug("pikepdf mmap not available")
|
except AttributeError:
|
||||||
# We found a race condition probably related to pybind issue #2252 that can
|
log.debug("pikepdf mmap not available")
|
||||||
# cause a crash. For now, disable pikepdf mmap to be on the safe side.
|
log.debug("pikepdf mmap disabled")
|
||||||
# Fix is not in pybind11 2.6.0
|
|
||||||
# log.debug("pikepdf mmap disabled")
|
|
||||||
return
|
|
||||||
|
|
||||||
|
|
||||||
def deprecated(func):
|
|
||||||
"""Warn that function is deprecated."""
|
|
||||||
|
|
||||||
@wraps(func)
|
|
||||||
def new_func(*args, **kwargs):
|
|
||||||
warnings.simplefilter('always', DeprecationWarning) # turn off filter
|
|
||||||
warnings.warn(
|
|
||||||
f"Call to deprecated function {func.__name__}.",
|
|
||||||
category=DeprecationWarning,
|
|
||||||
stacklevel=2,
|
|
||||||
)
|
|
||||||
warnings.simplefilter('default', DeprecationWarning) # reset filter
|
|
||||||
return func(*args, **kwargs)
|
|
||||||
|
|
||||||
return new_func
|
|
||||||
|
|||||||
@@ -14,10 +14,12 @@ import re
|
|||||||
import warnings
|
import warnings
|
||||||
from math import atan, cos, sin
|
from math import atan, cos, sin
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, NamedTuple, Optional, Tuple, Union
|
from typing import Any, NamedTuple
|
||||||
from xml.etree import ElementTree
|
from xml.etree import ElementTree
|
||||||
|
|
||||||
with warnings.catch_warnings():
|
with warnings.catch_warnings():
|
||||||
|
# reportlab uses deprecated load_module
|
||||||
|
# shim can be removed when we require reportlab >= 3.7
|
||||||
warnings.filterwarnings(
|
warnings.filterwarnings(
|
||||||
'ignore', category=DeprecationWarning, message=r".*load_module.*"
|
'ignore', category=DeprecationWarning, message=r".*load_module.*"
|
||||||
)
|
)
|
||||||
@@ -98,11 +100,10 @@ class HocrTransformError(Exception):
|
|||||||
|
|
||||||
|
|
||||||
class HocrTransform:
|
class HocrTransform:
|
||||||
|
"""A class for converting documents from the hOCR format.
|
||||||
|
|
||||||
"""
|
|
||||||
A class for converting documents from the hOCR format.
|
|
||||||
For details of the hOCR format, see:
|
For details of the hOCR format, see:
|
||||||
http://kba.cloud/hocr-spec/
|
http://kba.cloud/hocr-spec/.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
box_pattern = re.compile(r'bbox((\s+\d+){4})')
|
box_pattern = re.compile(r'bbox((\s+\d+){4})')
|
||||||
@@ -118,6 +119,7 @@ class HocrTransform:
|
|||||||
)
|
)
|
||||||
|
|
||||||
def __init__(self, *, hocr_filename: str | Path, dpi: float):
|
def __init__(self, *, hocr_filename: str | Path, dpi: float):
|
||||||
|
"""Initialize the HocrTransform object."""
|
||||||
self.dpi = dpi
|
self.dpi = dpi
|
||||||
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
||||||
|
|
||||||
@@ -142,9 +144,7 @@ class HocrTransform:
|
|||||||
raise HocrTransformError("hocr file is missing page dimensions")
|
raise HocrTransformError("hocr file is missing page dimensions")
|
||||||
|
|
||||||
def __str__(self): # pragma: no cover
|
def __str__(self): # pragma: no cover
|
||||||
"""
|
"""Return the textual content of the HTML body."""
|
||||||
Return the textual content of the HTML body
|
|
||||||
"""
|
|
||||||
if self.hocr is None:
|
if self.hocr is None:
|
||||||
return ''
|
return ''
|
||||||
body = self.hocr.find(self._child_xpath('body'))
|
body = self.hocr.find(self._child_xpath('body'))
|
||||||
@@ -154,9 +154,7 @@ class HocrTransform:
|
|||||||
return ''
|
return ''
|
||||||
|
|
||||||
def _get_element_text(self, element: Element):
|
def _get_element_text(self, element: Element):
|
||||||
"""
|
"""Return the textual content of the element and its children."""
|
||||||
Return the textual content of the element and its children
|
|
||||||
"""
|
|
||||||
text = ''
|
text = ''
|
||||||
if element.text is not None:
|
if element.text is not None:
|
||||||
text += element.text
|
text += element.text
|
||||||
@@ -168,10 +166,7 @@ class HocrTransform:
|
|||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def element_coordinates(cls, element: Element) -> Rect:
|
def element_coordinates(cls, element: Element) -> Rect:
|
||||||
"""
|
"""Get coordinates of the bounding box around an element."""
|
||||||
Returns a tuple containing the coordinates of the bounding box around
|
|
||||||
an element
|
|
||||||
"""
|
|
||||||
out = Rect._make(0 for _ in range(4))
|
out = Rect._make(0 for _ in range(4))
|
||||||
if 'title' in element.attrib:
|
if 'title' in element.attrib:
|
||||||
matches = cls.box_pattern.search(element.attrib['title'])
|
matches = cls.box_pattern.search(element.attrib['title'])
|
||||||
@@ -182,9 +177,7 @@ class HocrTransform:
|
|||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def baseline(cls, element: Element) -> tuple[float, float]:
|
def baseline(cls, element: Element) -> tuple[float, float]:
|
||||||
"""
|
"""Get baseline's slope and intercept."""
|
||||||
Returns a tuple containing the baseline slope and intercept.
|
|
||||||
"""
|
|
||||||
if 'title' in element.attrib:
|
if 'title' in element.attrib:
|
||||||
matches = cls.baseline_pattern.search(element.attrib['title'])
|
matches = cls.baseline_pattern.search(element.attrib['title'])
|
||||||
if matches:
|
if matches:
|
||||||
@@ -192,9 +185,7 @@ class HocrTransform:
|
|||||||
return (0.0, 0.0)
|
return (0.0, 0.0)
|
||||||
|
|
||||||
def pt_from_pixel(self, pxl) -> Rect:
|
def pt_from_pixel(self, pxl) -> Rect:
|
||||||
"""
|
"""Returns the quantity in PDF units (pt) given quantity in pixels."""
|
||||||
Returns the quantity in PDF units (pt) given quantity in pixels
|
|
||||||
"""
|
|
||||||
return Rect._make((c / self.dpi * inch) for c in pxl)
|
return Rect._make((c / self.dpi * inch) for c in pxl)
|
||||||
|
|
||||||
def _child_xpath(self, html_tag: str, html_class: str | None = None) -> str:
|
def _child_xpath(self, html_tag: str, html_class: str | None = None) -> str:
|
||||||
@@ -205,21 +196,9 @@ class HocrTransform:
|
|||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def replace_unsupported_chars(cls, s: str) -> str:
|
def replace_unsupported_chars(cls, s: str) -> str:
|
||||||
"""
|
"""Replaces characters with those available in the Helvetica typeface."""
|
||||||
Given an input string, returns the corresponding string that:
|
|
||||||
* is available in the Helvetica facetype
|
|
||||||
* does not contain any ligature (to allow easy search in the PDF file)
|
|
||||||
"""
|
|
||||||
return s.translate(cls.ligatures)
|
return s.translate(cls.ligatures)
|
||||||
|
|
||||||
def topdown_position(self, element):
|
|
||||||
pxl_line_coords = self.element_coordinates(element)
|
|
||||||
line_box = self.pt_from_pixel(pxl_line_coords)
|
|
||||||
# Coordinates here are still in the hocr coordinate system, so 0 on the y axis
|
|
||||||
# is the top of the page and increasing values of y will move towards the
|
|
||||||
# bottom of the page.
|
|
||||||
return line_box.y2
|
|
||||||
|
|
||||||
def to_pdf(
|
def to_pdf(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
@@ -230,8 +209,8 @@ class HocrTransform:
|
|||||||
invisible_text: bool = False,
|
invisible_text: bool = False,
|
||||||
interword_spaces: bool = False,
|
interword_spaces: bool = False,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""
|
"""Creates a PDF file with an image superimposed on top of the text.
|
||||||
Creates a PDF file with an image superimposed on top of the text.
|
|
||||||
Text is positioned according to the bounding box of the lines in
|
Text is positioned according to the bounding box of the lines in
|
||||||
the hOCR file.
|
the hOCR file.
|
||||||
The image need not be identical to the image used to create the hOCR
|
The image need not be identical to the image used to create the hOCR
|
||||||
@@ -322,6 +301,7 @@ class HocrTransform:
|
|||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def polyval(cls, poly, x): # pragma: no cover
|
def polyval(cls, poly, x): # pragma: no cover
|
||||||
|
"""Calculate the value of a polynomial at a point."""
|
||||||
return x * poly[0] + poly[1]
|
return x * poly[0] + poly[1]
|
||||||
|
|
||||||
def _do_line(
|
def _do_line(
|
||||||
|
|||||||
@@ -0,0 +1,168 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
"""OCR-related image manipulation."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
from math import floor, sqrt
|
||||||
|
from typing import Optional, Tuple
|
||||||
|
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
# Remove this workaround when we require Pillow >= 9.1.0
|
||||||
|
try:
|
||||||
|
Resampling = Image.Resampling # type: ignore
|
||||||
|
except AttributeError:
|
||||||
|
# Pillow 9 shim
|
||||||
|
Resampling = Image # type: ignore
|
||||||
|
|
||||||
|
|
||||||
|
# While from __future__ import annotations, we use singledispatch here, which
|
||||||
|
# does not support annotations. Disable check about using old-style typing
|
||||||
|
# until Python 3.10, OR when drop singledispatch in ocrmypdf 15.
|
||||||
|
# ruff: noqa: UP006
|
||||||
|
# ruff: noqa: UP007
|
||||||
|
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def bytes_per_pixel(mode: str) -> int:
|
||||||
|
"""Return the number of padded bytes per pixel for a given PIL image mode.
|
||||||
|
|
||||||
|
In RGB mode we assume 4 bytes per pixel, which is the case for most
|
||||||
|
consumers.
|
||||||
|
"""
|
||||||
|
if mode in ('1', 'L', 'P'):
|
||||||
|
return 1
|
||||||
|
if mode in ('LA', 'PA', 'La') or mode.startswith('I;16'):
|
||||||
|
return 2
|
||||||
|
return 4
|
||||||
|
|
||||||
|
|
||||||
|
def _calculate_downsample(
|
||||||
|
image_size: Tuple[int, int],
|
||||||
|
bytes_per_pixel: int,
|
||||||
|
*,
|
||||||
|
max_size: Optional[Tuple[int, int]] = None,
|
||||||
|
max_pixels: Optional[int] = None,
|
||||||
|
max_bytes: Optional[int] = None,
|
||||||
|
) -> Tuple[int, int]:
|
||||||
|
"""Calculate image size required to downsample an image to fit limits.
|
||||||
|
|
||||||
|
If no limit is exceeded, the input image's size is returned.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
image_size: Dimensions of image.
|
||||||
|
bytes_per_pixel: Number of bytes per pixel.
|
||||||
|
max_size: The maximum width and height of the image.
|
||||||
|
max_pixels: The maximum number of pixels in the image. Some image consumers
|
||||||
|
limit the total number of pixels as some value other than width*height.
|
||||||
|
max_bytes: The maximum number of bytes in the image. RGB is counted as 4
|
||||||
|
bytes; all other modes are counted as 1 byte.
|
||||||
|
"""
|
||||||
|
size = image_size
|
||||||
|
|
||||||
|
if max_size is not None:
|
||||||
|
overage = max_size[0] / size[0], max_size[1] / size[1]
|
||||||
|
size_factor = min(overage)
|
||||||
|
if size_factor < 1.0:
|
||||||
|
log.debug("Resizing image to fit image dimensions limit")
|
||||||
|
size = floor(size[0] * size_factor), floor(size[1] * size_factor)
|
||||||
|
if size[0] == 0:
|
||||||
|
size = 1, min(size[1], max_size[1])
|
||||||
|
elif size[1] == 0:
|
||||||
|
size = min(size[0], max_size[0]), 1
|
||||||
|
|
||||||
|
if max_pixels is not None:
|
||||||
|
if size[0] * size[1] > max_pixels:
|
||||||
|
log.debug("Resizing image to fit image pixel limit")
|
||||||
|
pixels_factor = sqrt(max_pixels / (size[0] * size[1]))
|
||||||
|
size = floor(size[0] * pixels_factor), floor(size[1] * pixels_factor)
|
||||||
|
|
||||||
|
if max_bytes is not None:
|
||||||
|
bpp = bytes_per_pixel
|
||||||
|
# stride = bytes per line
|
||||||
|
stride = size[0] * bpp
|
||||||
|
height = size[1]
|
||||||
|
if stride * height > max_bytes:
|
||||||
|
log.debug("Resizing image to fit image byte size limit")
|
||||||
|
bytes_factor = sqrt(max_bytes / (stride * height))
|
||||||
|
scaled_stride = floor(stride * bytes_factor)
|
||||||
|
scaled_height = floor(height * bytes_factor)
|
||||||
|
if scaled_stride == 0:
|
||||||
|
scaled_stride = bpp
|
||||||
|
scaled_height = min(max_bytes // bpp, scaled_height)
|
||||||
|
if scaled_height == 0:
|
||||||
|
scaled_height = 1
|
||||||
|
scaled_stride = min(max_bytes // scaled_height, scaled_stride)
|
||||||
|
size = floor(scaled_stride / bpp), scaled_height
|
||||||
|
|
||||||
|
return size
|
||||||
|
|
||||||
|
|
||||||
|
def calculate_downsample(
|
||||||
|
image: Image.Image,
|
||||||
|
*,
|
||||||
|
max_size: Optional[Tuple[int, int]] = None,
|
||||||
|
max_pixels: Optional[int] = None,
|
||||||
|
max_bytes: Optional[int] = None,
|
||||||
|
) -> Tuple[int, int]:
|
||||||
|
"""Calculate image size required to downsample an image to fit limits.
|
||||||
|
|
||||||
|
If no limit is exceeded, the input image's size is returned.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
image: The image to downsample.
|
||||||
|
max_size: The maximum width and height of the image.
|
||||||
|
max_pixels: The maximum number of pixels in the image. Some image consumers
|
||||||
|
limit the total number of pixels as some value other than width*height.
|
||||||
|
max_bytes: The maximum number of bytes in the image. RGB is counted as 4
|
||||||
|
bytes; all other modes are counted as 1 byte.
|
||||||
|
"""
|
||||||
|
return _calculate_downsample(
|
||||||
|
image.size,
|
||||||
|
bytes_per_pixel(image.mode),
|
||||||
|
max_size=max_size,
|
||||||
|
max_pixels=max_pixels,
|
||||||
|
max_bytes=max_bytes,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def downsample_image(
|
||||||
|
image: Image.Image,
|
||||||
|
new_size: tuple[int, int],
|
||||||
|
*,
|
||||||
|
resample_mode: Image.Resampling = Resampling.BICUBIC,
|
||||||
|
reducing_gap: int = 3,
|
||||||
|
) -> Image.Image:
|
||||||
|
"""Downsample an image to fit within the given limits.
|
||||||
|
|
||||||
|
The DPI is adjusted to match the new size, which is how we can ensure the
|
||||||
|
OCR is positioned correctly.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
image: The image to downsample
|
||||||
|
new_size: The new size of the image.
|
||||||
|
resample_mode: The resampling mode to use when downsampling.
|
||||||
|
reducing_gap: The reducing gap to use when downsampling (for larger
|
||||||
|
reductions).
|
||||||
|
"""
|
||||||
|
if new_size == image.size:
|
||||||
|
return image
|
||||||
|
|
||||||
|
original_size = image.size
|
||||||
|
original_dpi = image.info['dpi']
|
||||||
|
image = image.resize(
|
||||||
|
new_size,
|
||||||
|
resample=resample_mode,
|
||||||
|
reducing_gap=reducing_gap,
|
||||||
|
)
|
||||||
|
image.info['dpi'] = (
|
||||||
|
round(original_dpi[0] * new_size[0] / original_size[0]),
|
||||||
|
round(original_dpi[1] * new_size[1] / original_size[1]),
|
||||||
|
)
|
||||||
|
log.debug(f"Rescaled image to {image.size} pixels and {image.info['dpi']} dpi")
|
||||||
|
return image
|
||||||
+129
-75
@@ -53,21 +53,25 @@ class XrefExt(NamedTuple):
|
|||||||
|
|
||||||
|
|
||||||
def img_name(root: Path, xref: Xref, ext: str) -> Path:
|
def img_name(root: Path, xref: Xref, ext: str) -> Path:
|
||||||
|
"""Return the name of an image file for a given xref and extension."""
|
||||||
return root / f'{xref:08d}{ext}'
|
return root / f'{xref:08d}{ext}'
|
||||||
|
|
||||||
|
|
||||||
def png_name(root: Path, xref: Xref) -> Path:
|
def png_name(root: Path, xref: Xref) -> Path:
|
||||||
|
"""Return the name of a PNG file for a given xref."""
|
||||||
return img_name(root, xref, '.png')
|
return img_name(root, xref, '.png')
|
||||||
|
|
||||||
|
|
||||||
def jpg_name(root: Path, xref: Xref) -> Path:
|
def jpg_name(root: Path, xref: Xref) -> Path:
|
||||||
|
"""Return the name of a JPEG file for a given xref."""
|
||||||
return img_name(root, xref, '.jpg')
|
return img_name(root, xref, '.jpg')
|
||||||
|
|
||||||
|
|
||||||
def extract_image_filter(
|
def extract_image_filter(
|
||||||
pike: Pdf, root: Path, image: Stream, xref: Xref
|
pdf: Pdf, root: Path, image: Stream, xref: Xref
|
||||||
) -> tuple[PdfImage, tuple[Name, Object]] | None:
|
) -> tuple[PdfImage, tuple[Name, Object]] | None:
|
||||||
del pike # unused args
|
"""Determine if an image is extractable."""
|
||||||
|
del pdf # unused args
|
||||||
del root
|
del root
|
||||||
|
|
||||||
if image.Subtype != Name.Image:
|
if image.Subtype != Name.Image:
|
||||||
@@ -122,11 +126,12 @@ def extract_image_filter(
|
|||||||
|
|
||||||
|
|
||||||
def extract_image_jbig2(
|
def extract_image_jbig2(
|
||||||
*, pike: Pdf, root: Path, image: Stream, xref: Xref, options
|
*, pdf: Pdf, root: Path, image: Stream, xref: Xref, options
|
||||||
) -> XrefExt | None:
|
) -> XrefExt | None:
|
||||||
|
"""Extract an image, saving it as a JBIG2 file."""
|
||||||
del options # unused arg
|
del options # unused arg
|
||||||
|
|
||||||
result = extract_image_filter(pike, root, image, xref)
|
result = extract_image_filter(pdf, root, image, xref)
|
||||||
if result is None:
|
if result is None:
|
||||||
return None
|
return None
|
||||||
pim, filtdp = result
|
pim, filtdp = result
|
||||||
@@ -163,9 +168,10 @@ def extract_image_jbig2(
|
|||||||
|
|
||||||
|
|
||||||
def extract_image_generic(
|
def extract_image_generic(
|
||||||
*, pike: Pdf, root: Path, image: Stream, xref: Xref, options
|
*, pdf: Pdf, root: Path, image: Stream, xref: Xref, options
|
||||||
) -> XrefExt | None:
|
) -> XrefExt | None:
|
||||||
result = extract_image_filter(pike, root, image, xref)
|
"""Generic image extraction."""
|
||||||
|
result = extract_image_filter(pdf, root, image, xref)
|
||||||
if result is None:
|
if result is None:
|
||||||
return None
|
return None
|
||||||
pim, filtdp = result
|
pim, filtdp = result
|
||||||
@@ -224,13 +230,72 @@ def extract_image_generic(
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _find_image_xrefs_container(
|
||||||
|
pdf: Pdf,
|
||||||
|
container: Object,
|
||||||
|
pageno: int,
|
||||||
|
include_xrefs: MutableSet[Xref],
|
||||||
|
exclude_xrefs: MutableSet[Xref],
|
||||||
|
pageno_for_xref: dict[Xref, int],
|
||||||
|
depth: int = 0,
|
||||||
|
):
|
||||||
|
"""Find all image XRefs or Form XObject and add to the include/exclude sets."""
|
||||||
|
if depth > 10:
|
||||||
|
log.warning("Recursion depth exceeded in _find_image_xrefs_page")
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
xobjs = container.Resources.XObject
|
||||||
|
except AttributeError:
|
||||||
|
return
|
||||||
|
for _imname, image in dict(xobjs).items():
|
||||||
|
if image.objgen[1] != 0:
|
||||||
|
continue # Ignore images in an incremental PDF
|
||||||
|
if Name.Subtype in image and image.Subtype == Name.Form:
|
||||||
|
# Recurse into Form XObjects
|
||||||
|
log.debug(f"Recursing into Form XObject {_imname} in page {pageno}")
|
||||||
|
_find_image_xrefs_container(
|
||||||
|
pdf,
|
||||||
|
image,
|
||||||
|
pageno,
|
||||||
|
include_xrefs,
|
||||||
|
exclude_xrefs,
|
||||||
|
pageno_for_xref,
|
||||||
|
depth + 1,
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
xref = Xref(image.objgen[0])
|
||||||
|
if Name.SMask in image:
|
||||||
|
# Ignore soft masks
|
||||||
|
smask_xref = Xref(image.SMask.objgen[0])
|
||||||
|
exclude_xrefs.add(smask_xref)
|
||||||
|
log.debug(f"xref {smask_xref}: skipping image because it is an SMask")
|
||||||
|
include_xrefs.add(xref)
|
||||||
|
log.debug(f"xref {xref}: treating as an optimization candidate")
|
||||||
|
if xref not in pageno_for_xref:
|
||||||
|
pageno_for_xref[xref] = pageno
|
||||||
|
|
||||||
|
|
||||||
|
def _find_image_xrefs(pdf: Pdf):
|
||||||
|
include_xrefs: MutableSet[Xref] = set()
|
||||||
|
exclude_xrefs: MutableSet[Xref] = set()
|
||||||
|
pageno_for_xref: dict[Xref, int] = {}
|
||||||
|
|
||||||
|
for pageno, page in enumerate(pdf.pages):
|
||||||
|
_find_image_xrefs_container(
|
||||||
|
pdf, page, pageno, include_xrefs, exclude_xrefs, pageno_for_xref
|
||||||
|
)
|
||||||
|
|
||||||
|
working_xrefs = include_xrefs - exclude_xrefs
|
||||||
|
return working_xrefs, pageno_for_xref
|
||||||
|
|
||||||
|
|
||||||
def extract_images(
|
def extract_images(
|
||||||
pike: Pdf,
|
pdf: Pdf,
|
||||||
root: Path,
|
root: Path,
|
||||||
options,
|
options,
|
||||||
extract_fn: Callable[..., XrefExt | None],
|
extract_fn: Callable[..., XrefExt | None],
|
||||||
) -> Iterator[tuple[int, XrefExt]]:
|
) -> Iterator[tuple[int, XrefExt]]:
|
||||||
"""Extract image using extract_fn
|
"""Extract image using extract_fn.
|
||||||
|
|
||||||
Enumerate images on each page, lookup their xref/ID number in the PDF.
|
Enumerate images on each page, lookup their xref/ID number in the PDF.
|
||||||
Exclude images that are soft masks (i.e. alpha transparency related).
|
Exclude images that are soft masks (i.e. alpha transparency related).
|
||||||
@@ -244,36 +309,13 @@ def extract_images(
|
|||||||
it does a tuple should be returned: (xref, ext) where .ext is the file
|
it does a tuple should be returned: (xref, ext) where .ext is the file
|
||||||
extension. extract_fn must also extract the file it finds interesting.
|
extension. extract_fn must also extract the file it finds interesting.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
include_xrefs: MutableSet[Xref] = set()
|
|
||||||
exclude_xrefs: MutableSet[Xref] = set()
|
|
||||||
pageno_for_xref = {}
|
|
||||||
errors = 0
|
errors = 0
|
||||||
for pageno, page in enumerate(pike.pages):
|
working_xrefs, pageno_for_xref = _find_image_xrefs(pdf)
|
||||||
try:
|
|
||||||
xobjs = page.Resources.XObject
|
|
||||||
except AttributeError:
|
|
||||||
continue
|
|
||||||
for _imname, image in dict(xobjs).items():
|
|
||||||
if image.objgen[1] != 0:
|
|
||||||
continue # Ignore images in an incremental PDF
|
|
||||||
xref = Xref(image.objgen[0])
|
|
||||||
if Name.SMask in image:
|
|
||||||
# Ignore soft masks
|
|
||||||
smask_xref = Xref(image.SMask.objgen[0])
|
|
||||||
exclude_xrefs.add(smask_xref)
|
|
||||||
log.debug(f"xref {smask_xref}: skipping image because it is an SMask")
|
|
||||||
include_xrefs.add(xref)
|
|
||||||
log.debug(f"xref {xref}: treating as an optimization candidate")
|
|
||||||
if xref not in pageno_for_xref:
|
|
||||||
pageno_for_xref[xref] = pageno
|
|
||||||
|
|
||||||
working_xrefs = include_xrefs - exclude_xrefs
|
|
||||||
for xref in working_xrefs:
|
for xref in working_xrefs:
|
||||||
image = pike.get_object((xref, 0))
|
image = pdf.get_object((xref, 0))
|
||||||
try:
|
try:
|
||||||
result = extract_fn(
|
result = extract_fn(
|
||||||
pike=pike, root=root, image=image, xref=xref, options=options
|
pdf=pdf, root=root, image=image, xref=xref, options=options
|
||||||
)
|
)
|
||||||
except Exception: # pylint: disable=broad-except
|
except Exception: # pylint: disable=broad-except
|
||||||
log.exception(
|
log.exception(
|
||||||
@@ -287,13 +329,12 @@ def extract_images(
|
|||||||
|
|
||||||
|
|
||||||
def extract_images_generic(
|
def extract_images_generic(
|
||||||
pike: Pdf, root: Path, options
|
pdf: Pdf, root: Path, options
|
||||||
) -> tuple[list[Xref], list[Xref]]:
|
) -> tuple[list[Xref], list[Xref]]:
|
||||||
"""Extract any >=2bpp image we think we can improve"""
|
"""Extract any >=2bpp image we think we can improve."""
|
||||||
|
|
||||||
jpegs = []
|
jpegs = []
|
||||||
pngs = []
|
pngs = []
|
||||||
for _, xref_ext in extract_images(pike, root, options, extract_image_generic):
|
for _, xref_ext in extract_images(pdf, root, options, extract_image_generic):
|
||||||
log.debug('%s', xref_ext)
|
log.debug('%s', xref_ext)
|
||||||
if xref_ext.ext == '.png':
|
if xref_ext.ext == '.png':
|
||||||
pngs.append(xref_ext.xref)
|
pngs.append(xref_ext.xref)
|
||||||
@@ -303,11 +344,10 @@ def extract_images_generic(
|
|||||||
return jpegs, pngs
|
return jpegs, pngs
|
||||||
|
|
||||||
|
|
||||||
def extract_images_jbig2(pike: Pdf, root: Path, options) -> dict[int, list[XrefExt]]:
|
def extract_images_jbig2(pdf: Pdf, root: Path, options) -> dict[int, list[XrefExt]]:
|
||||||
"""Extract any bitonal image that we think we can improve as JBIG2"""
|
"""Extract any bitonal image that we think we can improve as JBIG2."""
|
||||||
|
|
||||||
jbig2_groups = defaultdict(list)
|
jbig2_groups = defaultdict(list)
|
||||||
for pageno, xref_ext in extract_images(pike, root, options, extract_image_jbig2):
|
for pageno, xref_ext in extract_images(pdf, root, options, extract_image_jbig2):
|
||||||
group = pageno // options.jbig2_page_group_size
|
group = pageno // options.jbig2_page_group_size
|
||||||
jbig2_groups[group].append(xref_ext)
|
jbig2_groups[group].append(xref_ext)
|
||||||
|
|
||||||
@@ -318,7 +358,7 @@ def extract_images_jbig2(pike: Pdf, root: Path, options) -> dict[int, list[XrefE
|
|||||||
def _produce_jbig2_images(
|
def _produce_jbig2_images(
|
||||||
jbig2_groups: dict[int, list[XrefExt]], root: Path, options, executor: Executor
|
jbig2_groups: dict[int, list[XrefExt]], root: Path, options, executor: Executor
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Produce JBIG2 images from their groups"""
|
"""Produce JBIG2 images from their groups."""
|
||||||
|
|
||||||
def jbig2_group_args(root: Path, groups: dict[int, list[XrefExt]]):
|
def jbig2_group_args(root: Path, groups: dict[int, list[XrefExt]]):
|
||||||
for group, xref_exts in groups.items():
|
for group, xref_exts in groups.items():
|
||||||
@@ -327,6 +367,7 @@ def _produce_jbig2_images(
|
|||||||
fspath(root), # =cwd
|
fspath(root), # =cwd
|
||||||
(img_name(root, xref, ext) for xref, ext in xref_exts), # =infiles
|
(img_name(root, xref, ext) for xref, ext in xref_exts), # =infiles
|
||||||
prefix, # =out_prefix
|
prefix, # =out_prefix
|
||||||
|
options.jbig2_threshold,
|
||||||
)
|
)
|
||||||
|
|
||||||
def jbig2_single_args(root, groups: dict[int, list[XrefExt]]):
|
def jbig2_single_args(root, groups: dict[int, list[XrefExt]]):
|
||||||
@@ -339,6 +380,7 @@ def _produce_jbig2_images(
|
|||||||
fspath(root),
|
fspath(root),
|
||||||
img_name(root, xref, ext),
|
img_name(root, xref, ext),
|
||||||
root / f'{prefix}.{n:04d}',
|
root / f'{prefix}.{n:04d}',
|
||||||
|
options.jbig2_threshold,
|
||||||
)
|
)
|
||||||
|
|
||||||
if options.jbig2_page_group_size > 1:
|
if options.jbig2_page_group_size > 1:
|
||||||
@@ -363,7 +405,7 @@ def _produce_jbig2_images(
|
|||||||
|
|
||||||
|
|
||||||
def convert_to_jbig2(
|
def convert_to_jbig2(
|
||||||
pike: Pdf,
|
pdf: Pdf,
|
||||||
jbig2_groups: dict[int, list[XrefExt]],
|
jbig2_groups: dict[int, list[XrefExt]],
|
||||||
root: Path,
|
root: Path,
|
||||||
options,
|
options,
|
||||||
@@ -390,7 +432,7 @@ def convert_to_jbig2(
|
|||||||
jbig2_symfile = root / (prefix + '.sym')
|
jbig2_symfile = root / (prefix + '.sym')
|
||||||
if jbig2_symfile.exists():
|
if jbig2_symfile.exists():
|
||||||
jbig2_globals_data = jbig2_symfile.read_bytes()
|
jbig2_globals_data = jbig2_symfile.read_bytes()
|
||||||
jbig2_globals = Stream(pike, jbig2_globals_data)
|
jbig2_globals = Stream(pdf, jbig2_globals_data)
|
||||||
jbig2_globals_dict = Dictionary(JBIG2Globals=jbig2_globals)
|
jbig2_globals_dict = Dictionary(JBIG2Globals=jbig2_globals)
|
||||||
elif options.jbig2_page_group_size == 1:
|
elif options.jbig2_page_group_size == 1:
|
||||||
jbig2_globals_dict = None
|
jbig2_globals_dict = None
|
||||||
@@ -401,7 +443,7 @@ def convert_to_jbig2(
|
|||||||
xref, _ = xref_ext
|
xref, _ = xref_ext
|
||||||
jbig2_im_file = root / (prefix + f'.{n:04d}')
|
jbig2_im_file = root / (prefix + f'.{n:04d}')
|
||||||
jbig2_im_data = jbig2_im_file.read_bytes()
|
jbig2_im_data = jbig2_im_file.read_bytes()
|
||||||
im_obj = pike.get_object(xref, 0)
|
im_obj = pdf.get_object(xref, 0)
|
||||||
im_obj.write(
|
im_obj.write(
|
||||||
jbig2_im_data, filter=Name.JBIG2Decode, decode_parms=jbig2_globals_dict
|
jbig2_im_data, filter=Name.JBIG2Decode, decode_parms=jbig2_globals_dict
|
||||||
)
|
)
|
||||||
@@ -421,8 +463,10 @@ def _optimize_jpeg(args: tuple[Xref, Path, Path, int]) -> tuple[Xref, Path | Non
|
|||||||
|
|
||||||
|
|
||||||
def transcode_jpegs(
|
def transcode_jpegs(
|
||||||
pike: Pdf, jpegs: Sequence[Xref], root: Path, options, executor: Executor
|
pdf: Pdf, jpegs: Sequence[Xref], root: Path, options, executor: Executor
|
||||||
) -> None:
|
) -> None:
|
||||||
|
"""Optimize JPEGs according to optimization settings."""
|
||||||
|
|
||||||
def jpeg_args() -> Iterator[tuple[Xref, Path, Path, int]]:
|
def jpeg_args() -> Iterator[tuple[Xref, Path, Path, int]]:
|
||||||
for xref in jpegs:
|
for xref in jpegs:
|
||||||
in_jpg = jpg_name(root, xref)
|
in_jpg = jpg_name(root, xref)
|
||||||
@@ -433,7 +477,7 @@ def transcode_jpegs(
|
|||||||
xref, opt_jpg = result
|
xref, opt_jpg = result
|
||||||
if opt_jpg:
|
if opt_jpg:
|
||||||
compdata = opt_jpg.read_bytes() # JPEG can inserted into PDF as is
|
compdata = opt_jpg.read_bytes() # JPEG can inserted into PDF as is
|
||||||
im_obj = pike.get_object(xref, 0)
|
im_obj = pdf.get_object(xref, 0)
|
||||||
im_obj.write(compdata, filter=Name.DCTDecode)
|
im_obj.write(compdata, filter=Name.DCTDecode)
|
||||||
pbar.update()
|
pbar.update()
|
||||||
|
|
||||||
@@ -453,9 +497,9 @@ def transcode_jpegs(
|
|||||||
|
|
||||||
|
|
||||||
def _find_deflatable_jpeg(
|
def _find_deflatable_jpeg(
|
||||||
*, pike: Pdf, root: Path, image: Stream, xref: Xref, options
|
*, pdf: Pdf, root: Path, image: Stream, xref: Xref, options
|
||||||
) -> XrefExt | None:
|
) -> XrefExt | None:
|
||||||
result = extract_image_filter(pike, root, image, xref)
|
result = extract_image_filter(pdf, root, image, xref)
|
||||||
if result is None:
|
if result is None:
|
||||||
return None
|
return None
|
||||||
_pim, filtdp = result
|
_pim, filtdp = result
|
||||||
@@ -467,9 +511,9 @@ def _find_deflatable_jpeg(
|
|||||||
|
|
||||||
|
|
||||||
def _deflate_jpeg(args: tuple[Pdf, threading.Lock, Xref, int]) -> tuple[Xref, bytes]:
|
def _deflate_jpeg(args: tuple[Pdf, threading.Lock, Xref, int]) -> tuple[Xref, bytes]:
|
||||||
pike, lock, xref, complevel = args
|
pdf, lock, xref, complevel = args
|
||||||
with lock:
|
with lock:
|
||||||
xobj = pike.get_object(xref, 0)
|
xobj = pdf.get_object(xref, 0)
|
||||||
try:
|
try:
|
||||||
data = xobj.read_raw_bytes()
|
data = xobj.read_raw_bytes()
|
||||||
except PdfError:
|
except PdfError:
|
||||||
@@ -480,9 +524,15 @@ def _deflate_jpeg(args: tuple[Pdf, threading.Lock, Xref, int]) -> tuple[Xref, by
|
|||||||
return xref, compdata
|
return xref, compdata
|
||||||
|
|
||||||
|
|
||||||
def deflate_jpegs(pike: Pdf, root: Path, options, executor: Executor) -> None:
|
def deflate_jpegs(pdf: Pdf, root: Path, options, executor: Executor) -> None:
|
||||||
|
"""Apply FlateDecode to JPEGs.
|
||||||
|
|
||||||
|
This is a lossless compression method that is supported by all PDF viewers,
|
||||||
|
and generally results in a smaller file size compared to straight DCTDecode
|
||||||
|
images.
|
||||||
|
"""
|
||||||
jpegs = []
|
jpegs = []
|
||||||
for _pageno, xref_ext in extract_images(pike, root, options, _find_deflatable_jpeg):
|
for _pageno, xref_ext in extract_images(pdf, root, options, _find_deflatable_jpeg):
|
||||||
xref = xref_ext.xref
|
xref = xref_ext.xref
|
||||||
log.debug(f'xref {xref}: marking this JPEG as deflatable')
|
log.debug(f'xref {xref}: marking this JPEG as deflatable')
|
||||||
jpegs.append(xref)
|
jpegs.append(xref)
|
||||||
@@ -494,13 +544,13 @@ def deflate_jpegs(pike: Pdf, root: Path, options, executor: Executor) -> None:
|
|||||||
|
|
||||||
def deflate_args() -> Iterator:
|
def deflate_args() -> Iterator:
|
||||||
for xref in jpegs:
|
for xref in jpegs:
|
||||||
yield pike, lock, xref, complevel
|
yield pdf, lock, xref, complevel
|
||||||
|
|
||||||
def finish(result, pbar):
|
def finish(result, pbar):
|
||||||
xref, compdata = result
|
xref, compdata = result
|
||||||
if len(compdata) > 0:
|
if len(compdata) > 0:
|
||||||
with lock:
|
with lock:
|
||||||
xobj = pike.get_object(xref, 0)
|
xobj = pdf.get_object(xref, 0)
|
||||||
xobj.write(compdata, filter=[Name.FlateDecode, Name.DCTDecode])
|
xobj.write(compdata, filter=[Name.FlateDecode, Name.DCTDecode])
|
||||||
pbar.update()
|
pbar.update()
|
||||||
|
|
||||||
@@ -519,16 +569,16 @@ def deflate_jpegs(pike: Pdf, root: Path, options, executor: Executor) -> None:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool:
|
def _transcode_png(pdf: Pdf, filename: Path, xref: Xref) -> bool:
|
||||||
output = filename.with_suffix('.png.pdf')
|
output = filename.with_suffix('.png.pdf')
|
||||||
with output.open('wb') as f:
|
with output.open('wb') as f:
|
||||||
img2pdf.convert(fspath(filename), outputstream=f, **IMG2PDF_KWARGS)
|
img2pdf.convert(fspath(filename), outputstream=f, **IMG2PDF_KWARGS)
|
||||||
|
|
||||||
with Pdf.open(output) as pdf_image:
|
with Pdf.open(output) as pdf_image:
|
||||||
foreign_image = next(iter(pdf_image.pages[0].images.values()))
|
foreign_image = next(iter(pdf_image.pages[0].images.values()))
|
||||||
local_image = pike.copy_foreign(foreign_image)
|
local_image = pdf.copy_foreign(foreign_image)
|
||||||
|
|
||||||
im_obj = pike.get_object(xref, 0)
|
im_obj = pdf.get_object(xref, 0)
|
||||||
im_obj.write(
|
im_obj.write(
|
||||||
local_image.read_raw_bytes(),
|
local_image.read_raw_bytes(),
|
||||||
filter=local_image.Filter,
|
filter=local_image.Filter,
|
||||||
@@ -561,13 +611,14 @@ def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool:
|
|||||||
|
|
||||||
|
|
||||||
def transcode_pngs(
|
def transcode_pngs(
|
||||||
pike: Pdf,
|
pdf: Pdf,
|
||||||
images: Sequence[Xref],
|
images: Sequence[Xref],
|
||||||
image_name_fn: Callable[[Path, Xref], Path],
|
image_name_fn: Callable[[Path, Xref], Path],
|
||||||
root: Path,
|
root: Path,
|
||||||
options,
|
options,
|
||||||
executor,
|
executor,
|
||||||
) -> None:
|
) -> None:
|
||||||
|
"""Apply lossy transcoding to PNGs."""
|
||||||
modified: MutableSet[Xref] = set()
|
modified: MutableSet[Xref] = set()
|
||||||
if options.optimize >= 2:
|
if options.optimize >= 2:
|
||||||
png_quality = (
|
png_quality = (
|
||||||
@@ -601,7 +652,7 @@ def transcode_pngs(
|
|||||||
|
|
||||||
for xref in modified:
|
for xref in modified:
|
||||||
filename = png_name(root, xref)
|
filename = png_name(root, xref)
|
||||||
_transcode_png(pike, filename, xref)
|
_transcode_png(pdf, filename, xref)
|
||||||
|
|
||||||
|
|
||||||
DEFAULT_EXECUTOR = SerialExecutor()
|
DEFAULT_EXECUTOR = SerialExecutor()
|
||||||
@@ -614,6 +665,7 @@ def optimize(
|
|||||||
save_settings,
|
save_settings,
|
||||||
executor: Executor = DEFAULT_EXECUTOR,
|
executor: Executor = DEFAULT_EXECUTOR,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
|
"""Optimize images in a PDF file."""
|
||||||
options = context.options
|
options = context.options
|
||||||
if options.optimize == 0:
|
if options.optimize == 0:
|
||||||
safe_symlink(input_file, output_file)
|
safe_symlink(input_file, output_file)
|
||||||
@@ -626,24 +678,24 @@ def optimize(
|
|||||||
if options.jbig2_page_group_size == 0:
|
if options.jbig2_page_group_size == 0:
|
||||||
options.jbig2_page_group_size = 10 if options.jbig2_lossy else 1
|
options.jbig2_page_group_size = 10 if options.jbig2_lossy else 1
|
||||||
|
|
||||||
with Pdf.open(input_file) as pike:
|
with Pdf.open(input_file) as pdf:
|
||||||
root = output_file.parent / 'images'
|
root = output_file.parent / 'images'
|
||||||
root.mkdir(exist_ok=True)
|
root.mkdir(exist_ok=True)
|
||||||
|
|
||||||
jpegs, pngs = extract_images_generic(pike, root, options)
|
jpegs, pngs = extract_images_generic(pdf, root, options)
|
||||||
transcode_jpegs(pike, jpegs, root, options, executor)
|
transcode_jpegs(pdf, jpegs, root, options, executor)
|
||||||
deflate_jpegs(pike, root, options, executor)
|
deflate_jpegs(pdf, root, options, executor)
|
||||||
# if options.optimize >= 2:
|
# if options.optimize >= 2:
|
||||||
# Try pngifying the jpegs
|
# Try pngifying the jpegs
|
||||||
# transcode_pngs(pike, jpegs, jpg_name, root, options)
|
# transcode_pngs(pdf, jpegs, jpg_name, root, options)
|
||||||
transcode_pngs(pike, pngs, png_name, root, options, executor)
|
transcode_pngs(pdf, pngs, png_name, root, options, executor)
|
||||||
|
|
||||||
jbig2_groups = extract_images_jbig2(pike, root, options)
|
jbig2_groups = extract_images_jbig2(pdf, root, options)
|
||||||
convert_to_jbig2(pike, jbig2_groups, root, options, executor)
|
convert_to_jbig2(pdf, jbig2_groups, root, options, executor)
|
||||||
|
|
||||||
target_file = output_file.with_suffix('.opt.pdf')
|
target_file = output_file.with_suffix('.opt.pdf')
|
||||||
pike.remove_unreferenced_resources()
|
pdf.remove_unreferenced_resources()
|
||||||
pike.save(target_file, **save_settings)
|
pdf.save(target_file, **save_settings)
|
||||||
|
|
||||||
input_size = input_file.stat().st_size
|
input_size = input_file.stat().st_size
|
||||||
output_size = target_file.stat().st_size
|
output_size = target_file.stat().st_size
|
||||||
@@ -660,9 +712,9 @@ def optimize(
|
|||||||
"optimizations will not be used"
|
"optimizations will not be used"
|
||||||
)
|
)
|
||||||
# We still need to save the file
|
# We still need to save the file
|
||||||
with Pdf.open(input_file) as pike:
|
with Pdf.open(input_file) as pdf:
|
||||||
pike.remove_unreferenced_resources()
|
pdf.remove_unreferenced_resources()
|
||||||
pike.save(output_file, **save_settings)
|
pdf.save(output_file, **save_settings)
|
||||||
else:
|
else:
|
||||||
safe_symlink(target_file, output_file)
|
safe_symlink(target_file, output_file)
|
||||||
|
|
||||||
@@ -670,11 +722,12 @@ def optimize(
|
|||||||
|
|
||||||
|
|
||||||
def main(infile, outfile, level, jobs=1):
|
def main(infile, outfile, level, jobs=1):
|
||||||
|
"""Entry point for direct optimization of a file."""
|
||||||
from shutil import copy # pylint: disable=import-outside-toplevel
|
from shutil import copy # pylint: disable=import-outside-toplevel
|
||||||
from tempfile import TemporaryDirectory # pylint: disable=import-outside-toplevel
|
from tempfile import TemporaryDirectory # pylint: disable=import-outside-toplevel
|
||||||
|
|
||||||
class OptimizeOptions:
|
class OptimizeOptions:
|
||||||
"""Emulate ocrmypdf's options"""
|
"""Emulate ocrmypdf's options."""
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self, input_file, jobs, optimize_, jpeg_quality, png_quality, jb2lossy
|
self, input_file, jobs, optimize_, jpeg_quality, png_quality, jb2lossy
|
||||||
@@ -686,6 +739,7 @@ def main(infile, outfile, level, jobs=1):
|
|||||||
self.png_quality = png_quality
|
self.png_quality = png_quality
|
||||||
self.jbig2_page_group_size = 0
|
self.jbig2_page_group_size = 0
|
||||||
self.jbig2_lossy = jb2lossy
|
self.jbig2_lossy = jb2lossy
|
||||||
|
self.jbig2_threshold = 0.85
|
||||||
self.quiet = True
|
self.quiet = True
|
||||||
self.progress_bar = False
|
self.progress_bar = False
|
||||||
|
|
||||||
|
|||||||
+6
-13
@@ -1,21 +1,15 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""
|
"""Utilities for PDF/A production and confirmation with Ghostspcript."""
|
||||||
Utilities for PDF/A production and confirmation with Ghostspcript.
|
|
||||||
"""
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import base64
|
import base64
|
||||||
|
from importlib.resources import files as package_files
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Iterator
|
from typing import Iterator
|
||||||
|
|
||||||
try:
|
|
||||||
from importlib.resources import files as package_files
|
|
||||||
except ImportError:
|
|
||||||
from importlib_resources import files as package_files # type: ignore
|
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
|
|
||||||
SRGB_ICC_PROFILE_NAME = 'sRGB.icc'
|
SRGB_ICC_PROFILE_NAME = 'sRGB.icc'
|
||||||
@@ -25,8 +19,8 @@ def _postscript_objdef(
|
|||||||
alias: str,
|
alias: str,
|
||||||
dictionary: dict[str, str],
|
dictionary: dict[str, str],
|
||||||
*,
|
*,
|
||||||
stream_name: str = None,
|
stream_name: str | None = None,
|
||||||
stream_data: bytes = None,
|
stream_data: bytes | None = None,
|
||||||
) -> Iterator[str]:
|
) -> Iterator[str]:
|
||||||
assert (stream_name is None) == (stream_data is None)
|
assert (stream_name is None) == (stream_data is None)
|
||||||
|
|
||||||
@@ -75,7 +69,7 @@ def _make_postscript(icc_name: str, icc_data: bytes, colors: int) -> Iterator[st
|
|||||||
|
|
||||||
|
|
||||||
def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
||||||
"""Create a Postscript PDFMARK file for Ghostscript PDF/A conversion
|
"""Create a Postscript PDFMARK file for Ghostscript PDF/A conversion.
|
||||||
|
|
||||||
pdfmark is an extension to the Postscript language that describes some PDF
|
pdfmark is an extension to the Postscript language that describes some PDF
|
||||||
features like bookmarks and annotations. It was originally specified Adobe
|
features like bookmarks and annotations. It was originally specified Adobe
|
||||||
@@ -84,7 +78,7 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
|||||||
Ghostscript uses pdfmark for PDF to PDF/A conversion as well. To use Ghostscript
|
Ghostscript uses pdfmark for PDF to PDF/A conversion as well. To use Ghostscript
|
||||||
to create a PDF/A, we need to create a pdfmark file with the necessary metadata.
|
to create a PDF/A, we need to create a pdfmark file with the necessary metadata.
|
||||||
|
|
||||||
This function takes care of the many version-specific bugs and pecularities in
|
This function takes care of the many version-specific bugs and peculiarities in
|
||||||
Ghostscript's handling of pdfmark.
|
Ghostscript's handling of pdfmark.
|
||||||
|
|
||||||
The only information we put in specifies that we want the file to be a
|
The only information we put in specifies that we want the file to be a
|
||||||
@@ -118,7 +112,6 @@ def file_claims_pdfa(filename: Path):
|
|||||||
This only checks if the XMP metadata contains a PDF/A marker. It does not
|
This only checks if the XMP metadata contains a PDF/A marker. It does not
|
||||||
do full PDF/A validation.
|
do full PDF/A validation.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
with pikepdf.open(filename) as pdf:
|
with pikepdf.open(filename) as pdf:
|
||||||
pdfmeta = pdf.open_metadata()
|
pdfmeta = pdf.open_metadata()
|
||||||
if not pdfmeta.pdfa_status:
|
if not pdfmeta.pdfa_status:
|
||||||
|
|||||||
@@ -6,4 +6,6 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PdfInfo
|
from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PageInfo, PdfInfo
|
||||||
|
|
||||||
|
__all__ = ["Colorspace", "Encoding", "PageInfo", "PdfInfo"]
|
||||||
|
|||||||
+198
-67
@@ -9,6 +9,8 @@ from __future__ import annotations
|
|||||||
import atexit
|
import atexit
|
||||||
import logging
|
import logging
|
||||||
import re
|
import re
|
||||||
|
import statistics
|
||||||
|
import sys
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from contextlib import ExitStack
|
from contextlib import ExitStack
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
@@ -17,20 +19,13 @@ from functools import partial
|
|||||||
from math import hypot, inf, isclose
|
from math import hypot, inf, isclose
|
||||||
from os import PathLike
|
from os import PathLike
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import (
|
from typing import Container, Iterable, Iterator, Mapping, NamedTuple, Sequence, Tuple
|
||||||
Container,
|
|
||||||
Iterable,
|
|
||||||
Iterator,
|
|
||||||
Mapping,
|
|
||||||
NamedTuple,
|
|
||||||
Optional,
|
|
||||||
Sequence,
|
|
||||||
Tuple,
|
|
||||||
)
|
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
from pikepdf import (
|
from pikepdf import (
|
||||||
|
Name,
|
||||||
Object,
|
Object,
|
||||||
|
Page,
|
||||||
Pdf,
|
Pdf,
|
||||||
PdfImage,
|
PdfImage,
|
||||||
PdfInlineImage,
|
PdfInlineImage,
|
||||||
@@ -173,7 +168,7 @@ class TextMarker:
|
|||||||
|
|
||||||
|
|
||||||
def _normalize_stack(graphobjs):
|
def _normalize_stack(graphobjs):
|
||||||
"""Convert runs of qQ's in the stack into single graphobjs"""
|
"""Convert runs of qQ's in the stack into single graphobjs."""
|
||||||
for operands, operator in graphobjs:
|
for operands, operator in graphobjs:
|
||||||
operator = str(operator)
|
operator = str(operator)
|
||||||
if re.match(r'Q*q+$', operator): # Zero or more Q, one or more q
|
if re.match(r'Q*q+$', operator): # Zero or more Q, one or more q
|
||||||
@@ -209,7 +204,6 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
|||||||
undefined in the spec, but we just pretend nothing happened and leave the
|
undefined in the spec, but we just pretend nothing happened and leave the
|
||||||
CTM unchanged.
|
CTM unchanged.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
stack = []
|
stack = []
|
||||||
ctm = PdfMatrix(initial_shorthand)
|
ctm = PdfMatrix(initial_shorthand)
|
||||||
xobject_settings: list[XobjectSettings] = []
|
xobject_settings: list[XobjectSettings] = []
|
||||||
@@ -316,7 +310,6 @@ def _get_dpi(ctm_shorthand, image_size) -> Resolution:
|
|||||||
/MediaBox.
|
/MediaBox.
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
a, b, c, d, _, _ = ctm_shorthand # pylint: disable=invalid-name
|
a, b, c, d, _, _ = ctm_shorthand # pylint: disable=invalid-name
|
||||||
|
|
||||||
# Calculate the width and height of the image in PDF units
|
# Calculate the width and height of the image in PDF units
|
||||||
@@ -333,7 +326,12 @@ def _get_dpi(ctm_shorthand, image_size) -> Resolution:
|
|||||||
|
|
||||||
|
|
||||||
class ImageInfo:
|
class ImageInfo:
|
||||||
"""Information about an image found in a PDF."""
|
"""Information about an image found in a PDF.
|
||||||
|
|
||||||
|
This gathers information from pikepdf and pdfminer.six, and is pickle-able
|
||||||
|
so that it can be passed to a worker process, unlike objects from those
|
||||||
|
libraries.
|
||||||
|
"""
|
||||||
|
|
||||||
DPI_PREC = Decimal('1.000')
|
DPI_PREC = Decimal('1.000')
|
||||||
|
|
||||||
@@ -348,6 +346,7 @@ class ImageInfo:
|
|||||||
inline: PdfInlineImage | None = None,
|
inline: PdfInlineImage | None = None,
|
||||||
shorthand=None,
|
shorthand=None,
|
||||||
):
|
):
|
||||||
|
"""Initialize an ImageInfo."""
|
||||||
self._name = str(name)
|
self._name = str(name)
|
||||||
self._shorthand = shorthand
|
self._shorthand = shorthand
|
||||||
|
|
||||||
@@ -414,54 +413,77 @@ class ImageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def name(self):
|
def name(self):
|
||||||
|
"""Name of the image as it appears in the PDF."""
|
||||||
return self._name
|
return self._name
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def type_(self):
|
def type_(self):
|
||||||
|
"""Type of image, either 'image' or 'stencil'."""
|
||||||
return self._type
|
return self._type
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def width(self):
|
def width(self) -> int:
|
||||||
|
"""Width of the image in pixels."""
|
||||||
return self._width
|
return self._width
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def height(self):
|
def height(self) -> int:
|
||||||
|
"""Height of the image in pixels."""
|
||||||
return self._height
|
return self._height
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def bpc(self):
|
def bpc(self):
|
||||||
|
"""Bits per component."""
|
||||||
return self._bpc
|
return self._bpc
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def color(self):
|
def color(self):
|
||||||
|
"""Colorspace of the image."""
|
||||||
return self._color if self._color is not None else '?'
|
return self._color if self._color is not None else '?'
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def comp(self):
|
def comp(self):
|
||||||
|
"""Number of components/channels in the image."""
|
||||||
return self._comp if self._comp is not None else '?'
|
return self._comp if self._comp is not None else '?'
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def enc(self):
|
def enc(self):
|
||||||
|
"""Encoding of the image."""
|
||||||
return self._enc if self._enc is not None else 'image'
|
return self._enc if self._enc is not None else 'image'
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def renderable(self):
|
def renderable(self) -> bool:
|
||||||
|
"""Whether the image is renderable.
|
||||||
|
|
||||||
|
Some PDFs in the wild have invalid images that are not renderable.
|
||||||
|
"""
|
||||||
return self.dpi.is_finite and self.width >= 0 and self.height >= 0
|
return self.dpi.is_finite and self.width >= 0 and self.height >= 0
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def dpi(self):
|
def dpi(self) -> Resolution:
|
||||||
|
"""Dots per inch of the image.
|
||||||
|
|
||||||
|
Calculated based on where and how the image is drawn in the PDF.
|
||||||
|
"""
|
||||||
return _get_dpi(self._shorthand, (self._width, self._height))
|
return _get_dpi(self._shorthand, (self._width, self._height))
|
||||||
|
|
||||||
|
@property
|
||||||
|
def printed_area(self) -> float:
|
||||||
|
"""Physical area of the image in square inches."""
|
||||||
|
if not self.renderable:
|
||||||
|
return 0.0
|
||||||
|
return float(self.width * self.dpi.x * self.height * self.dpi.y)
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
|
"""Return a string representation of the image."""
|
||||||
return (
|
return (
|
||||||
f"<ImageInfo '{self.name}' {self.type_} {self.width}x{self.height} "
|
f"<ImageInfo '{self.name}' {self.type_} {self.width}×{self.height} "
|
||||||
f"{self.color} {self.comp} {self.bpc} {self.enc} {self.dpi}>"
|
f"{self.color} {self.comp} {self.bpc} {self.enc} {self.dpi}>"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
||||||
"Find inline images in the contentstream"
|
"""Find inline images in the contentstream."""
|
||||||
|
|
||||||
for n, inline in enumerate(contentsinfo.inline_images):
|
for n, inline in enumerate(contentsinfo.inline_images):
|
||||||
yield ImageInfo(
|
yield ImageInfo(
|
||||||
name=f'inline-{n:02d}', shorthand=inline.shorthand, inline=inline.iimage
|
name=f'inline-{n:02d}', shorthand=inline.shorthand, inline=inline.iimage
|
||||||
@@ -469,7 +491,7 @@ def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
|||||||
|
|
||||||
|
|
||||||
def _image_xobjects(container) -> Iterator[tuple[Object, str]]:
|
def _image_xobjects(container) -> Iterator[tuple[Object, str]]:
|
||||||
"""Search for all XObject-based images in the container
|
"""Search for all XObject-based images in the container.
|
||||||
|
|
||||||
Usually the container is a page, but it could also be a Form XObject
|
Usually the container is a page, but it could also be a Form XObject
|
||||||
that contains images. Filter out the Form XObjects which are dealt with
|
that contains images. Filter out the Form XObjects which are dealt with
|
||||||
@@ -480,33 +502,29 @@ def _image_xobjects(container) -> Iterator[tuple[Object, str]]:
|
|||||||
since the object does not know its own name.
|
since the object does not know its own name.
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
if Name.Resources not in container:
|
||||||
if '/Resources' not in container:
|
|
||||||
return
|
return
|
||||||
resources = container['/Resources']
|
resources = container[Name.Resources]
|
||||||
if '/XObject' not in resources:
|
if Name.XObject not in resources:
|
||||||
return
|
return
|
||||||
xobjs = resources['/XObject'].as_dict()
|
for key, candidate in resources[Name.XObject].items():
|
||||||
for xobj in xobjs:
|
if candidate is None or Name.Subtype not in candidate:
|
||||||
candidate: Object = xobjs[xobj]
|
|
||||||
if '/Subtype' not in candidate:
|
|
||||||
continue
|
continue
|
||||||
if candidate['/Subtype'] == '/Image':
|
if candidate[Name.Subtype] == Name.Image:
|
||||||
pdfimage = candidate
|
pdfimage = candidate
|
||||||
yield (pdfimage, xobj)
|
yield (pdfimage, key)
|
||||||
|
|
||||||
|
|
||||||
def _find_regular_images(
|
def _find_regular_images(
|
||||||
container: Object, contentsinfo: ContentsInfo
|
container: Object, contentsinfo: ContentsInfo
|
||||||
) -> Iterator[ImageInfo]:
|
) -> Iterator[ImageInfo]:
|
||||||
"""Find images stored in the container's /Resources /XObject
|
"""Find images stored in the container's /Resources /XObject.
|
||||||
|
|
||||||
Usually the container is a page, but it could also be a Form XObject
|
Usually the container is a page, but it could also be a Form XObject
|
||||||
that contains images.
|
that contains images.
|
||||||
|
|
||||||
Generates images with their DPI at time of drawing.
|
Generates images with their DPI at time of drawing.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
for pdfimage, xobj in _image_xobjects(container):
|
for pdfimage, xobj in _image_xobjects(container):
|
||||||
if xobj not in contentsinfo.name_index:
|
if xobj not in contentsinfo.name_index:
|
||||||
continue
|
continue
|
||||||
@@ -523,20 +541,20 @@ def _find_regular_images(
|
|||||||
|
|
||||||
|
|
||||||
def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo):
|
def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo):
|
||||||
"""Find any images that are in Form XObjects in the container
|
"""Find any images that are in Form XObjects in the container.
|
||||||
|
|
||||||
The container may be a page, or a parent Form XObject.
|
The container may be a page, or a parent Form XObject.
|
||||||
|
|
||||||
"""
|
"""
|
||||||
if '/Resources' not in container:
|
if Name.Resources not in container:
|
||||||
return
|
return
|
||||||
resources = container['/Resources']
|
resources = container[Name.Resources]
|
||||||
if '/XObject' not in resources:
|
if Name.XObject not in resources:
|
||||||
return
|
return
|
||||||
xobjs = resources['/XObject'].as_dict()
|
xobjs = resources[Name.XObject].as_dict()
|
||||||
for xobj in xobjs:
|
for xobj in xobjs:
|
||||||
candidate = xobjs[xobj]
|
candidate = xobjs[xobj]
|
||||||
if candidate['/Subtype'] != '/Form':
|
if candidate is None or candidate[Name.Subtype] != Name.Form:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
form_xobject = candidate
|
form_xobject = candidate
|
||||||
@@ -557,7 +575,7 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
|||||||
def _process_content_streams(
|
def _process_content_streams(
|
||||||
*, pdf: Pdf, container: Object, shorthand=None
|
*, pdf: Pdf, container: Object, shorthand=None
|
||||||
) -> Iterator[VectorMarker | TextMarker | ImageInfo]:
|
) -> Iterator[VectorMarker | TextMarker | ImageInfo]:
|
||||||
"""Find all individual instances of images drawn in the container
|
"""Find all individual instances of images drawn in the container.
|
||||||
|
|
||||||
Usually the container is a page, but it may also be a Form XObject.
|
Usually the container is a page, but it may also be a Form XObject.
|
||||||
|
|
||||||
@@ -574,17 +592,19 @@ def _process_content_streams(
|
|||||||
downsampling.
|
downsampling.
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
if container.get(Name.Type) == Name.Page and Name.Contents in container:
|
||||||
if container.get('/Type') == '/Page' and '/Contents' in container:
|
|
||||||
initial_shorthand = shorthand or UNIT_SQUARE
|
initial_shorthand = shorthand or UNIT_SQUARE
|
||||||
elif container.get('/Type') == '/XObject' and container['/Subtype'] == '/Form':
|
elif (
|
||||||
|
container.get(Name.Type) == Name.XObject
|
||||||
|
and container[Name.Subtype] == Name.Form
|
||||||
|
):
|
||||||
# Set the CTM to the state it was when the "Do" operator was
|
# Set the CTM to the state it was when the "Do" operator was
|
||||||
# encountered that is drawing this instance of the Form XObject
|
# encountered that is drawing this instance of the Form XObject
|
||||||
ctm = PdfMatrix(shorthand) if shorthand else PdfMatrix.identity()
|
ctm = PdfMatrix(shorthand) if shorthand else PdfMatrix.identity()
|
||||||
|
|
||||||
# A Form XObject may provide its own matrix to map form space into
|
# A Form XObject may provide its own matrix to map form space into
|
||||||
# user space. Get this if one exists
|
# user space. Get this if one exists
|
||||||
form_shorthand = container.get('/Matrix', PdfMatrix.identity())
|
form_shorthand = container.get(Name.Matrix, PdfMatrix.identity())
|
||||||
form_matrix = PdfMatrix(form_shorthand)
|
form_matrix = PdfMatrix(form_shorthand)
|
||||||
|
|
||||||
# Concatenate form matrix with CTM to ensure CTM is correct for
|
# Concatenate form matrix with CTM to ensure CTM is correct for
|
||||||
@@ -606,8 +626,7 @@ def _process_content_streams(
|
|||||||
|
|
||||||
|
|
||||||
def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) -> bool:
|
def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) -> bool:
|
||||||
"""Smarter text detection that ignores text in margins"""
|
"""Smarter text detection that ignores text in margins."""
|
||||||
|
|
||||||
pw, ph = float(page_width), float(page_height) # pylint: disable=invalid-name
|
pw, ph = float(page_width), float(page_height) # pylint: disable=invalid-name
|
||||||
|
|
||||||
margin_ratio = 0.125
|
margin_ratio = 0.125
|
||||||
@@ -619,10 +638,11 @@ def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) ->
|
|||||||
)
|
)
|
||||||
|
|
||||||
def rects_intersect(a: FloatRect, b: FloatRect) -> bool:
|
def rects_intersect(a: FloatRect, b: FloatRect) -> bool:
|
||||||
"""
|
"""Check if two 4-tuple rects intersect.
|
||||||
|
|
||||||
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
|
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
|
||||||
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
|
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
|
||||||
Formula assumes all boxes are in first quadrant
|
Formula assumes all boxes are in first quadrant.
|
||||||
"""
|
"""
|
||||||
return a[0] < b[2] and a[2] > b[0] and a[1] > b[3] and a[3] < b[1]
|
return a[0] < b[2] and a[2] > b[0] and a[1] > b[3] and a[3] < b[1]
|
||||||
|
|
||||||
@@ -635,7 +655,7 @@ def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) ->
|
|||||||
|
|
||||||
|
|
||||||
def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]:
|
def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]:
|
||||||
"""Extract only limited content from text boxes
|
"""Extract only limited content from text boxes.
|
||||||
|
|
||||||
We do this to save memory and ensure that our objects are pickleable.
|
We do this to save memory and ensure that our objects are pickleable.
|
||||||
"""
|
"""
|
||||||
@@ -736,12 +756,38 @@ def _pdf_pageinfo_concurrent(
|
|||||||
return pages
|
return pages
|
||||||
|
|
||||||
|
|
||||||
|
class PageResolutionProfile(NamedTuple):
|
||||||
|
"""Information about the resolutions of a page."""
|
||||||
|
|
||||||
|
weighted_dpi: float
|
||||||
|
"""The weighted average DPI of the page, weighted by the area of each image."""
|
||||||
|
|
||||||
|
max_dpi: float
|
||||||
|
"""The maximum DPI of an image on the page."""
|
||||||
|
|
||||||
|
average_to_max_dpi_ratio: float
|
||||||
|
"""The average DPI of the page divided by the maximum DPI of the page.
|
||||||
|
|
||||||
|
This indicates the intensity of the resolution variation on the page.
|
||||||
|
|
||||||
|
If the average is 1.0 or close to 1.0, has all of its content at a uniform
|
||||||
|
resolution. If the average is much lower than 1.0, some content is at a
|
||||||
|
higher resolution than the rest of the page.
|
||||||
|
"""
|
||||||
|
|
||||||
|
area_ratio: float
|
||||||
|
"""The maximum-DPI area of the page divided by the total drawn area.
|
||||||
|
|
||||||
|
This indicates the prevalence of high-resolution content on the page.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
class PageInfo:
|
class PageInfo:
|
||||||
"""Information about type of contents on each page in a PDF."""
|
"""Information about type of contents on each page in a PDF."""
|
||||||
|
|
||||||
_has_text: bool | None
|
_has_text: bool | None
|
||||||
_has_vector: bool | None
|
_has_vector: bool | None
|
||||||
_images: list[ImageInfo]
|
_images: list[ImageInfo] = []
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
@@ -751,6 +797,7 @@ class PageInfo:
|
|||||||
check_pages: Container[int],
|
check_pages: Container[int],
|
||||||
detailed_analysis: bool = False,
|
detailed_analysis: bool = False,
|
||||||
):
|
):
|
||||||
|
"""Initialize a PageInfo object."""
|
||||||
self._pageno = pageno
|
self._pageno = pageno
|
||||||
self._infile = infile
|
self._infile = infile
|
||||||
self._detailed_analysis = detailed_analysis
|
self._detailed_analysis = detailed_analysis
|
||||||
@@ -764,7 +811,7 @@ class PageInfo:
|
|||||||
check_pages: Container[int],
|
check_pages: Container[int],
|
||||||
detailed_analysis: bool,
|
detailed_analysis: bool,
|
||||||
):
|
):
|
||||||
page = pdf.pages[pageno]
|
page: Page = pdf.pages[pageno]
|
||||||
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
|
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
|
||||||
width_pt = mediabox[2] - mediabox[0]
|
width_pt = mediabox[2] - mediabox[0]
|
||||||
height_pt = mediabox[3] - mediabox[1]
|
height_pt = mediabox[3] - mediabox[1]
|
||||||
@@ -772,7 +819,7 @@ class PageInfo:
|
|||||||
check_this_page = pageno in check_pages
|
check_this_page = pageno in check_pages
|
||||||
|
|
||||||
if check_this_page and detailed_analysis:
|
if check_this_page and detailed_analysis:
|
||||||
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
|
pscript5_mode = str(pdf.docinfo.get(Name.Creator)).startswith('PScript5')
|
||||||
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
||||||
self._textboxes = list(simplify_textboxes(miner, get_text_boxes))
|
self._textboxes = list(simplify_textboxes(miner, get_text_boxes))
|
||||||
bboxes = (box.bbox for box in self._textboxes)
|
bboxes = (box.bbox for box in self._textboxes)
|
||||||
@@ -782,17 +829,13 @@ class PageInfo:
|
|||||||
self._textboxes = []
|
self._textboxes = []
|
||||||
self._has_text = None # i.e. "no information"
|
self._has_text = None # i.e. "no information"
|
||||||
|
|
||||||
userunit = page.get('/UserUnit', Decimal(1.0))
|
userunit = page.get(Name.UserUnit, Decimal(1.0))
|
||||||
if not isinstance(userunit, Decimal):
|
if not isinstance(userunit, Decimal):
|
||||||
userunit = Decimal(userunit)
|
userunit = Decimal(userunit)
|
||||||
self._userunit = userunit
|
self._userunit = userunit
|
||||||
self._width_inches = width_pt * userunit / Decimal(72.0)
|
self._width_inches = width_pt * userunit / Decimal(72.0)
|
||||||
self._height_inches = height_pt * userunit / Decimal(72.0)
|
self._height_inches = height_pt * userunit / Decimal(72.0)
|
||||||
|
self._rotate = int(getattr(page.obj, 'Rotate', 0))
|
||||||
try:
|
|
||||||
self._rotate = int(page['/Rotate'])
|
|
||||||
except KeyError:
|
|
||||||
self._rotate = 0
|
|
||||||
|
|
||||||
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
|
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
|
||||||
|
|
||||||
@@ -827,40 +870,56 @@ class PageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def pageno(self) -> int:
|
def pageno(self) -> int:
|
||||||
|
"""Return page number (0-based)."""
|
||||||
return self._pageno
|
return self._pageno
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_text(self) -> bool:
|
def has_text(self) -> bool:
|
||||||
|
"""Return True if page has text, False if not or unknown."""
|
||||||
return bool(self._has_text)
|
return bool(self._has_text)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_corrupt_text(self) -> bool:
|
def has_corrupt_text(self) -> bool:
|
||||||
|
"""Return True if page has corrupt text, False if not or unknown."""
|
||||||
if not self._detailed_analysis:
|
if not self._detailed_analysis:
|
||||||
raise NotImplementedError('Did not do detailed analysis')
|
raise NotImplementedError('Did not do detailed analysis')
|
||||||
return any(tbox.is_corrupt for tbox in self._textboxes)
|
return any(tbox.is_corrupt for tbox in self._textboxes)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_vector(self) -> bool:
|
def has_vector(self) -> bool:
|
||||||
|
"""Return True if page has vector graphics, False if not or unknown.
|
||||||
|
|
||||||
|
Vector graphics are sometimes used to draw fonts, so it may not be
|
||||||
|
obvious on visual inspection whether a page has text or not.
|
||||||
|
"""
|
||||||
return bool(self._has_vector)
|
return bool(self._has_vector)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def width_inches(self) -> Decimal:
|
def width_inches(self) -> Decimal:
|
||||||
|
"""Return width of page in inches."""
|
||||||
return self._width_inches
|
return self._width_inches
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def height_inches(self) -> Decimal:
|
def height_inches(self) -> Decimal:
|
||||||
|
"""Return height of page in inches."""
|
||||||
return self._height_inches
|
return self._height_inches
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def width_pixels(self) -> int:
|
def width_pixels(self) -> int:
|
||||||
|
"""Return width of page in pixels."""
|
||||||
return int(round(float(self.width_inches) * self.dpi.x))
|
return int(round(float(self.width_inches) * self.dpi.x))
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def height_pixels(self) -> int:
|
def height_pixels(self) -> int:
|
||||||
|
"""Return height of page in pixels."""
|
||||||
return int(round(float(self.height_inches) * self.dpi.y))
|
return int(round(float(self.height_inches) * self.dpi.y))
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def rotation(self) -> int:
|
def rotation(self) -> int:
|
||||||
|
"""Return rotation of page in degrees.
|
||||||
|
|
||||||
|
Will only be a multiple of 90.
|
||||||
|
"""
|
||||||
return self._rotate
|
return self._rotate
|
||||||
|
|
||||||
@rotation.setter
|
@rotation.setter
|
||||||
@@ -871,10 +930,13 @@ class PageInfo:
|
|||||||
raise ValueError("rotation must be a cardinal angle")
|
raise ValueError("rotation must be a cardinal angle")
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def images(self):
|
def images(self) -> list[ImageInfo]:
|
||||||
|
"""Return images."""
|
||||||
return self._images
|
return self._images
|
||||||
|
|
||||||
def get_textareas(self, visible: bool | None = None, corrupt: bool | None = None):
|
def get_textareas(self, visible: bool | None = None, corrupt: bool | None = None):
|
||||||
|
"""Return textareas bounding boxes in PDF coordinates on the page."""
|
||||||
|
|
||||||
def predicate(obj, want_visible, want_corrupt):
|
def predicate(obj, want_visible, want_corrupt):
|
||||||
result = True
|
result = True
|
||||||
if want_visible is not None:
|
if want_visible is not None:
|
||||||
@@ -894,22 +956,67 @@ class PageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def dpi(self) -> Resolution:
|
def dpi(self) -> Resolution:
|
||||||
|
"""Return DPI needed to render all images on the page."""
|
||||||
if self._dpi is None:
|
if self._dpi is None:
|
||||||
return Resolution(0.0, 0.0)
|
return Resolution(0.0, 0.0)
|
||||||
return self._dpi
|
return self._dpi
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def userunit(self) -> Decimal:
|
def userunit(self) -> Decimal:
|
||||||
|
"""Return user unit of page."""
|
||||||
return self._userunit
|
return self._userunit
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def min_version(self) -> str:
|
def min_version(self) -> str:
|
||||||
|
"""Return minimum PDF version needed to render this page."""
|
||||||
if self.userunit is not None:
|
if self.userunit is not None:
|
||||||
return '1.6'
|
return '1.6'
|
||||||
else:
|
else:
|
||||||
return '1.5'
|
return '1.5'
|
||||||
|
|
||||||
|
def page_dpi_profile(self) -> PageResolutionProfile | None:
|
||||||
|
"""Return information about the DPIs of the page.
|
||||||
|
|
||||||
|
This is useful to detect pages with a small proportion of high-resolution
|
||||||
|
content that is forcing us to use a high DPI for the whole page. The ratio
|
||||||
|
is weighted by the area of each image. If images overlap, the overlapped
|
||||||
|
area counts.
|
||||||
|
|
||||||
|
Vector graphics and text are ignored.
|
||||||
|
|
||||||
|
Returns None if there is no meaningful DPI for the page.
|
||||||
|
"""
|
||||||
|
image_dpis = [
|
||||||
|
image.dpi.to_scalar() for image in self._images if image.renderable
|
||||||
|
]
|
||||||
|
image_areas = [image.printed_area for image in self._images if image.renderable]
|
||||||
|
total_drawn_area = sum(image_areas)
|
||||||
|
if total_drawn_area == 0:
|
||||||
|
return None
|
||||||
|
|
||||||
|
weights = [area / total_drawn_area for area in image_areas]
|
||||||
|
# Calculate harmonic mean of DPIs weighted by area
|
||||||
|
if sys.version_info >= (3, 10):
|
||||||
|
weighted_dpi = statistics.harmonic_mean(image_dpis, weights)
|
||||||
|
else:
|
||||||
|
weighted_dpi = sum(weights) / sum(
|
||||||
|
weight / dpi for weight, dpi in zip(weights, image_dpis)
|
||||||
|
)
|
||||||
|
max_dpi = max(image_dpis)
|
||||||
|
dpi_average_max_ratio = weighted_dpi / max_dpi
|
||||||
|
|
||||||
|
arg_max_dpi = image_dpis.index(max_dpi)
|
||||||
|
max_area_ratio = image_areas[arg_max_dpi] / total_drawn_area
|
||||||
|
|
||||||
|
return PageResolutionProfile(
|
||||||
|
weighted_dpi,
|
||||||
|
max_dpi,
|
||||||
|
dpi_average_max_ratio,
|
||||||
|
max_area_ratio,
|
||||||
|
)
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
|
"""Return string representation."""
|
||||||
return (
|
return (
|
||||||
f'<PageInfo '
|
f'<PageInfo '
|
||||||
f'pageno={self.pageno} {self.width_inches}"x{self.height_inches}" '
|
f'pageno={self.pageno} {self.width_inches}"x{self.height_inches}" '
|
||||||
@@ -921,7 +1028,11 @@ DEFAULT_EXECUTOR = SerialExecutor()
|
|||||||
|
|
||||||
|
|
||||||
class PdfInfo:
|
class PdfInfo:
|
||||||
"""Get summary information about a PDF"""
|
"""Get summary information about a PDF."""
|
||||||
|
|
||||||
|
_has_acroform: bool = False
|
||||||
|
_has_signature: bool = False
|
||||||
|
_needs_rendering: bool = False
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
@@ -929,10 +1040,11 @@ class PdfInfo:
|
|||||||
*,
|
*,
|
||||||
detailed_analysis: bool = False,
|
detailed_analysis: bool = False,
|
||||||
progbar: bool = False,
|
progbar: bool = False,
|
||||||
max_workers: int = None,
|
max_workers: int | None = None,
|
||||||
check_pages=None,
|
check_pages=None,
|
||||||
executor: Executor = DEFAULT_EXECUTOR,
|
executor: Executor = DEFAULT_EXECUTOR,
|
||||||
):
|
):
|
||||||
|
"""Initialize."""
|
||||||
self._infile = infile
|
self._infile = infile
|
||||||
if check_pages is None:
|
if check_pages is None:
|
||||||
check_pages = range(0, 1_000_000_000)
|
check_pages = range(0, 1_000_000_000)
|
||||||
@@ -949,52 +1061,71 @@ class PdfInfo:
|
|||||||
check_pages=check_pages,
|
check_pages=check_pages,
|
||||||
detailed_analysis=detailed_analysis,
|
detailed_analysis=detailed_analysis,
|
||||||
)
|
)
|
||||||
self._needs_rendering = pdf.Root.get('/NeedsRendering', False)
|
self._needs_rendering = pdf.Root.get(Name.NeedsRendering, False)
|
||||||
self._has_acroform = False
|
if Name.AcroForm in pdf.Root:
|
||||||
if '/AcroForm' in pdf.Root:
|
if len(pdf.Root.AcroForm.get(Name.Fields, [])) > 0:
|
||||||
if len(pdf.Root.AcroForm.get('/Fields', [])) > 0:
|
|
||||||
self._has_acroform = True
|
self._has_acroform = True
|
||||||
elif '/XFA' in pdf.Root.AcroForm:
|
elif Name.XFA in pdf.Root.AcroForm:
|
||||||
self._has_acroform = True
|
self._has_acroform = True
|
||||||
|
self._has_signature = bool(pdf.Root.AcroForm.get(Name.SigFlags, 0) & 1)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def pages(self) -> Sequence[PageInfo | None]:
|
def pages(self) -> Sequence[PageInfo | None]:
|
||||||
|
"""Return list of PageInfo objects, one per page in the PDF."""
|
||||||
return self._pages
|
return self._pages
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def min_version(self) -> str:
|
def min_version(self) -> str:
|
||||||
|
"""Return minimum PDF version needed to render this PDF."""
|
||||||
# The minimum PDF is the maximum version that any particular page needs
|
# The minimum PDF is the maximum version that any particular page needs
|
||||||
return max(page.min_version for page in self.pages if page)
|
return max(page.min_version for page in self.pages if page)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_userunit(self) -> bool:
|
def has_userunit(self) -> bool:
|
||||||
|
"""Return True if any page has a user unit."""
|
||||||
return any(page.userunit != 1.0 for page in self.pages if page)
|
return any(page.userunit != 1.0 for page in self.pages if page)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_acroform(self) -> bool:
|
def has_acroform(self) -> bool:
|
||||||
|
"""Return True if the document catalog has an AcroForm."""
|
||||||
return self._has_acroform
|
return self._has_acroform
|
||||||
|
|
||||||
|
@property
|
||||||
|
def has_signature(self) -> bool:
|
||||||
|
"""Return True if the document annotations has a digital signature."""
|
||||||
|
return self._has_signature
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def filename(self) -> str | Path:
|
def filename(self) -> str | Path:
|
||||||
|
"""Return filename of PDF."""
|
||||||
if not isinstance(self._infile, (str, Path)):
|
if not isinstance(self._infile, (str, Path)):
|
||||||
raise NotImplementedError("can't get filename from stream")
|
raise NotImplementedError("can't get filename from stream")
|
||||||
return self._infile
|
return self._infile
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def needs_rendering(self) -> bool:
|
def needs_rendering(self) -> bool:
|
||||||
|
"""Return True if PDF contains XFA forms.
|
||||||
|
|
||||||
|
XFA forms are not supported by most standard PDF renderers, so we
|
||||||
|
need to detect and suppress them.
|
||||||
|
"""
|
||||||
return self._needs_rendering
|
return self._needs_rendering
|
||||||
|
|
||||||
def __getitem__(self, item) -> PageInfo:
|
def __getitem__(self, item) -> PageInfo:
|
||||||
|
"""Return PageInfo object for page number `item`."""
|
||||||
return self._pages[item]
|
return self._pages[item]
|
||||||
|
|
||||||
def __len__(self):
|
def __len__(self):
|
||||||
|
"""Return number of pages in PDF."""
|
||||||
return len(self._pages)
|
return len(self._pages)
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
|
"""Return string representation."""
|
||||||
return f"<PdfInfo('...'), page count={len(self)}>"
|
return f"<PdfInfo('...'), page count={len(self)}>"
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
|
"""Run as a script."""
|
||||||
import argparse # pylint: disable=import-outside-toplevel
|
import argparse # pylint: disable=import-outside-toplevel
|
||||||
from pprint import pprint # pylint: disable=import-outside-toplevel
|
from pprint import pprint # pylint: disable=import-outside-toplevel
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
"""Detailed text position and layout analysis, building on pdfminer.six."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import re
|
import re
|
||||||
@@ -23,45 +25,68 @@ from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
|||||||
STRIP_NAME = re.compile(r'[0-9]+')
|
STRIP_NAME = re.compile(r'[0-9]+')
|
||||||
|
|
||||||
|
|
||||||
original_PDFSimpleFont_init = PDFSimpleFont.__init__
|
original_pdfsimplefont_init = PDFSimpleFont.__init__
|
||||||
|
|
||||||
|
|
||||||
def PDFSimpleFont__init__(self, descriptor, widths, spec):
|
def pdfsimplefont__init__(self, descriptor, widths, spec):
|
||||||
|
"""Monkeypatch pdfminer.six PDFSimpleFont.__init__.
|
||||||
|
|
||||||
|
If there is no ToUnicode and no Encoding, pdfminer.six assumes that Unicode
|
||||||
|
conversion is possible. This is incorrect, according to PDF Reference Manual
|
||||||
|
9.10.2. This patch fixes that.
|
||||||
|
"""
|
||||||
# Font encoding is specified either by a name of
|
# Font encoding is specified either by a name of
|
||||||
# built-in encoding or a dictionary that describes
|
# built-in encoding or a dictionary that describes
|
||||||
# the differences.
|
# the differences.
|
||||||
original_PDFSimpleFont_init(self, descriptor, widths, spec)
|
original_pdfsimplefont_init(self, descriptor, widths, spec)
|
||||||
# pdfminer is incorrect. If there is no ToUnicode and no Encoding, do not
|
|
||||||
# assume Unicode conversion is possible. RM 9.10.2
|
|
||||||
if not self.unicode_map and 'Encoding' not in spec:
|
if not self.unicode_map and 'Encoding' not in spec:
|
||||||
self.cid2unicode = {}
|
self.cid2unicode = {}
|
||||||
return
|
return
|
||||||
|
|
||||||
|
|
||||||
PDFSimpleFont.__init__ = PDFSimpleFont__init__
|
PDFSimpleFont.__init__ = pdfsimplefont__init__
|
||||||
|
|
||||||
#
|
#
|
||||||
# pdfminer patches when creator is PScript5.dll
|
# pdfminer patches when creator is PScript5.dll
|
||||||
#
|
#
|
||||||
|
|
||||||
|
|
||||||
def PDFType3Font__PScript5_get_height(self):
|
def pdftype3font__pscript5_get_height(self):
|
||||||
|
"""Monkeypatch for PScript5.dll PDFs.
|
||||||
|
|
||||||
|
The height of Type3 fonts is known to be incorrect in PScript5.dll
|
||||||
|
generated PDFs. This patch attempts to correct the height by
|
||||||
|
using the bbox height if it is available, otherwise using the
|
||||||
|
ascent and descent.
|
||||||
|
"""
|
||||||
h = self.bbox[3] - self.bbox[1]
|
h = self.bbox[3] - self.bbox[1]
|
||||||
if h == 0:
|
if h == 0:
|
||||||
h = self.ascent - self.descent
|
h = self.ascent - self.descent
|
||||||
return h * copysign(1.0, self.vscale)
|
return h * copysign(1.0, self.vscale)
|
||||||
|
|
||||||
|
|
||||||
def PDFType3Font__PScript5_get_descent(self):
|
def pdftype3font__pscript5_get_descent(self):
|
||||||
|
"""Monkeypatch for PScript5.dll PDFs.
|
||||||
|
|
||||||
|
The descent of Type3 fonts is known to be incorrect in PScript5.dll
|
||||||
|
generated PDFs. This patch attempts to correct the descent by
|
||||||
|
using the vscale.
|
||||||
|
"""
|
||||||
return self.descent * copysign(1.0, self.vscale)
|
return self.descent * copysign(1.0, self.vscale)
|
||||||
|
|
||||||
|
|
||||||
def PDFType3Font__PScript5_get_ascent(self):
|
def pdftype3font__pscript5_get_ascent(self):
|
||||||
|
"""Monkeypatch for PScript5.dll PDFs.
|
||||||
|
|
||||||
|
The ascent of Type3 fonts is known to be incorrect in PScript5.dll
|
||||||
|
generated PDFs. This patch attempts to correct the ascent by
|
||||||
|
using the vscale.
|
||||||
|
"""
|
||||||
return self.ascent * copysign(1.0, self.vscale)
|
return self.ascent * copysign(1.0, self.vscale)
|
||||||
|
|
||||||
|
|
||||||
class LTStateAwareChar(LTChar):
|
class LTStateAwareChar(LTChar):
|
||||||
"""A subclass of LTChar that tracks text render mode at time of drawing"""
|
"""A subclass of LTChar that tracks text render mode at time of drawing."""
|
||||||
|
|
||||||
__slots__ = (
|
__slots__ = (
|
||||||
'rendermode',
|
'rendermode',
|
||||||
@@ -94,6 +119,7 @@ class LTStateAwareChar(LTChar):
|
|||||||
graphicstate,
|
graphicstate,
|
||||||
textstate,
|
textstate,
|
||||||
):
|
):
|
||||||
|
"""Initialize."""
|
||||||
super().__init__(
|
super().__init__(
|
||||||
matrix,
|
matrix,
|
||||||
font,
|
font,
|
||||||
@@ -109,7 +135,7 @@ class LTStateAwareChar(LTChar):
|
|||||||
self.rendermode = textstate.render
|
self.rendermode = textstate.render
|
||||||
|
|
||||||
def is_compatible(self, obj):
|
def is_compatible(self, obj):
|
||||||
"""Check if characters can be combined into a textline
|
"""Check if characters can be combined into a textline.
|
||||||
|
|
||||||
We consider characters compatible if:
|
We consider characters compatible if:
|
||||||
- the Unicode mapping is known, and both have the same render mode
|
- the Unicode mapping is known, and both have the same render mode
|
||||||
@@ -127,36 +153,41 @@ class LTStateAwareChar(LTChar):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
def get_text(self):
|
def get_text(self):
|
||||||
|
"""Get text from this character."""
|
||||||
if isinstance(self._text, tuple):
|
if isinstance(self._text, tuple):
|
||||||
return '\ufffd' # standard 'Unknown symbol'
|
return '\ufffd' # standard 'Unknown symbol'
|
||||||
return self._text
|
return self._text
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
return '<{} {} matrix={} rendermode={!r} font={!r} adv={} text={!r}>'.format(
|
"""Return a string representation of this object."""
|
||||||
self.__class__.__name__,
|
return (
|
||||||
bbox2str(self.bbox),
|
f"<{self.__class__.__name__} "
|
||||||
matrix2str(self.matrix),
|
f"{bbox2str(self.bbox)} "
|
||||||
self.rendermode,
|
f"matrix={matrix2str(self.matrix)} "
|
||||||
self.fontname,
|
f"rendermode={self.rendermode!r} "
|
||||||
self.adv,
|
f"font={self.fontname!r} "
|
||||||
self.get_text(),
|
f"adv={self.adv} "
|
||||||
|
f"text={self.get_text()!r}>"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
class TextPositionTracker(PDFLayoutAnalyzer):
|
class TextPositionTracker(PDFLayoutAnalyzer):
|
||||||
"""A page layout analyzer that pays attention to text visibility"""
|
"""A page layout analyzer that pays attention to text visibility."""
|
||||||
|
|
||||||
def __init__(self, rsrcmgr, pageno=1, laparams=None):
|
def __init__(self, rsrcmgr, pageno=1, laparams=None):
|
||||||
|
"""Initialize the layout analyzer."""
|
||||||
super().__init__(rsrcmgr, pageno, laparams)
|
super().__init__(rsrcmgr, pageno, laparams)
|
||||||
self.textstate = None
|
self.textstate = None
|
||||||
self.result = None
|
self.result = None
|
||||||
self.cur_item = None # not defined in pdfminer code as it should be
|
self.cur_item = None # not defined in pdfminer code as it should be
|
||||||
|
|
||||||
def begin_page(self, page, ctm):
|
def begin_page(self, page, ctm):
|
||||||
|
"""Begin processing of a page."""
|
||||||
super().begin_page(page, ctm)
|
super().begin_page(page, ctm)
|
||||||
self.cur_item = LTPage(self.pageno, page.mediabox)
|
self.cur_item = LTPage(self.pageno, page.mediabox)
|
||||||
|
|
||||||
def end_page(self, page):
|
def end_page(self, page):
|
||||||
|
"""End processing of a page."""
|
||||||
assert not self._stack, str(len(self._stack))
|
assert not self._stack, str(len(self._stack))
|
||||||
assert isinstance(self.cur_item, LTPage), str(type(self.cur_item))
|
assert isinstance(self.cur_item, LTPage), str(type(self.cur_item))
|
||||||
if self.laparams is not None:
|
if self.laparams is not None:
|
||||||
@@ -165,12 +196,14 @@ class TextPositionTracker(PDFLayoutAnalyzer):
|
|||||||
self.receive_layout(self.cur_item)
|
self.receive_layout(self.cur_item)
|
||||||
|
|
||||||
def render_string(self, textstate, seq, ncs, graphicstate):
|
def render_string(self, textstate, seq, ncs, graphicstate):
|
||||||
|
"""Respond to render string event by updating text state."""
|
||||||
self.textstate = textstate.copy()
|
self.textstate = textstate.copy()
|
||||||
super().render_string(self.textstate, seq, ncs, graphicstate)
|
super().render_string(self.textstate, seq, ncs, graphicstate)
|
||||||
|
|
||||||
def render_char(
|
def render_char(
|
||||||
self, matrix, font, fontsize, scaling, rise, cid, ncs, graphicstate
|
self, matrix, font, fontsize, scaling, rise, cid, ncs, graphicstate
|
||||||
):
|
):
|
||||||
|
"""Respond to render char event by updating text state."""
|
||||||
try:
|
try:
|
||||||
text = font.to_unichr(cid)
|
text = font.to_unichr(cid)
|
||||||
assert isinstance(text, str), str(type(text))
|
assert isinstance(text, str), str(type(text))
|
||||||
@@ -195,23 +228,23 @@ class TextPositionTracker(PDFLayoutAnalyzer):
|
|||||||
return item.adv
|
return item.adv
|
||||||
|
|
||||||
def handle_undefined_char(self, font, cid):
|
def handle_undefined_char(self, font, cid):
|
||||||
|
"""Handle undefined character."""
|
||||||
# log.info('undefined: %r, %r', font, cid)
|
# log.info('undefined: %r, %r', font, cid)
|
||||||
return (font.fontname, cid)
|
return (font.fontname, cid)
|
||||||
|
|
||||||
def receive_layout(self, ltpage):
|
def receive_layout(self, ltpage):
|
||||||
|
"""Receive layout handler."""
|
||||||
self.result = ltpage
|
self.result = ltpage
|
||||||
|
|
||||||
def get_result(self):
|
def get_result(self):
|
||||||
|
"""Get the result of the analysis."""
|
||||||
return self.result
|
return self.result
|
||||||
|
|
||||||
|
|
||||||
def get_page_analysis(infile, pageno, pscript5_mode):
|
def get_page_analysis(infile, pageno, pscript5_mode):
|
||||||
|
"""Get the page analysis for a given page."""
|
||||||
rman = pdfminer.pdfinterp.PDFResourceManager(caching=True)
|
rman = pdfminer.pdfinterp.PDFResourceManager(caching=True)
|
||||||
if pdfminer.__version__ < '20200402':
|
disable_boxes_flow = None
|
||||||
# Workaround for https://github.com/pdfminer/pdfminer.six/issues/395
|
|
||||||
disable_boxes_flow = 2
|
|
||||||
else:
|
|
||||||
disable_boxes_flow = None
|
|
||||||
dev = TextPositionTracker(
|
dev = TextPositionTracker(
|
||||||
rman,
|
rman,
|
||||||
laparams=LAParams(
|
laparams=LAParams(
|
||||||
@@ -225,9 +258,9 @@ def get_page_analysis(infile, pageno, pscript5_mode):
|
|||||||
patcher = patch.multiple(
|
patcher = patch.multiple(
|
||||||
'pdfminer.pdffont.PDFType3Font',
|
'pdfminer.pdffont.PDFType3Font',
|
||||||
spec=True,
|
spec=True,
|
||||||
get_ascent=PDFType3Font__PScript5_get_ascent,
|
get_ascent=pdftype3font__pscript5_get_ascent,
|
||||||
get_descent=PDFType3Font__PScript5_get_descent,
|
get_descent=pdftype3font__pscript5_get_descent,
|
||||||
get_height=PDFType3Font__PScript5_get_height,
|
get_height=pdftype3font__pscript5_get_height,
|
||||||
)
|
)
|
||||||
patcher.start()
|
patcher.start()
|
||||||
|
|
||||||
@@ -250,6 +283,7 @@ def get_page_analysis(infile, pageno, pscript5_mode):
|
|||||||
|
|
||||||
|
|
||||||
def get_text_boxes(obj):
|
def get_text_boxes(obj):
|
||||||
|
"""Get the text boxes attached to the current node."""
|
||||||
for child in obj:
|
for child in obj:
|
||||||
if isinstance(child, (LTTextBox)):
|
if isinstance(child, (LTTextBox)):
|
||||||
yield child
|
yield child
|
||||||
|
|||||||
+67
-31
@@ -28,6 +28,7 @@ if TYPE_CHECKING:
|
|||||||
hookspec = pluggy.HookspecMarker('ocrmypdf')
|
hookspec = pluggy.HookspecMarker('ocrmypdf')
|
||||||
|
|
||||||
# pylint: disable=unused-argument
|
# pylint: disable=unused-argument
|
||||||
|
# mypy: disable-error-code=empty-body
|
||||||
|
|
||||||
|
|
||||||
@hookspec(firstresult=True)
|
@hookspec(firstresult=True)
|
||||||
@@ -43,7 +44,7 @@ def get_logging_console() -> Handler:
|
|||||||
|
|
||||||
|
|
||||||
@hookspec
|
@hookspec
|
||||||
def initialize(plugin_manager: pluggy.PluginManager):
|
def initialize(plugin_manager: pluggy.PluginManager) -> None:
|
||||||
"""Called when this plugin is first loaded into OCRmyPDF.
|
"""Called when this plugin is first loaded into OCRmyPDF.
|
||||||
|
|
||||||
The primary intended use of this is for plugins to check compatibility with other
|
The primary intended use of this is for plugins to check compatibility with other
|
||||||
@@ -99,6 +100,8 @@ def check_options(options: Namespace) -> None:
|
|||||||
ocrmypdf.exceptions.ExitCodeException: If options are not acceptable
|
ocrmypdf.exceptions.ExitCodeException: If options are not acceptable
|
||||||
and the application should terminate gracefully with an informative
|
and the application should terminate gracefully with an informative
|
||||||
message and error code.
|
message and error code.
|
||||||
|
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This hook will be called from the main process, and may modify global state
|
This hook will be called from the main process, and may modify global state
|
||||||
before child worker processes are forked.
|
before child worker processes are forked.
|
||||||
@@ -127,6 +130,8 @@ def get_executor(progressbar_class) -> Executor:
|
|||||||
Note:
|
Note:
|
||||||
This hook will be called from the main process, and may modify global state
|
This hook will be called from the main process, and may modify global state
|
||||||
before child worker processes are forked.
|
before child worker processes are forked.
|
||||||
|
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This is a :ref:`firstresult hook<firstresult>`.
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
"""
|
"""
|
||||||
@@ -159,7 +164,6 @@ def get_progressbar_class():
|
|||||||
Here is how OCRmyPDF will use the progress bar:
|
Here is how OCRmyPDF will use the progress bar:
|
||||||
|
|
||||||
Example:
|
Example:
|
||||||
|
|
||||||
pbar_class = pm.hook.get_progressbar_class()
|
pbar_class = pm.hook.get_progressbar_class()
|
||||||
with pbar_class(**tqdm_kwargs) as pbar:
|
with pbar_class(**tqdm_kwargs) as pbar:
|
||||||
...
|
...
|
||||||
@@ -181,6 +185,8 @@ def validate(pdfinfo: PdfInfo, options: Namespace) -> None:
|
|||||||
ocrmypdf.exceptions.ExitCodeException: If options or pdfinfo are not acceptable
|
ocrmypdf.exceptions.ExitCodeException: If options or pdfinfo are not acceptable
|
||||||
and the application should terminate gracefully with an informative
|
and the application should terminate gracefully with an informative
|
||||||
message and error code.
|
message and error code.
|
||||||
|
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This hook will be called from the main process, and may modify global state
|
This hook will be called from the main process, and may modify global state
|
||||||
before child worker processes are forked.
|
before child worker processes are forked.
|
||||||
@@ -197,6 +203,7 @@ def rasterize_pdf_page(
|
|||||||
page_dpi: Resolution | None,
|
page_dpi: Resolution | None,
|
||||||
rotation: int | None,
|
rotation: int | None,
|
||||||
filter_vector: bool,
|
filter_vector: bool,
|
||||||
|
stop_on_soft_error: bool,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units.
|
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units.
|
||||||
|
|
||||||
@@ -207,17 +214,26 @@ def rasterize_pdf_page(
|
|||||||
Args:
|
Args:
|
||||||
input_file: The PDF to rasterize.
|
input_file: The PDF to rasterize.
|
||||||
output_file: The desired name of the rasterized image.
|
output_file: The desired name of the rasterized image.
|
||||||
raster_device: Type of image to produce at output_file
|
raster_device: Type of image to produce at output_file.
|
||||||
raster_dpi: Resolution at which to rasterize page
|
raster_dpi: Resolution in dots per inch at which to rasterize page.
|
||||||
pageno: Page number to rasterize (beginning at page 1)
|
pageno: Page number to rasterize (beginning at page 1).
|
||||||
page_dpi: Resolution, overriding output image DPI
|
page_dpi: Resolution, overriding output image DPI.
|
||||||
rotation: Cardinal angle, clockwise, to rotate page
|
rotation: Cardinal angle, clockwise, to rotate page.
|
||||||
filter_vector: If True, remove vector graphics objects
|
filter_vector: If True, remove vector graphics objects.
|
||||||
|
stop_on_soft_error: If there is an "soft error" such that PDF page image
|
||||||
|
generation can proceed, but may visually differ from the original,
|
||||||
|
the implementer of this hook should raise a detailed exception. If
|
||||||
|
``False``, continue processing and report by logging it. If the hook
|
||||||
|
cannot proceed, it should always raise an exception, regardless of
|
||||||
|
this setting. One "soft error" would be a missing font that is
|
||||||
|
required to properly rasterize the PDF.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Path: output_file if successful
|
Path: output_file if successful
|
||||||
Note:
|
Note:
|
||||||
This hook will be called from child processes. Modifying global state
|
This hook will be called from child processes. Modifying global state
|
||||||
will not affect the main process or other child processes.
|
will not affect the main process or other child processes.
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This is a :ref:`firstresult hook<firstresult>`.
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
"""
|
"""
|
||||||
@@ -228,23 +244,32 @@ def filter_ocr_image(page: PageContext, image: Image.Image) -> Image.Image:
|
|||||||
"""Called to filter the image before it is sent to OCR.
|
"""Called to filter the image before it is sent to OCR.
|
||||||
|
|
||||||
This is the image that OCR sees, not what the user sees when they view the
|
This is the image that OCR sees, not what the user sees when they view the
|
||||||
PDF. If ``redo_ocr`` is enabled, portions of the image will be masked so
|
PDF. In certain modes such as ``--redo-ocr``, portions of the image may be
|
||||||
they are not shown to OCR. The main use of this hook is expected to be hiding
|
masked out to hide them from OCR.
|
||||||
content from OCR.
|
|
||||||
|
The main uses of this hook are expected to be hiding content from OCR,
|
||||||
|
conditioning images to OCR better with filters, and adjusting images to
|
||||||
|
match any constraints imposed by the OCR engine.
|
||||||
|
|
||||||
The input image may be color, grayscale, or monochrome, and the
|
The input image may be color, grayscale, or monochrome, and the
|
||||||
output image may differ. The pixel width and height of the
|
output image may differ. For example, if you know that a custom OCR engine
|
||||||
output image must be identical to the input image, or misalignment between
|
does not care about the color of the text, you could convert the image to
|
||||||
the OCR text layer and visual position of the text will occur. Likewise,
|
it to grayscale or monochrome.
|
||||||
the output must be a faithful representation of the input, or alignment
|
|
||||||
errors may occurs.
|
|
||||||
|
|
||||||
Tesseract OCR only deals with monochrome images, and internally converts
|
Generally speaking, the output image should be a faithful representation of
|
||||||
non-monochrome images to OCR.
|
of the input image. You *may* change the pixel width and height of the
|
||||||
|
the input image, but you must not change the aspect ratio, and you must
|
||||||
|
calculate the DPI of the output image based on the new pixel width and
|
||||||
|
height or the OCR text layer will be misaligned with the visual position.
|
||||||
|
|
||||||
|
The built-in Tesseract OCR engine uses this hook itself to downsample
|
||||||
|
very large images to fit its constraints.
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This hook will be called from child processes. Modifying global state
|
This hook will be called from child processes. Modifying global state
|
||||||
will not affect the main process or other child processes.
|
will not affect the main process or other child processes.
|
||||||
|
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This is a :ref:`firstresult hook<firstresult>`.
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
"""
|
"""
|
||||||
@@ -269,7 +294,7 @@ def filter_page_image(page: PageContext, image_filename: Path) -> Path:
|
|||||||
to enforce these constraints; it is up to the plugin to do sensible things.
|
to enforce these constraints; it is up to the plugin to do sensible things.
|
||||||
|
|
||||||
OCRmyPDF will create the PDF page based on the image format used (unless the
|
OCRmyPDF will create the PDF page based on the image format used (unless the
|
||||||
hook is overriden). If you convert the image to a JPEG, the output page will
|
hook is overridden). If you convert the image to a JPEG, the output page will
|
||||||
be created as a JPEG, etc. If you change the colorspace, that change will be
|
be created as a JPEG, etc. If you change the colorspace, that change will be
|
||||||
kept. Note that the OCRmyPDF image optimization stage, if enabled, may
|
kept. Note that the OCRmyPDF image optimization stage, if enabled, may
|
||||||
ultimately chose a different format.
|
ultimately chose a different format.
|
||||||
@@ -278,13 +303,11 @@ def filter_page_image(page: PageContext, image_filename: Path) -> Path:
|
|||||||
will occur. The return value should be a path to a file in the same folder
|
will occur. The return value should be a path to a file in the same folder
|
||||||
as ``image_filename``.
|
as ``image_filename``.
|
||||||
|
|
||||||
Implementation detail: If the value returned is falsy, OCRmyPDF will ignore
|
|
||||||
the return value and assume the input file was unmodified. This is deprecated.
|
|
||||||
To leave the image unmodified, ``image_filename`` should be returned.
|
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This hook will be called from child processes. Modifying global state
|
This hook will be called from child processes. Modifying global state
|
||||||
will not affect the main process or other child processes.
|
will not affect the main process or other child processes.
|
||||||
|
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This is a :ref:`firstresult hook<firstresult>`.
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
"""
|
"""
|
||||||
@@ -327,6 +350,7 @@ def filter_pdf_page(page: PageContext, image_filename: Path, output_pdf: Path) -
|
|||||||
Note:
|
Note:
|
||||||
This hook will be called from child processes. Modifying global state
|
This hook will be called from child processes. Modifying global state
|
||||||
will not affect the main process or other child processes.
|
will not affect the main process or other child processes.
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This is a :ref:`firstresult hook<firstresult>`.
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
"""
|
"""
|
||||||
@@ -385,7 +409,8 @@ class OcrEngine(ABC):
|
|||||||
"""Returns the set of all languages that are supported by the engine.
|
"""Returns the set of all languages that are supported by the engine.
|
||||||
|
|
||||||
Languages are typically given in 3-letter ISO 3166-1 codes, but actually
|
Languages are typically given in 3-letter ISO 3166-1 codes, but actually
|
||||||
can be any value understood by the OCR engine."""
|
can be any value understood by the OCR engine.
|
||||||
|
"""
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
@@ -417,6 +442,9 @@ class OcrEngine(ABC):
|
|||||||
a single page PDF with no visible content of any kind, sized
|
a single page PDF with no visible content of any kind, sized
|
||||||
to the dimensions implied by the input_file's width, height
|
to the dimensions implied by the input_file's width, height
|
||||||
and DPI. The image will be grafted onto the input PDF page.
|
and DPI. The image will be grafted onto the input PDF page.
|
||||||
|
output_text: The expected name of a text file containing the
|
||||||
|
recognized text.
|
||||||
|
options: The command line options.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
@@ -437,10 +465,11 @@ def generate_pdfa(
|
|||||||
pdf_pages: list[Path],
|
pdf_pages: list[Path],
|
||||||
pdfmark: Path,
|
pdfmark: Path,
|
||||||
output_file: Path,
|
output_file: Path,
|
||||||
compression: str,
|
context: PdfContext,
|
||||||
pdf_version: str,
|
pdf_version: str,
|
||||||
pdfa_part: str,
|
pdfa_part: str,
|
||||||
progressbar_class,
|
progressbar_class,
|
||||||
|
stop_on_soft_error: bool,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
"""Generate a PDF/A.
|
"""Generate a PDF/A.
|
||||||
|
|
||||||
@@ -454,11 +483,7 @@ def generate_pdfa(
|
|||||||
pdfmark: A PostScript file intended for Ghostscript with details on
|
pdfmark: A PostScript file intended for Ghostscript with details on
|
||||||
how to perform the PDF/A conversion.
|
how to perform the PDF/A conversion.
|
||||||
output_file: The name of the desired output file.
|
output_file: The name of the desired output file.
|
||||||
compression: One of ``'jpeg'``, ``'lossless'``, ``''``. For ``'jpeg'``,
|
context: The current context.
|
||||||
the PDF/A generator should convert all images to JPEG encoding where
|
|
||||||
possible. For lossless, all images should be converted to FlateEncode
|
|
||||||
(lossless PNG). If an empty string, the PDF generator should make its
|
|
||||||
own decisions about how to encode images.
|
|
||||||
pdf_version: The minimum PDF version that the output file should be.
|
pdf_version: The minimum PDF version that the output file should be.
|
||||||
At its own discretion, the PDF/A generator may raise the version,
|
At its own discretion, the PDF/A generator may raise the version,
|
||||||
but should not lower it.
|
but should not lower it.
|
||||||
@@ -471,6 +496,12 @@ def generate_pdfa(
|
|||||||
and the name of the work units ("page"). Then ``instance.update()``
|
and the name of the work units ("page"). Then ``instance.update()``
|
||||||
will be called when a work unit is completed. If ``None``, no
|
will be called when a work unit is completed. If ``None``, no
|
||||||
progress information is reported.
|
progress information is reported.
|
||||||
|
stop_on_soft_error: If there is an "soft error" such that PDF/A generation
|
||||||
|
can proceed and produce a valid PDF/A, but output may be invalid or
|
||||||
|
may not visually resemble the original, the implementer of this hook
|
||||||
|
should raise a detailed exception. If ``False``, continue processing
|
||||||
|
and report by logging it. If the hook cannot proceed, it should always
|
||||||
|
raise an exception, regardless of this setting.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Path: If successful, the hook should return ``output_file``.
|
Path: If successful, the hook should return ``output_file``.
|
||||||
@@ -478,7 +509,12 @@ def generate_pdfa(
|
|||||||
Note:
|
Note:
|
||||||
This is a :ref:`firstresult hook<firstresult>`.
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
|
|
||||||
See also:
|
Note:
|
||||||
|
Before version 15.0.0, the ``context`` was not provided and ``compression``
|
||||||
|
was provided instead. Plugins should now read the context object to determine
|
||||||
|
if compression is requested.
|
||||||
|
|
||||||
|
See Also:
|
||||||
https://github.com/tqdm/tqdm
|
https://github.com/tqdm/tqdm
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
"""Utilities to measure OCR quality"""
|
"""Utilities to measure OCR quality."""
|
||||||
|
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
"""Wrappers to manage subprocess calls"""
|
"""Wrappers to manage subprocess calls."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -9,7 +9,6 @@ import os
|
|||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from functools import lru_cache
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||||
from subprocess import run as subprocess_run
|
from subprocess import run as subprocess_run
|
||||||
@@ -35,20 +34,25 @@ def run(
|
|||||||
check: bool = False,
|
check: bool = False,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
) -> CompletedProcess:
|
) -> CompletedProcess:
|
||||||
"""Wrapper around :py:func:`subprocess.run`
|
"""Wrapper around :py:func:`subprocess.run`.
|
||||||
|
|
||||||
The main purpose of this wrapper is to log subprocess output in an orderly
|
The main purpose of this wrapper is to log subprocess output in an orderly
|
||||||
fashion that indentifies the responsible subprocess. An additional
|
fashion that identifies the responsible subprocess. An additional
|
||||||
task is that this function goes to greater lengths to find possible Windows
|
task is that this function goes to greater lengths to find possible Windows
|
||||||
locations of our dependencies when they are not on the system PATH.
|
locations of our dependencies when they are not on the system PATH.
|
||||||
|
|
||||||
Arguments should be identical to ``subprocess.run``, except for following:
|
Arguments should be identical to ``subprocess.run``, except for following:
|
||||||
|
|
||||||
Arguments:
|
Args:
|
||||||
|
args: Positional arguments to pass to ``subprocess.run``.
|
||||||
|
env: A set of environment variables. If None, the OS environment is used.
|
||||||
logs_errors_to_stdout: If True, indicates that the process writes its error
|
logs_errors_to_stdout: If True, indicates that the process writes its error
|
||||||
messages to stdout rather than stderr, so stdout should be logged
|
messages to stdout rather than stderr, so stdout should be logged
|
||||||
if there is an error. If False, stderr is logged. Could be used with
|
if there is an error. If False, stderr is logged. Could be used with
|
||||||
stderr=STDOUT, stdout=PIPE for example.
|
stderr=STDOUT, stdout=PIPE for example.
|
||||||
|
check: If True, raise an exception if the process exits with a non-zero
|
||||||
|
status code. If False, the return value will indicate success or failure.
|
||||||
|
kwargs: Additional arguments to pass to ``subprocess.run``.
|
||||||
"""
|
"""
|
||||||
args, env, process_log, _text = _fix_process_args(args, env, kwargs)
|
args, env, process_log, _text = _fix_process_args(args, env, kwargs)
|
||||||
|
|
||||||
@@ -114,8 +118,6 @@ def run_polling_stderr(
|
|||||||
def _fix_process_args(
|
def _fix_process_args(
|
||||||
args: Args, env: OsEnviron | None, kwargs
|
args: Args, env: OsEnviron | None, kwargs
|
||||||
) -> tuple[Args, OsEnviron, logging.Logger, bool]:
|
) -> tuple[Args, OsEnviron, logging.Logger, bool]:
|
||||||
assert 'universal_newlines' not in kwargs, "Use text= instead of universal_newlines"
|
|
||||||
|
|
||||||
if not env:
|
if not env:
|
||||||
env = os.environ
|
env = os.environ
|
||||||
|
|
||||||
@@ -135,7 +137,6 @@ def _fix_process_args(
|
|||||||
return args, env, process_log, text
|
return args, env, process_log, text
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=None)
|
|
||||||
def get_version(
|
def get_version(
|
||||||
program: str,
|
program: str,
|
||||||
*,
|
*,
|
||||||
@@ -143,7 +144,7 @@ def get_version(
|
|||||||
regex=r'(\d+(\.\d+)*)',
|
regex=r'(\d+(\.\d+)*)',
|
||||||
env: OsEnviron | None = None,
|
env: OsEnviron | None = None,
|
||||||
) -> str:
|
) -> str:
|
||||||
"""Get the version of the specified program
|
"""Get the version of the specified program.
|
||||||
|
|
||||||
Arguments:
|
Arguments:
|
||||||
program: The program to version check.
|
program: The program to version check.
|
||||||
@@ -291,21 +292,12 @@ def _error_old_version(
|
|||||||
_error_trailer(**locals())
|
_error_trailer(**locals())
|
||||||
|
|
||||||
|
|
||||||
def _remove_leading_v(s: str) -> str:
|
|
||||||
if sys.version_info >= (3, 9):
|
|
||||||
return s.removeprefix('v')
|
|
||||||
|
|
||||||
if s.startswith('v'):
|
|
||||||
return s[1:]
|
|
||||||
return s
|
|
||||||
|
|
||||||
|
|
||||||
def check_external_program(
|
def check_external_program(
|
||||||
*,
|
*,
|
||||||
program: str,
|
program: str,
|
||||||
package: str,
|
package: str,
|
||||||
version_checker: Callable[[], str],
|
version_checker: Callable[[], Version],
|
||||||
need_version: str,
|
need_version: str | Version,
|
||||||
required_for: str | None = None,
|
required_for: str | None = None,
|
||||||
recommended: bool = False,
|
recommended: bool = False,
|
||||||
version_parser: type[Version] = Version,
|
version_parser: type[Version] = Version,
|
||||||
@@ -316,7 +308,7 @@ def check_external_program(
|
|||||||
program: The name of the program to test.
|
program: The name of the program to test.
|
||||||
package: The name of a software package that typically supplies this program.
|
package: The name of a software package that typically supplies this program.
|
||||||
Usually the same as program.
|
Usually the same as program.
|
||||||
version_check: A callable without arguments that retrieves the installed
|
version_checker: A callable without arguments that retrieves the installed
|
||||||
version of program.
|
version of program.
|
||||||
need_version: The minimum required version.
|
need_version: The minimum required version.
|
||||||
required_for: The name of an argument of feature that requires this program.
|
required_for: The name of an argument of feature that requires this program.
|
||||||
@@ -325,12 +317,10 @@ def check_external_program(
|
|||||||
version_parser: A class that should be used to parse and compare version
|
version_parser: A class that should be used to parse and compare version
|
||||||
numbers. Used when version numbers do not follow standard conventions.
|
numbers. Used when version numbers do not follow standard conventions.
|
||||||
"""
|
"""
|
||||||
|
if not isinstance(need_version, Version):
|
||||||
|
need_version = version_parser(need_version)
|
||||||
try:
|
try:
|
||||||
if callable(version_checker):
|
found_version = version_checker()
|
||||||
found_version = version_checker()
|
|
||||||
else: # deprecated
|
|
||||||
found_version = version_checker
|
|
||||||
except (CalledProcessError, FileNotFoundError) as e:
|
except (CalledProcessError, FileNotFoundError) as e:
|
||||||
_error_missing_program(program, package, required_for, recommended)
|
_error_missing_program(program, package, required_for, recommended)
|
||||||
if not recommended:
|
if not recommended:
|
||||||
@@ -342,11 +332,10 @@ def check_external_program(
|
|||||||
raise
|
raise
|
||||||
return
|
return
|
||||||
|
|
||||||
found_version = _remove_leading_v(found_version)
|
if found_version and found_version < need_version:
|
||||||
need_version = _remove_leading_v(need_version)
|
_error_old_version(
|
||||||
|
program, package, str(need_version), str(found_version), required_for
|
||||||
if found_version and version_parser(found_version) < version_parser(need_version):
|
)
|
||||||
_error_old_version(program, package, need_version, found_version, required_for)
|
|
||||||
if not recommended:
|
if not recommended:
|
||||||
raise MissingDependencyError(program)
|
raise MissingDependencyError(program)
|
||||||
|
|
||||||
|
|||||||
@@ -6,12 +6,15 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
from itertools import chain
|
from itertools import chain
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, Iterable, Iterator, TypeVar
|
from typing import Any, Callable, Iterable, Iterator, TypeVar
|
||||||
|
|
||||||
|
from packaging.version import InvalidVersion, Version
|
||||||
|
|
||||||
if sys.version_info >= (3, 10):
|
if sys.version_info >= (3, 10):
|
||||||
from typing import TypeAlias
|
from typing import TypeAlias
|
||||||
else:
|
else:
|
||||||
@@ -29,12 +32,13 @@ else:
|
|||||||
spec=['HKEYType', 'EnumKey', 'EnumValue', 'HKEY_LOCAL_MACHINE', 'OpenKey']
|
spec=['HKEYType', 'EnumKey', 'EnumValue', 'HKEY_LOCAL_MACHINE', 'OpenKey']
|
||||||
)
|
)
|
||||||
# mypy does not understand winreg.HKeyType where winreg is a Mock (fair enough!)
|
# mypy does not understand winreg.HKeyType where winreg is a Mock (fair enough!)
|
||||||
HKEYType: TypeAlias = Any
|
HKEYType: TypeAlias = Any # type: ignore
|
||||||
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
T = TypeVar('T')
|
T = TypeVar('T')
|
||||||
|
Tkey = TypeVar('Tkey')
|
||||||
|
|
||||||
|
|
||||||
def ghostscript_version_key(s: str) -> tuple[int, int, int]:
|
def ghostscript_version_key(s: str) -> tuple[int, int, int]:
|
||||||
@@ -99,6 +103,30 @@ def registry_path_tesseract(env=None) -> Iterator[Path]:
|
|||||||
log.warning(e)
|
log.warning(e)
|
||||||
|
|
||||||
|
|
||||||
|
def _gs_version_in_path_key(path: Path) -> tuple[str, Version | None]:
|
||||||
|
"""Key function for comparing Ghostscript and Tesseract paths.
|
||||||
|
|
||||||
|
Ghostscript installs on Windows:
|
||||||
|
%PROGRAMFILES%/gs/gs9.56.1/bin -> ('gs', Version('9.56.1'))
|
||||||
|
%PROGRAMFILES%/gs/9.24/bin -> ('gs', Version('9.24'))
|
||||||
|
|
||||||
|
Tesseract looks like:
|
||||||
|
%PROGRAMFILES%/Tesseract-OCR -> ('Tesseract-OCR', None)
|
||||||
|
|
||||||
|
Thus ensuring the resulting tuple will order the alternatives correctly,
|
||||||
|
e.g. gs10.0 > gs9.99.
|
||||||
|
"""
|
||||||
|
match = re.search(r'gs[/\\]?([0-9.]+)[/\\]bin', str(path))
|
||||||
|
if match:
|
||||||
|
try:
|
||||||
|
version_str = match.group(1)
|
||||||
|
version = Version(version_str)
|
||||||
|
return 'gs', version
|
||||||
|
except InvalidVersion:
|
||||||
|
pass
|
||||||
|
return path.name, None
|
||||||
|
|
||||||
|
|
||||||
def program_files_paths(env=None) -> Iterator[Path]:
|
def program_files_paths(env=None) -> Iterator[Path]:
|
||||||
if not env:
|
if not env:
|
||||||
env = os.environ
|
env = os.environ
|
||||||
@@ -116,7 +144,7 @@ def program_files_paths(env=None) -> Iterator[Path]:
|
|||||||
return iter(
|
return iter(
|
||||||
sorted(
|
sorted(
|
||||||
(p for p in path_walker()),
|
(p for p in path_walker()),
|
||||||
key=lambda p: (p.name, p.parent.name),
|
key=_gs_version_in_path_key,
|
||||||
reverse=True,
|
reverse=True,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
@@ -141,13 +169,7 @@ SHIMS = [
|
|||||||
|
|
||||||
|
|
||||||
def fix_windows_args(program: str, args, env):
|
def fix_windows_args(program: str, args, env):
|
||||||
"""Adjust our desired program and command line arguments for use on Windows"""
|
"""Adjust our desired program and command line arguments for use on Windows."""
|
||||||
|
|
||||||
if sys.version_info < (3, 8):
|
|
||||||
# bpo-33617 - Windows needs manual Path -> str conversion
|
|
||||||
args = [os.fspath(arg) for arg in args]
|
|
||||||
program = os.fspath(program)
|
|
||||||
|
|
||||||
# If we are running a .py on Windows, ensure we call it with this Python
|
# If we are running a .py on Windows, ensure we call it with this Python
|
||||||
# (to support test suite shims)
|
# (to support test suite shims)
|
||||||
if program.lower().endswith('.py'):
|
if program.lower().endswith('.py'):
|
||||||
@@ -164,11 +186,11 @@ def fix_windows_args(program: str, args, env):
|
|||||||
return args
|
return args
|
||||||
|
|
||||||
|
|
||||||
def unique_everseen(iterable: Iterable[T], key: Callable[[T], T]) -> Iterator[T]:
|
def unique_everseen(iterable: Iterable[T], key: Callable[[T], Tkey]) -> Iterator[T]:
|
||||||
"List unique elements, preserving order."
|
"""List unique elements, preserving order."""
|
||||||
# unique_everseen('AAAABBBCCDAABBB') --> A B C D
|
# unique_everseen('AAAABBBCCDAABBB') --> A B C D
|
||||||
# unique_everseen('ABBCcAD', str.lower) --> A B C D
|
# unique_everseen('ABBCcAD', str.lower) --> A B C D
|
||||||
seen: set[T] = set()
|
seen: set[Tkey] = set()
|
||||||
seen_add = seen.add
|
seen_add = seen.add
|
||||||
for element in iterable:
|
for element in iterable:
|
||||||
k = key(element)
|
k = key(element)
|
||||||
@@ -177,11 +199,15 @@ def unique_everseen(iterable: Iterable[T], key: Callable[[T], T]) -> Iterator[T]
|
|||||||
yield element
|
yield element
|
||||||
|
|
||||||
|
|
||||||
|
def _casefold_path(path: Path) -> str:
|
||||||
|
return str.casefold(str(path))
|
||||||
|
|
||||||
|
|
||||||
def shim_env_path(env=None):
|
def shim_env_path(env=None):
|
||||||
if env is None:
|
if env is None:
|
||||||
env = os.environ
|
env = os.environ
|
||||||
|
|
||||||
shim_paths = chain.from_iterable(shim(env) for shim in SHIMS)
|
shim_paths = chain.from_iterable(shim(env) for shim in SHIMS)
|
||||||
return os.pathsep.join(
|
return os.pathsep.join(
|
||||||
str(p) for p in unique_everseen(shim_paths, key=lambda p: str.casefold(str(p)))
|
str(p) for p in unique_everseen(shim_paths, key=_casefold_path)
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -1,4 +1,6 @@
|
|||||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
"""Tests."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|||||||
BIN
Binary file not shown.
-1
@@ -1 +0,0 @@
|
|||||||
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
|
||||||
|
|||||||
+1
-4
@@ -1,7 +1,5 @@
|
|||||||
i a la Waterman
|
i a la Waterman
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
4 ons linzen
|
4 ons linzen
|
||||||
|
|
||||||
3 liter water
|
3 liter water
|
||||||
@@ -17,11 +15,10 @@ laurier, kruidnagel, kerrie, zout
|
|||||||
De linzgen wassen en in-l liter kokend wa-
|
De linzgen wassen en in-l liter kokend wa-
|
||||||
ter 1 dag laten weken, 2 liter water bij
|
ter 1 dag laten weken, 2 liter water bij
|
||||||
de linzen voegen, zonder het water waarin
|
de linzen voegen, zonder het water waarin
|
||||||
ze geweekt zijn af te gieten., De helft van
|
ze geweekt zijn af te gieten, De helft van
|
||||||
de uien bakken met laurier en Kruidnagel.
|
de uien bakken met laurier en Kruidnagel.
|
||||||
Alle uien, kerrie en zgout bij de linzen
|
Alle uien, kerrie en zgout bij de linzen
|
||||||
voegen, Alles aan de kook brengen, Van de
|
voegen, Alles aan de kook brengen, Van de
|
||||||
bloem met boter en melk een papje maken en
|
bloem met boter en melk een papje maken en
|
||||||
verder afmaken met de soep, Als de linzen
|
verder afmaken met de soep, Als de linzen
|
||||||
gfgaar Zijn is de soep klaar.
|
gfgaar Zijn is de soep klaar.
|
||||||
|
|
||||||
BIN
Binary file not shown.
-1
@@ -1 +0,0 @@
|
|||||||
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
|
||||||
|
|||||||
-15
@@ -1,25 +1,11 @@
|
|||||||
Tarnose
|
Tarnose
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Bokale oa
|
Bokale oa
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Lehuntze
|
Lehuntze
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Mugerre
|
Mugerre
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
Milafranga Komunikabideak
|
Milafranga Komunikabideak
|
||||||
|
|
||||||
BAIONA zeiteninsiie —
|
BAIONA zeiteninsiie —
|
||||||
@@ -27,4 +13,3 @@ BAIONA zeiteninsiie —
|
|||||||
7 Trenbideak -----
|
7 Trenbideak -----
|
||||||
|
|
||||||
t\ Basusarri — spmeans:20141004 ae: . _ ~
|
t\ Basusarri — spmeans:20141004 ae: . _ ~
|
||||||
|
|
||||||
BIN
Binary file not shown.
-1
@@ -1 +0,0 @@
|
|||||||
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
|
||||||
|
|||||||
-1
@@ -1,2 +1 @@
|
|||||||
Covfefe is a perfectly cromulent word.
|
Covfefe is a perfectly cromulent word.
|
||||||
|
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user