Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0a1216bf14 | ||
|
|
f10a0f7707 | ||
|
|
dc11802809 | ||
|
|
4cce0077d0 | ||
|
|
d293e05946 | ||
|
|
4030258bbc | ||
|
|
db388165a9 | ||
|
|
3d6907f7f6 | ||
|
|
701c3b371b | ||
|
|
684e5b4944 | ||
|
|
a964080f77 | ||
|
|
3f72f16958 | ||
|
|
b4f2582766 | ||
|
|
c77cc7c837 | ||
|
|
f3715daf15 | ||
|
|
c87221a4e6 | ||
|
|
c409fa5825 | ||
|
|
09c485bd88 | ||
|
|
9d51a1b5ab | ||
|
|
43e7765efd | ||
|
|
352f009c77 | ||
|
|
399b5548ca | ||
|
|
7b1e5b4f41 | ||
|
|
33e0b16174 | ||
|
|
a613722e96 | ||
|
|
ad0126185f | ||
|
|
be45871d10 | ||
|
|
252221fd8b | ||
|
|
d25c49ba81 | ||
|
|
5112e9e857 | ||
|
|
757b72b0af | ||
|
|
75c5b92cb9 | ||
|
|
d673126994 | ||
|
|
d89a633ba7 | ||
|
|
710d797299 | ||
|
|
fc75254c60 | ||
|
|
f453e94f14 | ||
|
|
8f8aaa93ed | ||
|
|
a90b9e669f | ||
|
|
e4f69cc1d6 | ||
|
|
a5852ba199 | ||
|
|
051b9da991 | ||
|
|
139d9f9841 | ||
|
|
532d65a355 | ||
|
|
9de38afb13 | ||
|
|
913c939dc9 | ||
|
|
9db9a3d6ec | ||
|
|
8423bd549b | ||
|
|
336d274a54 | ||
|
|
906d77b389 | ||
|
|
9416e850ff | ||
|
|
fd3248869c | ||
|
|
2a09a668f6 | ||
|
|
e788dde607 | ||
|
|
173a80864d | ||
|
|
aa115a8be3 | ||
|
|
e0441c4aa1 | ||
|
|
a861c58da2 | ||
|
|
b1306bd7a8 | ||
|
|
3a6eb383dc | ||
|
|
6c942ecefd | ||
|
|
a25f8ecc62 | ||
|
|
e1f4813d94 | ||
|
|
af526f078d | ||
|
|
16438c1312 | ||
|
|
dd6cb7ce20 | ||
|
|
653e2e23df | ||
|
|
a2033698fa | ||
|
|
ec1d585d40 | ||
|
|
a4e1f8e1f3 | ||
|
|
2e155c31bf | ||
|
|
e09ae9c68a | ||
|
|
c5bf1dd90d | ||
|
|
e96770c5e4 | ||
|
|
f4f0f3c022 | ||
|
|
0a42934c08 | ||
|
|
d8f47768f9 | ||
|
|
c9594a4a5f | ||
|
|
873f915212 | ||
|
|
079c162a96 | ||
|
|
25c8c4656f | ||
|
|
ffcae9a1a0 | ||
|
|
6e71fe1186 | ||
|
|
0885799010 | ||
|
|
8ffc99f648 | ||
|
|
2261c51eff | ||
|
|
5c470778a3 | ||
|
|
4124889f36 | ||
|
|
a23c22b0e8 | ||
|
|
dd1f5f7215 | ||
|
|
5e2206bae7 | ||
|
|
079ee86d43 | ||
|
|
3692868004 | ||
|
|
064f935699 | ||
|
|
8770fff968 | ||
|
|
82de78b6b0 | ||
|
|
2a52c6dec2 | ||
|
|
2898879be7 | ||
|
|
18e613657c | ||
|
|
a48ca556c7 | ||
|
|
9cba738b48 | ||
|
|
bccf2f423f | ||
|
|
390fdf8c05 | ||
|
|
166de3086b | ||
|
|
206c675df6 | ||
|
|
6c8f9223e9 | ||
|
|
85c6a974ca | ||
|
|
dccdcfaa91 | ||
|
|
b1da09f141 | ||
|
|
42c84531e4 | ||
|
|
a9ad805347 | ||
|
|
16bda74974 | ||
|
|
d274d88929 | ||
|
|
327df5cbbc | ||
|
|
46d0632fe2 | ||
|
|
ef1e7a814e | ||
|
|
1084724937 | ||
|
|
ecb0109d79 | ||
|
|
386cabff00 | ||
|
|
3bd5054634 | ||
|
|
6a8dd65aa2 | ||
|
|
6083b4f0a7 | ||
|
|
1a3ce59476 | ||
|
|
c395436ba3 | ||
|
|
8d23d0b441 | ||
|
|
c6a2716cdb | ||
|
|
5545bae76f | ||
|
|
7bccb8c748 | ||
|
|
173c0d1274 | ||
|
|
6953f32465 | ||
|
|
26b4d9bb4b | ||
|
|
34e564cd7d | ||
|
|
504d5776d2 | ||
|
|
ee23976858 | ||
|
|
f559316881 | ||
|
|
084610c242 | ||
|
|
9ff627472b | ||
|
|
956310d1ec | ||
|
|
1a982da442 | ||
|
|
4879a1f0de | ||
|
|
ce66bcc9c8 | ||
|
|
1ebf3144af | ||
|
|
7a1cccbc4e | ||
|
|
ebacff1b39 | ||
|
|
c7c447be66 | ||
|
|
91aa175602 | ||
|
|
b267494e4a | ||
|
|
f687180ecc | ||
|
|
6f4b38b103 | ||
|
|
d32324859c | ||
|
|
48222b87b5 | ||
|
|
62e5edc72b | ||
|
|
2846d46bb8 | ||
|
|
47ef1914d4 | ||
|
|
df157552f3 | ||
|
|
0b3a526049 | ||
|
|
1e80d412fa | ||
|
|
df6e106203 | ||
|
|
bd0f005861 | ||
|
|
6ba4b7b3f3 | ||
|
|
2c11349ee8 | ||
|
|
b0afef09ef | ||
|
|
72fa347c38 | ||
|
|
96d68c2413 | ||
|
|
babc76fa74 | ||
|
|
dc06990e5d | ||
|
|
0ff0d2f8d1 | ||
|
|
81602cf420 | ||
|
|
607e2d7e81 | ||
|
|
b01d9e07e8 | ||
|
|
91db94cf2e | ||
|
|
416df803d4 | ||
|
|
037b96ca16 | ||
|
|
bb258fc99c | ||
|
|
4b8ccbe8cb | ||
|
|
ab1ff3331b | ||
|
|
3675ae918c | ||
|
|
0ba32b96b7 | ||
|
|
add64e4fa2 | ||
|
|
7fe2954ede | ||
|
|
ad202693b3 | ||
|
|
594ef83551 | ||
|
|
78b71618c1 | ||
|
|
b8aa89e1ec | ||
|
|
b4c1f66bc1 | ||
|
|
5172dbde8d | ||
|
|
d2908640c6 | ||
|
|
997bf7578d | ||
|
|
043258242c | ||
|
|
156d5d9a9c | ||
|
|
0b7e52fb5e | ||
|
|
a5feef07d0 | ||
|
|
f11bb53e61 | ||
|
|
68a57a7839 | ||
|
|
4194430dc1 | ||
|
|
a707c56fae | ||
|
|
3cba50bfbd | ||
|
|
ed5e17d0a4 | ||
|
|
ce0e0ecd4d | ||
|
|
7e1223c12c | ||
|
|
b83d7f6d1a | ||
|
|
80e957908a | ||
|
|
f0e7bea8ba | ||
|
|
0cdb9bd04a | ||
|
|
8224d89bc6 | ||
|
|
a2bbbe2a26 | ||
|
|
43f41863fa | ||
|
|
d71e50e83d | ||
|
|
1f598da3c1 | ||
|
|
d0cdbd5e1c | ||
|
|
5c56f61209 | ||
|
|
9bec85470a | ||
|
|
a03863a17d | ||
|
|
22cd9b2364 | ||
|
|
4fc7d6d93e | ||
|
|
71f0e7f545 | ||
|
|
895fddd85e | ||
|
|
5a59e4d543 | ||
|
|
b51abf2249 | ||
|
|
6d3f9ff15a | ||
|
|
5d1d1a712b | ||
|
|
6d5f8133e0 | ||
|
|
13018d3d5c | ||
|
|
14a85f9473 | ||
|
|
d22a1b3367 | ||
|
|
b913e5dfef | ||
|
|
dd8a5a4c72 | ||
|
|
36e9a54f02 | ||
|
|
3707af3b74 | ||
|
|
ced7ad9164 | ||
|
|
54bbbfdeb3 | ||
|
|
7f73a6ed1e | ||
|
|
dce206d3dc | ||
|
|
9304c856cf | ||
|
|
e5df98cbdf | ||
|
|
19bf3aeb00 | ||
|
|
e86be0031c | ||
|
|
6425977998 | ||
|
|
d57df2d980 | ||
|
|
664d0c7969 | ||
|
|
a354663ee1 | ||
|
|
b21b048ec4 | ||
|
|
709c65b41a | ||
|
|
67f99c5bb7 | ||
|
|
d55e673d9c | ||
|
|
21b90d2d14 | ||
|
|
2def7e3392 | ||
|
|
b0dcaa7512 | ||
|
|
e8285b1d10 | ||
|
|
5ba56adb53 | ||
|
|
ca735278e0 | ||
|
|
b5ccbfdf25 | ||
|
|
8c35d6e6e4 | ||
|
|
d1e0c81eda | ||
|
|
10c8e4f8b4 | ||
|
|
6be2242c21 | ||
|
|
204c9d6ae1 | ||
|
|
6eb393590b | ||
|
|
07c6654057 | ||
|
|
4e15eb8d14 | ||
|
|
8b01ab8ad2 | ||
|
|
e0a522ad50 | ||
|
|
a1a8788c5a | ||
|
|
cccdc178c3 | ||
|
|
4eacb3454f | ||
|
|
82b8b41e80 | ||
|
|
581c5020ab | ||
|
|
3ef8872a1e | ||
|
|
28eec73eed | ||
|
|
bfe4a5b329 | ||
|
|
29097837d6 | ||
|
|
a40361db3c | ||
|
|
8b29e3cbab | ||
|
|
b170be120b | ||
|
|
9a6cd95e5f | ||
|
|
d464d3122e | ||
|
|
1327ab37d4 | ||
|
|
67553fc5c6 | ||
|
|
306a903854 | ||
|
|
b93cf51c0f | ||
|
|
6b994221c6 | ||
|
|
8b5b02e0d8 | ||
|
|
624df9bb23 | ||
|
|
fa06ea3600 | ||
|
|
31994258fb | ||
|
|
1f15ecbca5 | ||
|
|
bcf5657e5c | ||
|
|
2ae028bf38 | ||
|
|
b51a5887e5 | ||
|
|
fc523e837c | ||
|
|
caeba76a61 | ||
|
|
cd35216f21 | ||
|
|
e6a7b58863 | ||
|
|
56184a762f | ||
|
|
07ab98f5af | ||
|
|
04fb1892b4 | ||
|
|
173ce2f215 | ||
|
|
9b641055e1 | ||
|
|
4fa28d7e74 | ||
|
|
bed74501fc | ||
|
|
8c90f7c972 | ||
|
|
aa0ec40102 | ||
|
|
12c567ee10 | ||
|
|
d39778ce3a | ||
|
|
e824cdbc4e | ||
|
|
1d91c09963 | ||
|
|
e821ca46d5 | ||
|
|
a29e4952fb | ||
|
|
4cc0dc6b4a | ||
|
|
7263702de9 |
-24
@@ -1,24 +0,0 @@
|
|||||||
[paths]
|
|
||||||
source =
|
|
||||||
src
|
|
||||||
*/site-packages
|
|
||||||
|
|
||||||
[run]
|
|
||||||
branch = true
|
|
||||||
parallel = true
|
|
||||||
concurrency =
|
|
||||||
thread
|
|
||||||
multiprocessing
|
|
||||||
source =
|
|
||||||
src/ocrmypdf
|
|
||||||
|
|
||||||
[report]
|
|
||||||
exclude_lines =
|
|
||||||
pragma: no cover
|
|
||||||
def __repr__
|
|
||||||
raise AssertionError
|
|
||||||
raise NotImplementedError
|
|
||||||
if 0:
|
|
||||||
if False:
|
|
||||||
if __name__ == .__main__.:
|
|
||||||
if TYPE_CHECKING:
|
|
||||||
+6
-9
@@ -10,8 +10,10 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
|||||||
build-essential autoconf automake libtool \
|
build-essential autoconf automake libtool \
|
||||||
libleptonica-dev \
|
libleptonica-dev \
|
||||||
zlib1g-dev \
|
zlib1g-dev \
|
||||||
python3 \
|
python3-dev \
|
||||||
python3-distutils \
|
python3-distutils \
|
||||||
|
libffi-dev \
|
||||||
|
libqpdf-dev \
|
||||||
ca-certificates \
|
ca-certificates \
|
||||||
curl \
|
curl \
|
||||||
git
|
git
|
||||||
@@ -35,12 +37,7 @@ COPY . /app
|
|||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
RUN pip3 install --no-cache-dir \
|
RUN pip3 install --no-cache-dir .[test,webservice,watcher]
|
||||||
-r requirements/main.txt \
|
|
||||||
-r requirements/webservice.txt \
|
|
||||||
-r requirements/test.txt \
|
|
||||||
-r requirements/watcher.txt \
|
|
||||||
.
|
|
||||||
|
|
||||||
FROM base
|
FROM base
|
||||||
|
|
||||||
@@ -62,7 +59,8 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
|||||||
tesseract-ocr-fra \
|
tesseract-ocr-fra \
|
||||||
tesseract-ocr-por \
|
tesseract-ocr-por \
|
||||||
tesseract-ocr-spa \
|
tesseract-ocr-spa \
|
||||||
unpaper
|
unpaper \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
@@ -76,6 +74,5 @@ COPY --from=builder /app/misc/watcher.py /app/
|
|||||||
COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/
|
COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/
|
||||||
COPY --from=builder /app/requirements /app/requirements
|
COPY --from=builder /app/requirements /app/requirements
|
||||||
COPY --from=builder /app/tests /app/tests
|
COPY --from=builder /app/tests /app/tests
|
||||||
COPY --from=builder /app/src /app/src
|
|
||||||
|
|
||||||
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
||||||
|
|||||||
@@ -0,0 +1,32 @@
|
|||||||
|
---
|
||||||
|
name: General issues
|
||||||
|
about: Installation, packages, dependencies, "nothing works", test suite failures...
|
||||||
|
title: ''
|
||||||
|
labels: ''
|
||||||
|
assignees: ''
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Describe the bug**
|
||||||
|
What's the problem?
|
||||||
|
|
||||||
|
**To Reproduce**
|
||||||
|
Steps to reproduce the behavior.
|
||||||
|
|
||||||
|
**Expected behavior**
|
||||||
|
What did you expected to happen?
|
||||||
|
|
||||||
|
**Screenshots**
|
||||||
|
If applicable, add screenshots to help explain your problem.
|
||||||
|
|
||||||
|
**System (please complete the following information):**
|
||||||
|
- OS:
|
||||||
|
- Python version:
|
||||||
|
- OCRmyPDF version:
|
||||||
|
|
||||||
|
**Installation**
|
||||||
|
How did you install OCRmyPDF? Did you install it from your operating system's
|
||||||
|
package manager, or using pip?
|
||||||
|
|
||||||
|
**Additional context**
|
||||||
|
Add any other context about the problem here.
|
||||||
+5
-5
@@ -1,6 +1,6 @@
|
|||||||
---
|
---
|
||||||
name: Bug report
|
name: Problem with a specific input file
|
||||||
about: Create a report to help us improve
|
about: Something went wrong while trying to OCR a specific file
|
||||||
title: ''
|
title: ''
|
||||||
labels: ''
|
labels: ''
|
||||||
assignees: ''
|
assignees: ''
|
||||||
@@ -20,13 +20,13 @@ ocrmypdf ...arguments... input.pdf output.pdf
|
|||||||
Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful.
|
Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful.
|
||||||
|
|
||||||
**Example file**
|
**Example file**
|
||||||
Include an input PDF or image that demonstrates your issue.
|
If your issue is a problem that affects only certain files, and we will require an input file (PDF or image) that demonstrates your issue.
|
||||||
|
|
||||||
Please provide an input file with no personal or confidential information. At your option you may `GPG-encrypt the file <https://github.com/jbarlow83/OCRmyPDF/wiki>` for OCRmyPDF's author only.
|
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/jbarlow83/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
||||||
|
|
||||||
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
||||||
|
|
||||||
(Exceptions: Issues with installation, command line argument parsing, test suite failures.Issues without example files usually cannot be resolved.)
|
*(Issues without example files usually cannot be resolved. It's like reporting an issue against a web browser without providing a URL.)*
|
||||||
|
|
||||||
**Expected behavior**
|
**Expected behavior**
|
||||||
A clear and concise description of what you expected to happen.
|
A clear and concise description of what you expected to happen.
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
---
|
||||||
|
name: Feature request
|
||||||
|
about: Suggest an idea for this project
|
||||||
|
title: ''
|
||||||
|
labels: ''
|
||||||
|
assignees: ''
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Is your feature request related to a problem? Please describe.**
|
||||||
|
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
||||||
|
|
||||||
|
**Describe the solution you'd like**
|
||||||
|
A clear and concise description of what you want to happen.
|
||||||
|
|
||||||
|
**Describe alternatives you've considered**
|
||||||
|
A clear and concise description of any alternative solutions or features you've considered. Please include the versions of OCRmyPDF and other supporting programs (Tesseract OCR, Ghostscript) - maybe an alternative already exists in a newer version.
|
||||||
|
|
||||||
|
**Example file**
|
||||||
|
If your issue concerns how OCRmyPDF processes certain files, and please provide an example file that helps illustrate how OCRmyPDF's output could be improve.
|
||||||
|
|
||||||
|
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/jbarlow83/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
||||||
|
|
||||||
|
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
||||||
|
|
||||||
|
**Additional context**
|
||||||
|
Add any other context or screenshots about the feature request here.
|
||||||
@@ -1,17 +0,0 @@
|
|||||||
---
|
|
||||||
name: Feature request
|
|
||||||
about: Suggest an idea for this project
|
|
||||||
title: ''
|
|
||||||
labels: enhancement
|
|
||||||
assignees: ''
|
|
||||||
|
|
||||||
---
|
|
||||||
|
|
||||||
**Is your feature request related to a problem? Please describe.**
|
|
||||||
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
|
||||||
|
|
||||||
**Describe the solution you'd like**
|
|
||||||
A clear and concise description of what you want to happen.
|
|
||||||
|
|
||||||
**Additional context**
|
|
||||||
Add any other context or screenshots about the feature request here.
|
|
||||||
@@ -1,33 +0,0 @@
|
|||||||
**Describe the issue**
|
|
||||||
A clear and concise description of what the issue is.
|
|
||||||
|
|
||||||
**To Reproduce**
|
|
||||||
What command line were you trying to run?
|
|
||||||
|
|
||||||
```bash
|
|
||||||
ocrmypdf ...arguments... input.pdf output.pdf
|
|
||||||
```
|
|
||||||
|
|
||||||
**Example file**
|
|
||||||
Please include an example *input* PDF (or image). The input file is more helpful.
|
|
||||||
|
|
||||||
Please check any or all that apply about the test file:
|
|
||||||
|
|
||||||
- [ ] This is the input file
|
|
||||||
- [ ] The file contains no personal or confidential information
|
|
||||||
- [ ] I am the copyright holder for this file
|
|
||||||
- [ ] I permit this file to be included in the OCRmyPDF test suite under the CC-BY-SA 4.0 license
|
|
||||||
- [ ] I am not the copyright holder, but this file is available under a free software license
|
|
||||||
|
|
||||||
Files that are not free for inclusion in this project are quite welcome, but we like to collect free files for our test suite when possible. Please do *not* submit files with confidential information. At your option you may encrypt files for OCRmyPDF's author only.
|
|
||||||
|
|
||||||
**Expected behavior**
|
|
||||||
A clear and concise description of what you expected to happen. Include screenshots if applicable.
|
|
||||||
|
|
||||||
**System:**
|
|
||||||
|
|
||||||
- OS: [e.g. Linux, macOS]
|
|
||||||
- OCRmyPDF Version: [e.g. v7.4.0]
|
|
||||||
|
|
||||||
**Additional context**
|
|
||||||
Add any other context about the problem here.
|
|
||||||
@@ -0,0 +1,275 @@
|
|||||||
|
name: Test and deploy
|
||||||
|
|
||||||
|
on:
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- master
|
||||||
|
- ci
|
||||||
|
- release/*
|
||||||
|
tags:
|
||||||
|
- v*
|
||||||
|
paths-ignore:
|
||||||
|
- README*
|
||||||
|
pull_request:
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
test_linux:
|
||||||
|
name: Test ${{ matrix.os }} with Python ${{ matrix.python }}
|
||||||
|
runs-on: ${{ matrix.os }}
|
||||||
|
strategy:
|
||||||
|
matrix:
|
||||||
|
os: [ubuntu-18.04] #, ubuntu-20.04]
|
||||||
|
python: ["3.6"] #, "3.7", "3.8", "3.9"]
|
||||||
|
|
||||||
|
env:
|
||||||
|
OS: ${{ matrix.os }}
|
||||||
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v2
|
||||||
|
with:
|
||||||
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
|
- uses: actions/setup-python@v2
|
||||||
|
name: Install Python
|
||||||
|
with:
|
||||||
|
python-version: ${{ matrix.python }}
|
||||||
|
|
||||||
|
- name: Install common packages
|
||||||
|
run: |
|
||||||
|
sudo apt-get update
|
||||||
|
sudo apt-get install -y --no-install-recommends \
|
||||||
|
curl \
|
||||||
|
ghostscript \
|
||||||
|
img2pdf \
|
||||||
|
libffi-dev \
|
||||||
|
liblept5 \
|
||||||
|
libsm6 libxext6 libxrender-dev \
|
||||||
|
pngquant \
|
||||||
|
poppler-utils \
|
||||||
|
tesseract-ocr \
|
||||||
|
tesseract-ocr-deu \
|
||||||
|
tesseract-ocr-eng \
|
||||||
|
unpaper \
|
||||||
|
zlib1g
|
||||||
|
|
||||||
|
- name: Install Ubuntu 18.04 packages
|
||||||
|
if: matrix.os == 'ubuntu-18.04'
|
||||||
|
run: |
|
||||||
|
sudo apt-get install -y --no-install-recommends \
|
||||||
|
libexempi3
|
||||||
|
|
||||||
|
- name: Install Ubuntu 20.04 packages
|
||||||
|
if: matrix.os == 'ubuntu-20.04'
|
||||||
|
run: |
|
||||||
|
sudo apt-get install -y --no-install-recommends \
|
||||||
|
libexempi8
|
||||||
|
|
||||||
|
- name: Install Python packages
|
||||||
|
run: |
|
||||||
|
python -m pip install .[test]
|
||||||
|
|
||||||
|
- name: Report versions
|
||||||
|
run: |
|
||||||
|
tesseract --version
|
||||||
|
gs --version
|
||||||
|
pngquant --version
|
||||||
|
unpaper --version
|
||||||
|
img2pdf --version
|
||||||
|
|
||||||
|
- name: Test
|
||||||
|
run: |
|
||||||
|
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
|
- name: Upload coverage to Codecov
|
||||||
|
uses: codecov/codecov-action@v1
|
||||||
|
with:
|
||||||
|
files: ./coverage.xml
|
||||||
|
env_vars: OS,PYTHON
|
||||||
|
|
||||||
|
test_macos:
|
||||||
|
name: Test macOS
|
||||||
|
runs-on: ${{ matrix.os }}
|
||||||
|
strategy:
|
||||||
|
matrix:
|
||||||
|
os: [macos-latest]
|
||||||
|
python: ["3.9"]
|
||||||
|
|
||||||
|
env:
|
||||||
|
OS: ${{ matrix.os }}
|
||||||
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v2
|
||||||
|
with:
|
||||||
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
|
- uses: actions/setup-python@v2
|
||||||
|
name: Install Python
|
||||||
|
with:
|
||||||
|
python-version: ${{ matrix.python }}
|
||||||
|
|
||||||
|
- name: Install Homebrew deps
|
||||||
|
run: |
|
||||||
|
brew update
|
||||||
|
brew install \
|
||||||
|
exempi \
|
||||||
|
ghostscript \
|
||||||
|
jbig2enc \
|
||||||
|
leptonica \
|
||||||
|
openjpeg \
|
||||||
|
pngquant \
|
||||||
|
tesseract
|
||||||
|
|
||||||
|
- name: Install Python packages
|
||||||
|
run: |
|
||||||
|
python -m pip install --upgrade pip
|
||||||
|
python -m pip install .[test]
|
||||||
|
|
||||||
|
- name: Report versions
|
||||||
|
run: |
|
||||||
|
tesseract --version
|
||||||
|
gs --version
|
||||||
|
pngquant --version
|
||||||
|
img2pdf --version
|
||||||
|
|
||||||
|
- name: Test
|
||||||
|
run: |
|
||||||
|
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
|
- name: Upload coverage to Codecov
|
||||||
|
uses: codecov/codecov-action@v1
|
||||||
|
with:
|
||||||
|
files: ./coverage.xml
|
||||||
|
env_vars: OS,PYTHON
|
||||||
|
|
||||||
|
test_windows:
|
||||||
|
name: Test Windows
|
||||||
|
runs-on: ${{ matrix.os }}
|
||||||
|
strategy:
|
||||||
|
matrix:
|
||||||
|
os: [windows-latest]
|
||||||
|
python: ["3.9"]
|
||||||
|
|
||||||
|
env:
|
||||||
|
OS: ${{ matrix.os }}
|
||||||
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v2
|
||||||
|
with:
|
||||||
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
|
- uses: actions/setup-python@v2
|
||||||
|
name: Install Python
|
||||||
|
with:
|
||||||
|
python-version: ${{ matrix.python }}
|
||||||
|
|
||||||
|
- name: Install system packages
|
||||||
|
run: |
|
||||||
|
choco install --yes --no-progress --pre tesseract
|
||||||
|
choco install --yes --no-progress ghostscript
|
||||||
|
choco install --yes --no-progress pngquant
|
||||||
|
|
||||||
|
- name: Install Python packages
|
||||||
|
run: |
|
||||||
|
python -m pip install --upgrade pip
|
||||||
|
python -m pip install .[test]
|
||||||
|
|
||||||
|
- name: Test
|
||||||
|
run: |
|
||||||
|
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
|
- name: Upload coverage to Codecov
|
||||||
|
uses: codecov/codecov-action@v1
|
||||||
|
with:
|
||||||
|
files: ./coverage.xml
|
||||||
|
env_vars: OS,PYTHON
|
||||||
|
|
||||||
|
wheel_sdist_linux:
|
||||||
|
name: Build sdist and wheels
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v2
|
||||||
|
with:
|
||||||
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
|
- uses: actions/setup-python@v2
|
||||||
|
name: Install Python
|
||||||
|
with:
|
||||||
|
python-version: "3.6"
|
||||||
|
|
||||||
|
- name: Make wheels and sdist
|
||||||
|
run: |
|
||||||
|
python -m pip install --upgrade pip wheel
|
||||||
|
python setup.py sdist
|
||||||
|
python setup.py bdist_wheel
|
||||||
|
|
||||||
|
- uses: actions/upload-artifact@v2
|
||||||
|
with:
|
||||||
|
path: |
|
||||||
|
./dist/*.whl
|
||||||
|
./dist/*.tar.gz
|
||||||
|
|
||||||
|
upload_pypi:
|
||||||
|
name: Deploy artifacts to PyPI
|
||||||
|
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||||
|
steps:
|
||||||
|
- uses: actions/download-artifact@v2
|
||||||
|
with:
|
||||||
|
name: artifact
|
||||||
|
path: dist
|
||||||
|
|
||||||
|
- uses: pypa/gh-action-pypi-publish@master
|
||||||
|
with:
|
||||||
|
user: __token__
|
||||||
|
password: ${{ secrets.TOKEN_PYPI }}
|
||||||
|
# repository_url: https://test.pypi.org/legacy/
|
||||||
|
|
||||||
|
docker:
|
||||||
|
name: Build Docker images
|
||||||
|
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Set image tag to release or branch
|
||||||
|
run: echo "DOCKER_IMAGE_TAG=${GITHUB_REF##*/}" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
- name: If master, set to latest
|
||||||
|
run: echo 'DOCKER_IMAGE_TAG=latest' >> $GITHUB_ENV
|
||||||
|
if: env.DOCKER_IMAGE_TAG == 'master'
|
||||||
|
|
||||||
|
- name: Set Docker Hub repository to username
|
||||||
|
run: echo "DOCKER_REPOSITORY=jbarlow83" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
- name: Set image name
|
||||||
|
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
||||||
|
|
||||||
|
- uses: actions/checkout@v2
|
||||||
|
with:
|
||||||
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
|
- name: Login to Docker Hub
|
||||||
|
uses: docker/login-action@v1
|
||||||
|
with:
|
||||||
|
username: jbarlow83
|
||||||
|
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||||
|
|
||||||
|
- name: Set up QEMU
|
||||||
|
uses: docker/setup-qemu-action@v1
|
||||||
|
|
||||||
|
- name: Set up Docker Buildx
|
||||||
|
id: buildx
|
||||||
|
uses: docker/setup-buildx-action@v1
|
||||||
|
|
||||||
|
- name: Print image tag
|
||||||
|
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
||||||
|
|
||||||
|
- name: Build
|
||||||
|
run: |
|
||||||
|
docker buildx build \
|
||||||
|
--push \
|
||||||
|
--platform linux/arm64/v8,linux/amd64 \
|
||||||
|
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \
|
||||||
|
--file .docker/Dockerfile .
|
||||||
+4
-1
@@ -6,7 +6,9 @@
|
|||||||
!.gitattributes
|
!.gitattributes
|
||||||
!.gitignore
|
!.gitignore
|
||||||
!.pre-commit-config.yaml
|
!.pre-commit-config.yaml
|
||||||
!.readthedocs.yml
|
!.readthedocs.yaml
|
||||||
|
!.github/
|
||||||
|
!.docker/
|
||||||
|
|
||||||
# Dev scratch
|
# Dev scratch
|
||||||
*.ipynb
|
*.ipynb
|
||||||
@@ -23,6 +25,7 @@ venv*/
|
|||||||
/debug_tests.py
|
/debug_tests.py
|
||||||
*.traineddata
|
*.traineddata
|
||||||
/private
|
/private
|
||||||
|
/coverage.xml
|
||||||
|
|
||||||
# Package building
|
# Package building
|
||||||
*.egg-info/
|
*.egg-info/
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
repos:
|
repos:
|
||||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||||
rev: v3.1.0
|
rev: v3.4.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: check-case-conflict
|
- id: check-case-conflict
|
||||||
- id: check-merge-conflict
|
- id: check-merge-conflict
|
||||||
@@ -12,12 +12,12 @@ repos:
|
|||||||
hooks:
|
hooks:
|
||||||
- id: seed-isort-config
|
- id: seed-isort-config
|
||||||
- repo: https://github.com/pre-commit/mirrors-isort
|
- repo: https://github.com/pre-commit/mirrors-isort
|
||||||
rev: v5.0.5 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases
|
rev: v5.7.0 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases
|
||||||
hooks:
|
hooks:
|
||||||
- id: isort
|
- id: isort
|
||||||
- repo: https://github.com/psf/black
|
- repo: https://github.com/psf/black
|
||||||
rev: 19.10b0
|
rev: 20.8b1
|
||||||
hooks:
|
hooks:
|
||||||
- id: black
|
- id: black
|
||||||
language_version: python3.8
|
language_version: python
|
||||||
exclude: ^src/ocrmypdf/lib/_leptonica.py
|
exclude: ^src/ocrmypdf/lib/_leptonica.py
|
||||||
|
|||||||
@@ -0,0 +1,22 @@
|
|||||||
|
# Read the Docs configuration file
|
||||||
|
# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details
|
||||||
|
|
||||||
|
# Required
|
||||||
|
version: 2
|
||||||
|
|
||||||
|
# Build documentation in the docs/ directory with Sphinx
|
||||||
|
sphinx:
|
||||||
|
configuration: docs/conf.py
|
||||||
|
|
||||||
|
# Optionally build your docs in additional formats such as PDF
|
||||||
|
formats:
|
||||||
|
- pdf
|
||||||
|
|
||||||
|
# Optionally set the version of Python and requirements required to build your docs
|
||||||
|
python:
|
||||||
|
version: 3.7
|
||||||
|
install:
|
||||||
|
- method: pip
|
||||||
|
path: .
|
||||||
|
extra_requirements:
|
||||||
|
- docs
|
||||||
@@ -1,10 +0,0 @@
|
|||||||
build:
|
|
||||||
image: latest
|
|
||||||
|
|
||||||
python:
|
|
||||||
version: 3.6
|
|
||||||
|
|
||||||
formats:
|
|
||||||
- pdf
|
|
||||||
|
|
||||||
requirements_file: requirements/main.txt
|
|
||||||
@@ -1,6 +1,6 @@
|
|||||||
<img src="docs/images/logo.svg" width="240" alt="OCRmyPDF">
|
<img src="docs/images/logo.svg" width="240" alt="OCRmyPDF">
|
||||||
|
|
||||||
[![Build Status][azure]](https://dev.azure.com/jim0585/ocrmypdf/_build/latest?definitionId=2&branchName=master) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions]
|
[](https://github.com/jbarlow83/OCRmyPDF/actions/workflows/build.yml) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions]
|
||||||
|
|
||||||
[azure]: https://dev.azure.com/jim0585/ocrmypdf/_apis/build/status/jbarlow83.OCRmyPDF?branchName=master
|
[azure]: https://dev.azure.com/jim0585/ocrmypdf/_apis/build/status/jbarlow83.OCRmyPDF?branchName=master
|
||||||
[travis]: https://travis-ci.org/jbarlow83/OCRmyPDF.svg?branch=master "Travis build status"
|
[travis]: https://travis-ci.org/jbarlow83/OCRmyPDF.svg?branch=master "Travis build status"
|
||||||
@@ -57,25 +57,17 @@ I searched the web for a free command line tool to OCR PDF files: I found many,
|
|||||||
|
|
||||||
## Installation
|
## Installation
|
||||||
|
|
||||||
Linux, Windows, macOS and FreeBSD are supported. Docker images are also available.
|
Linux, Windows, macOS and FreeBSD are supported. Docker images are also available, for both x64 and ARM.
|
||||||
|
|
||||||
Users of Debian 9 or later or Ubuntu 16.10 or later may simply
|
| Operating system | Install command |
|
||||||
|
| ----------------------------- | ------------------------------|
|
||||||
```bash
|
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||||
apt-get install ocrmypdf
|
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||||
```
|
| Fedora | ``dnf install ocrmypdf`` |
|
||||||
|
| macOS | ``brew install ocrmypdf`` |
|
||||||
and users of Fedora 29 or later may simply
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
|
| FreeBSD | ``pkg install py37-ocrmypdf`` |
|
||||||
```bash
|
| Conda | ``conda install ocrmypdf`` |
|
||||||
dnf install ocrmypdf
|
|
||||||
```
|
|
||||||
|
|
||||||
and Homebrew users (macOS, Linux, Windows Subsystem for Linux) may simply
|
|
||||||
|
|
||||||
```bash
|
|
||||||
brew install ocrmypdf
|
|
||||||
```
|
|
||||||
|
|
||||||
For everyone else, [see our documentation](https://ocrmypdf.readthedocs.io/en/latest/installation.html) for installation steps.
|
For everyone else, [see our documentation](https://ocrmypdf.readthedocs.io/en/latest/installation.html) for installation steps.
|
||||||
|
|
||||||
@@ -92,6 +84,9 @@ apt-get install tesseract-ocr-chi-sim # Example: Install Chinese Simplified lan
|
|||||||
|
|
||||||
# Arch Linux users
|
# Arch Linux users
|
||||||
pacman -S tesseract-data-eng tesseract-data-deu # Example: Install the English and German language packs
|
pacman -S tesseract-data-eng tesseract-data-deu # Example: Install the English and German language packs
|
||||||
|
|
||||||
|
# brew macOS users
|
||||||
|
brew install tesseract-lang
|
||||||
```
|
```
|
||||||
|
|
||||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested.
|
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested.
|
||||||
@@ -120,6 +115,7 @@ In addition to the required Python version (3.6+), OCRmyPDF requires external pr
|
|||||||
- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670)
|
- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670)
|
||||||
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||||
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||||
|
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||||
|
|
||||||
## Business enquiries
|
## Business enquiries
|
||||||
|
|
||||||
@@ -127,11 +123,15 @@ OCRmyPDF would not be the software that it is today without companies and users
|
|||||||
|
|
||||||
## License
|
## License
|
||||||
|
|
||||||
The OCRmyPDF software is licensed under the GNU GPLv3. Certain files are covered by other licenses, as noted in their source files.
|
The OCRmyPDF software is licensed under the Mozilla Public License 2.0
|
||||||
|
(MPL-2.0). This license permits integration of OCRmyPDF with other code,
|
||||||
|
included commercial and closed source, but asks you to publish source-level
|
||||||
|
modifications you make to OCRmyPDF.
|
||||||
|
|
||||||
The license for each test file varies, and is noted in tests/resources/README.rst. The documentation is licensed under Creative Commons Attribution-ShareAlike 4.0 (CC-BY-SA 4.0).
|
Some components of OCRmyPDF have other licenses, as noted in those files and the
|
||||||
|
``debian/copyright`` file. Most files in ``misc/`` use the MIT license, and the
|
||||||
OCRmyPDF versions prior to 6.0 were distributed under the MIT License.
|
documentation and test files are generally licensed under Creative Commons
|
||||||
|
ShareAlike 4.0 (CC-BY-SA 4.0).
|
||||||
|
|
||||||
## Disclaimer
|
## Disclaimer
|
||||||
|
|
||||||
|
|||||||
@@ -1,259 +0,0 @@
|
|||||||
trigger:
|
|
||||||
tags:
|
|
||||||
include:
|
|
||||||
- v*
|
|
||||||
branches:
|
|
||||||
include:
|
|
||||||
- "*"
|
|
||||||
exclude:
|
|
||||||
- "travis"
|
|
||||||
|
|
||||||
stages:
|
|
||||||
- stage: "Test"
|
|
||||||
jobs:
|
|
||||||
- job: Windows
|
|
||||||
pool:
|
|
||||||
vmImage: "vs2017-win2016"
|
|
||||||
strategy:
|
|
||||||
matrix:
|
|
||||||
Python36:
|
|
||||||
python.version: "3.6"
|
|
||||||
Python37:
|
|
||||||
python.version: "3.7"
|
|
||||||
Python38:
|
|
||||||
python.version: "3.8"
|
|
||||||
steps:
|
|
||||||
- task: UsePythonVersion@0
|
|
||||||
inputs:
|
|
||||||
versionSpec: "$(python.version)"
|
|
||||||
- pwsh: |
|
|
||||||
choco install --yes --no-progress --pre tesseract
|
|
||||||
choco install --yes --no-progress python3
|
|
||||||
choco install --yes --no-progress ghostscript
|
|
||||||
choco install --yes --no-progress pngquant
|
|
||||||
displayName: "Install system packages"
|
|
||||||
- pwsh: |
|
|
||||||
refreshenv
|
|
||||||
python -m pip install --upgrade pip wheel
|
|
||||||
python -m pip install -r requirements/main.txt -r requirements/test.txt .
|
|
||||||
displayName: "Install Python packages"
|
|
||||||
- pwsh: |
|
|
||||||
refreshenv
|
|
||||||
$env:pathext += ';.py'
|
|
||||||
# -n auto helps Windows
|
|
||||||
python -m pytest -n auto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml
|
|
||||||
displayName: "Test"
|
|
||||||
- task: PublishTestResults@2
|
|
||||||
inputs:
|
|
||||||
testResultsFiles: "test.xml"
|
|
||||||
testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)"
|
|
||||||
condition: succeededOrFailed()
|
|
||||||
- job: "Ubuntu_1804"
|
|
||||||
pool:
|
|
||||||
vmImage: "ubuntu-18.04"
|
|
||||||
strategy:
|
|
||||||
matrix:
|
|
||||||
Python36:
|
|
||||||
python.version: "3.6"
|
|
||||||
Python37:
|
|
||||||
python.version: "3.7"
|
|
||||||
Python38:
|
|
||||||
python.version: "3.8"
|
|
||||||
steps:
|
|
||||||
- task: UsePythonVersion@0
|
|
||||||
inputs:
|
|
||||||
versionSpec: "$(python.version)"
|
|
||||||
- bash: |
|
|
||||||
sudo apt-get update
|
|
||||||
sudo apt-get install -y --no-install-recommends \
|
|
||||||
python3-software-properties \
|
|
||||||
curl \
|
|
||||||
ghostscript \
|
|
||||||
img2pdf \
|
|
||||||
libexempi3 \
|
|
||||||
libffi-dev \
|
|
||||||
liblept5 \
|
|
||||||
libsm6 libxext6 libxrender-dev \
|
|
||||||
pngquant \
|
|
||||||
poppler-utils \
|
|
||||||
tesseract-ocr \
|
|
||||||
tesseract-ocr-deu \
|
|
||||||
tesseract-ocr-eng \
|
|
||||||
unpaper \
|
|
||||||
zlib1g
|
|
||||||
displayName: "Install system packages"
|
|
||||||
- bash: |
|
|
||||||
curl https://bootstrap.pypa.io/get-pip.py | python3
|
|
||||||
pip3 install -r requirements/main.txt -r requirements/test.txt .
|
|
||||||
displayName: "Install Python packages"
|
|
||||||
- bash: |
|
|
||||||
tesseract --version
|
|
||||||
displayName: "Record versions"
|
|
||||||
- bash: |
|
|
||||||
# -n auto is slower on Linux and breaks on Python 3.8
|
|
||||||
pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml
|
|
||||||
displayName: "Test"
|
|
||||||
- task: PublishTestResults@2
|
|
||||||
inputs:
|
|
||||||
testResultsFiles: "test.xml"
|
|
||||||
testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)"
|
|
||||||
condition: succeededOrFailed()
|
|
||||||
- job: "Ubuntu_1604"
|
|
||||||
pool:
|
|
||||||
vmImage: "ubuntu-16.04"
|
|
||||||
strategy:
|
|
||||||
matrix:
|
|
||||||
Python36:
|
|
||||||
python.version: "3.6"
|
|
||||||
steps:
|
|
||||||
- task: UsePythonVersion@0
|
|
||||||
inputs:
|
|
||||||
versionSpec: "$(python.version)"
|
|
||||||
- bash: |
|
|
||||||
sudo apt-get update
|
|
||||||
sudo apt-get install -y --no-install-recommends \
|
|
||||||
software-properties-common
|
|
||||||
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr
|
|
||||||
sudo apt-get update
|
|
||||||
sudo apt-get install -y --no-install-recommends \
|
|
||||||
ghostscript \
|
|
||||||
img2pdf \
|
|
||||||
libexempi3 \
|
|
||||||
libffi-dev \
|
|
||||||
liblept5 \
|
|
||||||
libsm6 libxext6 libxrender-dev \
|
|
||||||
pngquant \
|
|
||||||
poppler-utils \
|
|
||||||
tesseract-ocr \
|
|
||||||
tesseract-ocr-deu \
|
|
||||||
tesseract-ocr-eng \
|
|
||||||
unpaper \
|
|
||||||
zlib1g
|
|
||||||
displayName: "Install system packages"
|
|
||||||
- bash: |
|
|
||||||
curl https://bootstrap.pypa.io/get-pip.py | python3
|
|
||||||
pip3 install -r requirements/main.txt -r requirements/test.txt .
|
|
||||||
displayName: "Install Python packages"
|
|
||||||
- bash: |
|
|
||||||
tesseract --version
|
|
||||||
displayName: "Record versions"
|
|
||||||
- bash: |
|
|
||||||
# -n auto is slower on Linux and breaks on Python 3.8
|
|
||||||
pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml
|
|
||||||
displayName: "Test"
|
|
||||||
- task: PublishTestResults@2
|
|
||||||
inputs:
|
|
||||||
testResultsFiles: "test.xml"
|
|
||||||
testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)"
|
|
||||||
condition: succeededOrFailed()
|
|
||||||
- job: "macOS_Mojave"
|
|
||||||
pool:
|
|
||||||
vmImage: "macos-10.14"
|
|
||||||
strategy:
|
|
||||||
matrix:
|
|
||||||
Python37:
|
|
||||||
python.version: ""
|
|
||||||
Python38:
|
|
||||||
python.version: "python@3.8"
|
|
||||||
steps:
|
|
||||||
# https://github.com/actions/virtual-environments/issues/664
|
|
||||||
# - task: UsePythonVersion@0
|
|
||||||
# inputs:
|
|
||||||
# versionSpec: "$(python.version)"
|
|
||||||
- bash: |
|
|
||||||
brew update
|
|
||||||
brew unlink python@2
|
|
||||||
if [ "$(python.version)" != "" ]; then
|
|
||||||
brew upgrade $(python.version)
|
|
||||||
else
|
|
||||||
echo "Using Python `python3 --version`"
|
|
||||||
fi
|
|
||||||
displayName: "Update brew and Python"
|
|
||||||
- bash: |
|
|
||||||
brew install \
|
|
||||||
exempi \
|
|
||||||
ghostscript \
|
|
||||||
jbig2enc \
|
|
||||||
leptonica \
|
|
||||||
openjpeg \
|
|
||||||
pngquant \
|
|
||||||
tesseract \
|
|
||||||
unpaper
|
|
||||||
displayName: "Install system packages"
|
|
||||||
- bash: |
|
|
||||||
pip3 install --upgrade pip
|
|
||||||
pip3 install -r requirements/main.txt -r requirements/test.txt .
|
|
||||||
displayName: "Install Python packages"
|
|
||||||
- bash: |
|
|
||||||
tesseract --version
|
|
||||||
displayName: "Record versions"
|
|
||||||
- bash: pytest -nauto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml
|
|
||||||
displayName: "Test"
|
|
||||||
- task: PublishTestResults@2
|
|
||||||
inputs:
|
|
||||||
testResultsFiles: "test.xml"
|
|
||||||
testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)"
|
|
||||||
condition: succeededOrFailed()
|
|
||||||
- task: PublishCodeCoverageResults@1
|
|
||||||
inputs:
|
|
||||||
codeCoverageTool: Cobertura
|
|
||||||
summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml"
|
|
||||||
|
|
||||||
- stage: "Artifacts"
|
|
||||||
jobs:
|
|
||||||
- job: "sdist_wheel"
|
|
||||||
pool:
|
|
||||||
vmImage: "ubuntu-18.04"
|
|
||||||
steps:
|
|
||||||
- task: UsePythonVersion@0
|
|
||||||
inputs:
|
|
||||||
versionSpec: "3.7"
|
|
||||||
- bash: |
|
|
||||||
python -m pip install --upgrade pip wheel
|
|
||||||
python setup.py sdist bdist_wheel
|
|
||||||
- publish: dist
|
|
||||||
artifact: sdist_wheel
|
|
||||||
|
|
||||||
- stage: "Deploy"
|
|
||||||
jobs:
|
|
||||||
- deployment: "PyPI"
|
|
||||||
pool:
|
|
||||||
vmImage: "ubuntu-18.04"
|
|
||||||
environment: "deploy"
|
|
||||||
strategy:
|
|
||||||
runOnce:
|
|
||||||
deploy:
|
|
||||||
steps:
|
|
||||||
- download: current
|
|
||||||
artifact: sdist_wheel
|
|
||||||
- script: |
|
|
||||||
mkdir -p dist
|
|
||||||
mv $(Pipeline.Workspace)/sdist_wheel/* dist
|
|
||||||
displayName: "Move dist files"
|
|
||||||
- task: UsePythonVersion@0
|
|
||||||
inputs:
|
|
||||||
versionSpec: "3.8"
|
|
||||||
architecture: x64
|
|
||||||
- script: |
|
|
||||||
pip install --upgrade twine
|
|
||||||
displayName: "Generate artifacts"
|
|
||||||
- script: |
|
|
||||||
cat <<FILE >.pypirc
|
|
||||||
[distutils]
|
|
||||||
index-servers =
|
|
||||||
pypi
|
|
||||||
|
|
||||||
[pypi]
|
|
||||||
username: __token__
|
|
||||||
password: $(TOKEN_PYPI)
|
|
||||||
|
|
||||||
FILE
|
|
||||||
displayName: "Generate PyPI auth file"
|
|
||||||
- script: |
|
|
||||||
python -m twine upload --config-file .pypirc dist/*
|
|
||||||
displayName: "Upload to PyPI"
|
|
||||||
condition: and(succeeded(), startsWith(variables['Build.SourceBranch'], 'refs/tags/'))
|
|
||||||
- script: |
|
|
||||||
curl -X POST -d "token=$(TOKEN_RTD)" https://readthedocs.org/api/v2/webhook/pikepdf/39557/
|
|
||||||
displayName: "Trigger ReadTheDocs"
|
|
||||||
condition: and(succeeded(), or(startsWith(variables['Build.SourceBranch'], 'refs/tags/'), startsWith(variables['Build.SourceBranch'], 'refs/heads/master')))
|
|
||||||
Vendored
+39
-10
@@ -5,15 +5,42 @@ Source: https://github.com/jbarlow83/OCRmyPDF
|
|||||||
|
|
||||||
Files: *
|
Files: *
|
||||||
Copyright:
|
Copyright:
|
||||||
(C) 2013-2017 The OCRmyPDF Authors
|
(C) 2013-2015 Julien Pfefferkorn
|
||||||
(C) 2013-2016, 2015-2017 2016, 2017, 2017-2018, 2018 James R. Barlow
|
(C) 2015-2020 James R. Barlow
|
||||||
License: GPL-3+
|
(C) 2019 Martin Wind
|
||||||
|
License: MPL-2.0
|
||||||
|
|
||||||
|
Files: misc/*
|
||||||
|
Copyright:
|
||||||
|
(C) 2020 James R. Barlow
|
||||||
|
License: Expat
|
||||||
|
|
||||||
|
Files: misc/completion/ocrmypdf.bash
|
||||||
|
Copyright:
|
||||||
|
(C) 2019 Frank Pille
|
||||||
|
(C) 2020 Alex Willner
|
||||||
|
License: Expat
|
||||||
|
|
||||||
|
Files: misc/completion/ocrmypdf.fish
|
||||||
|
Copyright:
|
||||||
|
(C) 2020 James R. Barlow
|
||||||
|
License: Expat
|
||||||
|
|
||||||
|
Files: misc/batch.py
|
||||||
|
Copyright:
|
||||||
|
(C) 2016 findingorder: https://github.com/findingorder
|
||||||
|
License: Expat
|
||||||
|
|
||||||
|
Files: misc/synology.py
|
||||||
|
Copyright:
|
||||||
|
(C) github.com/Enantiomerie
|
||||||
|
License: Expat
|
||||||
|
|
||||||
Files: misc/watcher.py
|
Files: misc/watcher.py
|
||||||
Copyright:
|
Copyright:
|
||||||
(C) 2019 Ian Alexander: https://github.com/ianalexander
|
(C) 2019 Ian Alexander: https://github.com/ianalexander
|
||||||
(C) 2020 James R. Barlow
|
(C) 2020 James R. Barlow
|
||||||
License: GPL-3+
|
License: Expat
|
||||||
|
|
||||||
Files: misc/webservice.py
|
Files: misc/webservice.py
|
||||||
Copyright: (C) 2019 James R. Barlow
|
Copyright: (C) 2019 James R. Barlow
|
||||||
@@ -33,17 +60,12 @@ Copyright: (C) 2010 Jonathan Brinley <jonathanbrinley@gmail.com>
|
|||||||
(C) 2015-16 James R. Barlow
|
(C) 2015-16 James R. Barlow
|
||||||
License: Expat
|
License: Expat
|
||||||
|
|
||||||
Files: src/ocrmypdf/pdfa.py
|
|
||||||
Copyright: (C) 2015 James R. Barlow
|
|
||||||
(C) 1986-2017 The authors of GhostScript
|
|
||||||
License: GPL-3+
|
|
||||||
|
|
||||||
Files: src/ocrmypdf/_unicodefun.py
|
Files: src/ocrmypdf/_unicodefun.py
|
||||||
Copyright: (C) 2014 Armin Ronacher
|
Copyright: (C) 2014 Armin Ronacher
|
||||||
(C) 2017 James R. Barlow
|
(C) 2017 James R. Barlow
|
||||||
License: BSD-3-clause
|
License: BSD-3-clause
|
||||||
|
|
||||||
Files: tests/spoof/*
|
Files: tests/plugins/*
|
||||||
Copyright: (C) 2016, 2017, 2016-2018 James R. Barlow
|
Copyright: (C) 2016, 2017, 2016-2018 James R. Barlow
|
||||||
License: Expat
|
License: Expat
|
||||||
|
|
||||||
@@ -132,6 +154,13 @@ Files: debian/*
|
|||||||
Copyright: (C) 2016 Sean Whitton <spwhitton@spwhitton.name>
|
Copyright: (C) 2016 Sean Whitton <spwhitton@spwhitton.name>
|
||||||
License: GPL-3+
|
License: GPL-3+
|
||||||
|
|
||||||
|
License: MPL-2.0
|
||||||
|
This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
|
License, v. 2.0.
|
||||||
|
.
|
||||||
|
On Debian systems the full text of the MPL-2.0 can be found in
|
||||||
|
/usr/share/common-licenses/MPL-2.0.
|
||||||
|
|
||||||
License: GPL-3+
|
License: GPL-3+
|
||||||
This program is free software; you can redistribute it and/or modify
|
This program is free software; you can redistribute it and/or modify
|
||||||
it under the terms of the GNU General Public License as published by
|
it under the terms of the GNU General Public License as published by
|
||||||
|
|||||||
+4
-5
@@ -125,8 +125,7 @@ include:
|
|||||||
.. envvar:: OMP_THREAD_LIMIT
|
.. envvar:: OMP_THREAD_LIMIT
|
||||||
|
|
||||||
Controls the number of threads Tesseract will use. OCRmyPDF will
|
Controls the number of threads Tesseract will use. OCRmyPDF will
|
||||||
manage this environment if it is not already set. (Currently, it will
|
manage this environment variable if it is not already set.
|
||||||
set it to 1 because this gives the best results in testing.)
|
|
||||||
|
|
||||||
For example, if you have a development build of Tesseract don't wish to
|
For example, if you have a development build of Tesseract don't wish to
|
||||||
use the system installation, you can launch OCRmyPDF as follows:
|
use the system installation, you can launch OCRmyPDF as follows:
|
||||||
@@ -229,8 +228,8 @@ preprocessing is specified, then the image layer is a new PDF.
|
|||||||
Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone
|
Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone
|
||||||
looking to customize how OCR is presented should look here. A major
|
looking to customize how OCR is presented should look here. A major
|
||||||
disadvantage of this renderer is it not capable of correctly handling
|
disadvantage of this renderer is it not capable of correctly handling
|
||||||
text outside the Latin alphabet. Pull requests to improve the situation
|
text outside the Latin alphabet (specifically, it supports the ISO 8859-1
|
||||||
are welcome.
|
character). Pull requests to improve the situation are welcome.
|
||||||
|
|
||||||
Currently, this renderer has the best compatibility with Mozilla's
|
Currently, this renderer has the best compatibility with Mozilla's
|
||||||
PDF.js viewer.
|
PDF.js viewer.
|
||||||
@@ -315,7 +314,7 @@ message is:
|
|||||||
.. code-block:: none
|
.. code-block:: none
|
||||||
|
|
||||||
Temporary working files retained at:
|
Temporary working files retained at:
|
||||||
/tmp/com.github.ocrmypdf.u20wpz07
|
/tmp/ocrmypdf.io.u20wpz07
|
||||||
|
|
||||||
The organization of this folder is an implementation detail and subject
|
The organization of this folder is an implementation detail and subject
|
||||||
to change between releases. However the general organization is that
|
to change between releases. However the general organization is that
|
||||||
|
|||||||
+21
-12
@@ -20,7 +20,8 @@ and largely have the same functions.
|
|||||||
|
|
||||||
import ocrmypdf
|
import ocrmypdf
|
||||||
|
|
||||||
ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True)
|
if __name__ == '__main__': # To ensure correct behavior on Windows and macOS
|
||||||
|
ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True)
|
||||||
|
|
||||||
With a few exceptions, all of the command line arguments are available
|
With a few exceptions, all of the command line arguments are available
|
||||||
and may be passed as equivalent keywords.
|
and may be passed as equivalent keywords.
|
||||||
@@ -35,29 +36,37 @@ The :func:`ocrmypdf.ocr` function runs OCRmyPDF similar to command line
|
|||||||
execution. To do this, it will:
|
execution. To do this, it will:
|
||||||
|
|
||||||
- create a monitoring thread
|
- create a monitoring thread
|
||||||
- create worker processes (forking itself)
|
- create worker processes (on Linux, forking itself; on Windows and macOS, by
|
||||||
- manage the signal flags of worker processes
|
spawning)
|
||||||
|
- manage the signal flags of its worker processes
|
||||||
- execute other subprocesses (forking and executing other programs)
|
- execute other subprocesses (forking and executing other programs)
|
||||||
|
|
||||||
The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently
|
The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently
|
||||||
privileged to perform these actions. If it is not, ``ocrmypdf()`` will
|
privileged to perform these actions.
|
||||||
fail.
|
|
||||||
|
|
||||||
There is no currently no option to manage how jobs are scheduled other
|
There is no currently no option to manage how jobs are scheduled other
|
||||||
than the argument ``jobs=`` which will limit the number of worker
|
than the argument ``jobs=`` which will limit the number of worker
|
||||||
processes.
|
processes.
|
||||||
|
|
||||||
Forking a child process to call ``ocrmypdf.ocr()`` is suggested. That
|
Creating a child process to call ``ocrmypdf.ocr()`` is suggested. That
|
||||||
way your application will survive and remain interactive even if
|
way your application will survive and remain interactive even if
|
||||||
OCRmyPDF does not.
|
OCRmyPDF fails for any reason.
|
||||||
|
|
||||||
|
Programs that call ``ocrmypdf.ocr()`` should also install a SIGBUS signal
|
||||||
|
handler (except on Windows), to raise an exception if access to a memory
|
||||||
|
mapped file fails. OCRmyPDF may use memory mapping.
|
||||||
|
|
||||||
|
``ocrmypdf.ocr()`` will take a threading lock to prevent multiple runs of itself
|
||||||
|
in the same Python interpreter process. This is not thread-safe, because of how
|
||||||
|
OCRmyPDF's plugins and Python's library import system work. If you need to parallelize
|
||||||
|
OCRmyPDF, use processes.
|
||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
|
|
||||||
On Windows, the script that calls ``ocrmypdf.ocr()`` must be protected
|
On Windows and macOS, the script that calls ``ocrmypdf.ocr()`` must be
|
||||||
by an "ifmain" guard (``if __name__ == '__main__'``) or you must use
|
protected by an "ifmain" guard (``if __name__ == '__main__'``). If you do
|
||||||
``ocrmypdf.ocr(...use_threads=True)``. If you do not take at least one
|
not take at least one of these steps, process semantics will prevent
|
||||||
of these steps, Windows process semantics will prevent OCRmyPDF from working
|
OCRmyPDF from working correctly.
|
||||||
correctly.
|
|
||||||
|
|
||||||
Logging
|
Logging
|
||||||
-------
|
-------
|
||||||
|
|||||||
@@ -5,6 +5,15 @@ API Reference
|
|||||||
This page summarizes the rest of the public API. Generally speaking this
|
This page summarizes the rest of the public API. Generally speaking this
|
||||||
should mainly of interest to plugin developers.
|
should mainly of interest to plugin developers.
|
||||||
|
|
||||||
|
ocrmypdf
|
||||||
|
========
|
||||||
|
|
||||||
|
.. autoclass:: ocrmypdf.PageContext
|
||||||
|
:members:
|
||||||
|
|
||||||
|
.. autoclass:: ocrmypdf.PdfContext
|
||||||
|
:members:
|
||||||
|
|
||||||
ocrmypdf.exceptions
|
ocrmypdf.exceptions
|
||||||
===================
|
===================
|
||||||
|
|
||||||
@@ -17,6 +26,9 @@ ocrmypdf.helpers
|
|||||||
|
|
||||||
.. automodule:: ocrmypdf.helpers
|
.. automodule:: ocrmypdf.helpers
|
||||||
:members:
|
:members:
|
||||||
|
:noindex: deprecated
|
||||||
|
|
||||||
|
.. autodecorator:: deprecated
|
||||||
|
|
||||||
ocrmypdf.hocrtransform
|
ocrmypdf.hocrtransform
|
||||||
======================
|
======================
|
||||||
|
|||||||
+2
-9
@@ -111,7 +111,7 @@ Users may need to customize the script to meet their requirements.
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip3 install -r requirements/watcher.txt
|
pip3 install ocrmypdf[watcher]
|
||||||
|
|
||||||
env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \
|
env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \
|
||||||
OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \
|
OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \
|
||||||
@@ -127,7 +127,7 @@ Users may need to customize the script to meet their requirements.
|
|||||||
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
|
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
|
||||||
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||||
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
||||||
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={"rotate_pages": true}'``.
|
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
|
||||||
"OCR_POLL_NEW_FILE_SECONDS", "Polling interval"
|
"OCR_POLL_NEW_FILE_SECONDS", "Polling interval"
|
||||||
"OCR_LOGLEVEL", "Level of log messages to report"
|
"OCR_LOGLEVEL", "Level of log messages to report"
|
||||||
|
|
||||||
@@ -202,13 +202,6 @@ Alternatives
|
|||||||
- `Watchman <https://facebook.github.io/watchman/>`__ is a more
|
- `Watchman <https://facebook.github.io/watchman/>`__ is a more
|
||||||
powerful alternative to ``watchmedo``.
|
powerful alternative to ``watchmedo``.
|
||||||
|
|
||||||
AWS Lambda is not viable
|
|
||||||
------------------------
|
|
||||||
|
|
||||||
AWS Lambda and its equivalents have low limits on execution time and payload
|
|
||||||
size, relative to OCRmyPDF's needs. As of this writing, the request/response
|
|
||||||
payload for AWS Lambda was 6 MB, which means many PDFs will not fit.
|
|
||||||
|
|
||||||
macOS Automator
|
macOS Automator
|
||||||
===============
|
===============
|
||||||
|
|
||||||
|
|||||||
+11
-1
@@ -32,7 +32,8 @@ If you are proposing a change that will require a new Python dependency, we
|
|||||||
prefer dependencies that are already packaged by Debian or Red Hat. This makes
|
prefer dependencies that are already packaged by Debian or Red Hat. This makes
|
||||||
life much easier for our downstream package maintainers.
|
life much easier for our downstream package maintainers.
|
||||||
|
|
||||||
Python dependencies must also be GPLv3 compatible.
|
Python dependencies must also be license-compatible. GPLv3 or AGPLv3 are likely
|
||||||
|
incompatible with the project's license, but LGPLv3 is compatible.
|
||||||
|
|
||||||
New non-Python dependencies
|
New non-Python dependencies
|
||||||
===========================
|
===========================
|
||||||
@@ -55,3 +56,12 @@ of that platform.
|
|||||||
|
|
||||||
Packager maintainers, please ensure that the command line completion scripts in
|
Packager maintainers, please ensure that the command line completion scripts in
|
||||||
``misc/`` are installed.
|
``misc/`` are installed.
|
||||||
|
|
||||||
|
Copyright and license
|
||||||
|
=====================
|
||||||
|
|
||||||
|
For contributions over 10 lines of code, please include your name to list of
|
||||||
|
copyright holders for that file. The core program is licensed under MPL-2.0,
|
||||||
|
test files and documentation under CC-BY-SA 4.0, and miscellaneous files under
|
||||||
|
MIT. Please contribute code only that you wrote and you have the permission to
|
||||||
|
contribute or license to us.
|
||||||
|
|||||||
+5
-3
@@ -58,9 +58,11 @@ portrait pages.
|
|||||||
You can increase (decrease) the parameter ``--rotate-pages-threshold``
|
You can increase (decrease) the parameter ``--rotate-pages-threshold``
|
||||||
to make page rotation more (less) aggressive. The threshold number is the ratio
|
to make page rotation more (less) aggressive. The threshold number is the ratio
|
||||||
of how confidence the OCR engine is that the document image should be changed,
|
of how confidence the OCR engine is that the document image should be changed,
|
||||||
compared to kept the same. A value of ``15.0`` is the default, and is fairly
|
compared to kept the same. The default value is quite conservative; on some files
|
||||||
conservative. A value of ``2.0`` will produce more rotations, and more false
|
it may not attempt rotations at all unless it is very confident that the current
|
||||||
positives.
|
rotation is wrong. A lower value of ``2.0`` will produce more rotations, and
|
||||||
|
more false positives. Run with ``-v1`` to see the confidence level for each
|
||||||
|
page to see if there may be a better value for your files.
|
||||||
|
|
||||||
If the page is "just a little off horizontal", like a crooked picture,
|
If the page is "just a little off horizontal", like a crooked picture,
|
||||||
then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal
|
then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal
|
||||||
|
|||||||
+32
-6
@@ -1,3 +1,5 @@
|
|||||||
|
.. _docker:
|
||||||
|
|
||||||
=====================
|
=====================
|
||||||
OCRmyPDF Docker image
|
OCRmyPDF Docker image
|
||||||
=====================
|
=====================
|
||||||
@@ -101,16 +103,41 @@ Adding languages to the Docker image
|
|||||||
By default the Docker image includes English, German, Simplified Chinese,
|
By default the Docker image includes English, German, Simplified Chinese,
|
||||||
French, Portuguese and Spanish, the most popular languages for OCRmyPDF
|
French, Portuguese and Spanish, the most popular languages for OCRmyPDF
|
||||||
users based on feedback. You may add other languages by creating a new
|
users based on feedback. You may add other languages by creating a new
|
||||||
Dockerfile based on the public one:
|
Dockerfile based on the public one.
|
||||||
|
|
||||||
.. code-block:: dockerfile
|
.. code-block:: dockerfile
|
||||||
|
|
||||||
FROM jbarlow83/ocrmypdf
|
FROM jbarlow83/ocrmypdf
|
||||||
|
|
||||||
# Add French
|
# Example: add Italian
|
||||||
RUN apt install tesseract-ocr-fra
|
RUN apt install tesseract-ocr-ita
|
||||||
|
|
||||||
You can also copy training data to ``/usr/share/tesseract-ocr/<tesseract version>/tessdata``.
|
To install language packs (training data) such as the
|
||||||
|
`tessdata_best <https://github.com/tesseract-ocr/tessdata_best>`_ suite or
|
||||||
|
custom data, you first need to determine the version of Tesseract data files, which
|
||||||
|
may differ from the Tesseract program version. Use this command to determine the data
|
||||||
|
file version:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
docker run -i --rm --entrypoint /bin/ls jbarlow83/ocrmypdf /usr/share/tesseract-ocr
|
||||||
|
|
||||||
|
As of 2021, the data file version is probably ``4.00``.
|
||||||
|
|
||||||
|
You can then add new data with either a Dockerfile:
|
||||||
|
|
||||||
|
.. code-block:: dockerfile
|
||||||
|
|
||||||
|
FROM jbarlow83/ocrmypdf
|
||||||
|
|
||||||
|
# Example: add a tessdata_best file
|
||||||
|
COPY chi_tra_vert.traineddata /usr/share/tesseract-ocr/<data version>/tessdata/
|
||||||
|
|
||||||
|
Alternately, you can copy training data into a Docker container as follows:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
docker cp mycustomtraining.traineddata name_of_container:/usr/share/tesseract-ocr/<tesseract version>/tessdata/
|
||||||
|
|
||||||
Executing the test suite
|
Executing the test suite
|
||||||
========================
|
========================
|
||||||
@@ -163,8 +190,7 @@ complete. This may entail setting a long timeout; this interface is more
|
|||||||
useful for internal HTTP API calls.
|
useful for internal HTTP API calls.
|
||||||
|
|
||||||
Unlike the rest of OCRmyPDF, this web service is licensed under the
|
Unlike the rest of OCRmyPDF, this web service is licensed under the
|
||||||
Affero GPLv3 (AGPLv3) since Ghostscript, a dependency of OCRmyPDF, is
|
Affero GPLv3 (AGPLv3) since Ghostscript is also licensed in this way.
|
||||||
also licensed in this way.
|
|
||||||
|
|
||||||
In addition to the above, please read our
|
In addition to the above, please read our
|
||||||
:ref:`general remarks on using OCRmyPDF as a service <ocr-service>`.
|
:ref:`general remarks on using OCRmyPDF as a service <ocr-service>`.
|
||||||
|
|||||||
+1
-1
@@ -1,7 +1,7 @@
|
|||||||
OCRmyPDF documentation
|
OCRmyPDF documentation
|
||||||
======================
|
======================
|
||||||
|
|
||||||
OCRmyPDF adds an optical charcter recognition (OCR) text layer to scanned PDF
|
OCRmyPDF adds an optical character recognition (OCR) text layer to scanned PDF
|
||||||
files, allowing them to be searched.
|
files, allowing them to be searched.
|
||||||
|
|
||||||
PDF is the best format for storing and exchanging scanned documents.
|
PDF is the best format for storing and exchanging scanned documents.
|
||||||
|
|||||||
+80
-49
@@ -12,19 +12,21 @@ system/platform. This version may be out of date, however.
|
|||||||
|
|
||||||
These platforms have one-liner installs:
|
These platforms have one-liner installs:
|
||||||
|
|
||||||
+-----------------------------+-------------------------------+
|
+-------------------------------+-------------------------------+
|
||||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||||
+-----------------------------+-------------------------------+
|
+-------------------------------+-------------------------------+
|
||||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||||
+-----------------------------+-------------------------------+
|
+-------------------------------+-------------------------------+
|
||||||
| Fedora | ``dnf install ocrmypdf`` |
|
| Fedora | ``dnf install ocrmypdf`` |
|
||||||
+-----------------------------+-------------------------------+
|
+-------------------------------+-------------------------------+
|
||||||
| macOS | ``brew install ocrmypdf`` |
|
| macOS | ``brew install ocrmypdf`` |
|
||||||
+-----------------------------+-------------------------------+
|
+-------------------------------+-------------------------------+
|
||||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
+-----------------------------+-------------------------------+
|
+-------------------------------+-------------------------------+
|
||||||
| FreeBSD | ``pkg install py37-ocrmypdf`` |
|
| FreeBSD | ``pkg install py37-ocrmypdf`` |
|
||||||
+-----------------------------+-------------------------------+
|
+-------------------------------+-------------------------------+
|
||||||
|
| Conda (WSL, macOS, Linux) | ``conda install ocrmypdf`` |
|
||||||
|
+-------------------------------+-------------------------------+
|
||||||
|
|
||||||
More detailed procedures are outlined below. If you want to do a manual
|
More detailed procedures are outlined below. If you want to do a manual
|
||||||
install, or install a more recent version than your platform provides, read on.
|
install, or install a more recent version than your platform provides, read on.
|
||||||
@@ -54,6 +56,9 @@ Debian and Ubuntu 18.04 or newer
|
|||||||
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
||||||
:alt: Ubuntu 20.04 LTS
|
:alt: Ubuntu 20.04 LTS
|
||||||
|
|
||||||
|
.. |ubu-2010| image:: https://repology.org/badge/version-for-repo/ubuntu_20_10/ocrmypdf.svg
|
||||||
|
:alt: Ubuntu 20.10
|
||||||
|
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| **OCRmyPDF versions in Debian & Ubuntu** |
|
| **OCRmyPDF versions in Debian & Ubuntu** |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
@@ -61,7 +66,7 @@ Debian and Ubuntu 18.04 or newer
|
|||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |deb-stable| |deb-testing| |deb-unstable| |
|
| |deb-stable| |deb-testing| |deb-unstable| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |ubu-1804| |ubu-2004| |
|
| |ubu-1804| |ubu-2004| |ubu-2010| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
|
|
||||||
Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users
|
Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users
|
||||||
@@ -90,15 +95,15 @@ For full details on version availability for your platform, check the
|
|||||||
automatically detect it (specifically the ``jbig2`` binary) on the
|
automatically detect it (specifically the ``jbig2`` binary) on the
|
||||||
``PATH``. To add JBIG2 encoding, see :ref:`jbig2`.
|
``PATH``. To add JBIG2 encoding, see :ref:`jbig2`.
|
||||||
|
|
||||||
Fedora 29 or newer
|
Fedora
|
||||||
------------------
|
------
|
||||||
|
|
||||||
.. |fedora-31| image:: https://repology.org/badge/version-for-repo/fedora_31/ocrmypdf.svg
|
|
||||||
:alt: Fedora 31
|
|
||||||
|
|
||||||
.. |fedora-32| image:: https://repology.org/badge/version-for-repo/fedora_32/ocrmypdf.svg
|
.. |fedora-32| image:: https://repology.org/badge/version-for-repo/fedora_32/ocrmypdf.svg
|
||||||
:alt: Fedora 32
|
:alt: Fedora 32
|
||||||
|
|
||||||
|
.. |fedora-33| image:: https://repology.org/badge/version-for-repo/fedora_33/ocrmypdf.svg
|
||||||
|
:alt: Fedora 33
|
||||||
|
|
||||||
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
||||||
:alt: Fedore Rawhide
|
:alt: Fedore Rawhide
|
||||||
|
|
||||||
@@ -107,7 +112,7 @@ Fedora 29 or newer
|
|||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |latest| |
|
| |latest| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |fedora-31| |fedora-32| |fedora-rawhide| |
|
| |fedora-32| |fedora-33| |fedora-rawhide| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
|
|
||||||
Users of Fedora 29 or later may simply
|
Users of Fedora 29 or later may simply
|
||||||
@@ -158,7 +163,7 @@ To install ocrmypdf for the system:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
sudo pip3 install ocrmypdf
|
pip3 install ocrmypdf
|
||||||
|
|
||||||
To install for the current user only:
|
To install for the current user only:
|
||||||
|
|
||||||
@@ -190,7 +195,8 @@ of ocrmypdf, and install the following dependencies:
|
|||||||
python3-reportlab \
|
python3-reportlab \
|
||||||
qpdf \
|
qpdf \
|
||||||
tesseract-ocr \
|
tesseract-ocr \
|
||||||
zlib1g
|
zlib1g \
|
||||||
|
unpaper
|
||||||
|
|
||||||
We will need a newer version of ``pip`` then was available for Ubuntu 18.04:
|
We will need a newer version of ``pip`` then was available for Ubuntu 18.04:
|
||||||
|
|
||||||
@@ -354,7 +360,8 @@ To install OCRmyPDF for Alpine Linux:
|
|||||||
Mageia 7
|
Mageia 7
|
||||||
--------
|
--------
|
||||||
|
|
||||||
Install the following dependencies:
|
There is no OS-level packaging available for Mageia, so you must install the
|
||||||
|
dependencies:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -379,12 +386,16 @@ Install the following dependencies:
|
|||||||
|
|
||||||
To install ocrmypdf for the system:
|
To install ocrmypdf for the system:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
# As root user
|
# As root user
|
||||||
pip3 install ocrmypdf
|
pip3 install ocrmypdf
|
||||||
ldconfig
|
ldconfig
|
||||||
|
|
||||||
Or, to install for the current user only:
|
Or, to install for the current user only:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
export PATH=$HOME/.local/bin:$PATH
|
export PATH=$HOME/.local/bin:$PATH
|
||||||
pip3 install --user ocrmypdf
|
pip3 install --user ocrmypdf
|
||||||
|
|
||||||
@@ -453,7 +464,7 @@ Update Homebrew:
|
|||||||
|
|
||||||
Install or upgrade the required Homebrew packages, if any are missing.
|
Install or upgrade the required Homebrew packages, if any are missing.
|
||||||
To do this, use ``brew edit ocrmypdf`` to obtain a recent list of Homebrew
|
To do this, use ``brew edit ocrmypdf`` to obtain a recent list of Homebrew
|
||||||
dependencies. You could also check the ``azure-pipelines.yml``.
|
dependencies. You could also check the ``.workflows/build.yml``.
|
||||||
|
|
||||||
This will include the English, French, German and Spanish language
|
This will include the English, French, German and Spanish language
|
||||||
packs. If you need other languages you can optionally install them all:
|
packs. If you need other languages you can optionally install them all:
|
||||||
@@ -494,10 +505,6 @@ Installing on Windows
|
|||||||
Native Windows
|
Native Windows
|
||||||
--------------
|
--------------
|
||||||
|
|
||||||
.. note::
|
|
||||||
|
|
||||||
It is easier to install OCRmyPDF on Windows Subsystem for Linux.
|
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
|
|
||||||
Administrator privileges will be required for some of these steps.
|
Administrator privileges will be required for some of these steps.
|
||||||
@@ -508,31 +515,41 @@ You must install the following for Windows:
|
|||||||
* Tesseract 4.0 or later
|
* Tesseract 4.0 or later
|
||||||
* Ghostscript 9.50 or later
|
* Ghostscript 9.50 or later
|
||||||
|
|
||||||
You can install these with the Chocolatey package manager:
|
Using the `Chocolatey <https://chocolatey.org/>`_ package manager, install the
|
||||||
|
following when running in an Administrator command prompt:
|
||||||
|
|
||||||
* ``choco install python3``
|
* ``choco install python3``
|
||||||
* ``choco install --pre tesseract``
|
* ``choco install --pre tesseract``
|
||||||
* ``choco install ghostscript``
|
* ``choco install ghostscript``
|
||||||
|
* ``choco install pngquant`` (optional)
|
||||||
|
|
||||||
Also consider adding:
|
The commands above will install Python 3.x (latest version), Tesseract, Ghostscript
|
||||||
|
and pngquant. Chocolatey may also need to install the Windows Visual C++ Runtime
|
||||||
|
DLLs or other Windows patches, and may require a reboot.
|
||||||
|
|
||||||
* ``choco install pngquant``
|
You may then use ``pip`` to install ocrmypdf. (This can performed by a user or
|
||||||
|
Administrator.):
|
||||||
Windows 10 64-bit and 64-bit versions of applications are recommended. Earlier
|
|
||||||
versions of Windows and 32-bit versions of these programs are not tested, and not
|
|
||||||
supported at this time.
|
|
||||||
|
|
||||||
OCRmyPDF will check for Tesseract-OCR and Ghostscript in your Program Files folder.
|
|
||||||
If they are in some other location, you may need to modify the ``PATH``
|
|
||||||
environment variable so Tesseract, Ghostscript, and other any optional executables can
|
|
||||||
be found. You can enter it in the command line or
|
|
||||||
`follow these directions <https://www.computerhope.com/issues/ch000549.htm#dospath>`_
|
|
||||||
to make the change persistent and system-wide.
|
|
||||||
|
|
||||||
You may then use pip to install ocrmypdf:
|
|
||||||
|
|
||||||
* ``pip install ocrmypdf``
|
* ``pip install ocrmypdf``
|
||||||
|
|
||||||
|
Chocolatey automatically selects appropriate versions of these applications. If you
|
||||||
|
are installing them manually, please install 64-bit versions of all applications for
|
||||||
|
64-bit Windows, or 32-bit versions of all applications for 32-bit Windows. Mixing
|
||||||
|
the "bitness" of these programs will lead to errors.
|
||||||
|
|
||||||
|
OCRmyPDF will check the Windows Registry and standard locations in your Program Files
|
||||||
|
for third party software it needs (specifically, Tesseract and Ghostscript). To
|
||||||
|
override the versions OCRmyPDF selects, you can modify the ``PATH`` environment
|
||||||
|
variable. `Follow these directions <https://www.computerhope.com/issues/ch000549.htm#dospath>`_
|
||||||
|
to change the PATH.
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
|
||||||
|
As of early 2021, users have reported problems with the Microsoft Store version of
|
||||||
|
Python and OCRmyPDF. These issues affect many other third party Python packages.
|
||||||
|
Please download Python from Python.org or Chocolatey instead, and do not use the
|
||||||
|
Microsoft Store version.
|
||||||
|
|
||||||
Windows Subsystem for Linux
|
Windows Subsystem for Linux
|
||||||
---------------------------
|
---------------------------
|
||||||
|
|
||||||
@@ -603,7 +620,7 @@ However, the OCR-to-text-layer functionality is available.
|
|||||||
Docker
|
Docker
|
||||||
------
|
------
|
||||||
|
|
||||||
You can also :ref:`Install the Docker <docker-install>` container on Windows. Ensure that
|
You can also :ref:`Install the Docker <docker>` container on Windows. Ensure that
|
||||||
your command prompt can run the docker "hello world" container.
|
your command prompt can run the docker "hello world" container.
|
||||||
|
|
||||||
Installing on FreeBSD
|
Installing on FreeBSD
|
||||||
@@ -629,7 +646,7 @@ Installing the Docker image
|
|||||||
For some users, installing the Docker image will be easier than
|
For some users, installing the Docker image will be easier than
|
||||||
installing all of OCRmyPDF's dependencies.
|
installing all of OCRmyPDF's dependencies.
|
||||||
|
|
||||||
See `OCRmyPDF Docker Image <docker>`__ for more information.
|
See :ref:`docker` for more information.
|
||||||
|
|
||||||
Installing with Python pip
|
Installing with Python pip
|
||||||
==========================
|
==========================
|
||||||
@@ -637,7 +654,22 @@ Installing with Python pip
|
|||||||
OCRmyPDF is delivered by PyPI because it is a convenient way to install
|
OCRmyPDF is delivered by PyPI because it is a convenient way to install
|
||||||
the latest version. However, PyPI and ``pip`` cannot address the fact
|
the latest version. However, PyPI and ``pip`` cannot address the fact
|
||||||
that ``ocrmypdf`` depends on certain non-Python system libraries and
|
that ``ocrmypdf`` depends on certain non-Python system libraries and
|
||||||
programs being instsalled.
|
programs being installed.
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
|
||||||
|
Debian and Ubuntu users: unfortunately, Debian and Ubuntu customize
|
||||||
|
Python in non-standard ways, and the nature of these customizations
|
||||||
|
varies from release to release. This can make for a frustrating
|
||||||
|
user experience. The instructions below work on almost all platforms that
|
||||||
|
have Python installed, except for Debian and Ubuntu, where you may need
|
||||||
|
to take additional steps. For best results on Debian and Ubuntu, use the
|
||||||
|
``apt`` packages; or if these are too old, run
|
||||||
|
``apt install python3-pip python3-venv``, create a virtual environment,
|
||||||
|
and install OCRmyPDF in that environment.
|
||||||
|
|
||||||
|
`See here for more inforation on Debian-Python issues
|
||||||
|
<https://gist.github.com/tiran/2dec9e03c6f901814f6d1e8dad09528e>`__.
|
||||||
|
|
||||||
For best results, first install `your platform's
|
For best results, first install `your platform's
|
||||||
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
|
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
|
||||||
@@ -780,8 +812,7 @@ To install all of the development and test requirements:
|
|||||||
python3 -m venv
|
python3 -m venv
|
||||||
source venv/bin/activate
|
source venv/bin/activate
|
||||||
cd OCRmyPDF
|
cd OCRmyPDF
|
||||||
pip install -e .
|
pip install -e .[test]
|
||||||
pip install -r requirements/dev.txt -r requirements/test.txt
|
|
||||||
|
|
||||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||||
|
|
||||||
|
|||||||
@@ -139,7 +139,7 @@ Limitations
|
|||||||
OCRmyPDF is limited by the Tesseract OCR engine. As such it experiences
|
OCRmyPDF is limited by the Tesseract OCR engine. As such it experiences
|
||||||
these limitations, as do any other programs that rely on Tesseract:
|
these limitations, as do any other programs that rely on Tesseract:
|
||||||
|
|
||||||
- The OCR is not as accurate as commercial solutions such as Abbyy.
|
- The OCR is not as accurate as commercial OCR solutions.
|
||||||
- It is not capable of recognizing handwriting.
|
- It is not capable of recognizing handwriting.
|
||||||
- It may find gibberish and report this as OCR output.
|
- It may find gibberish and report this as OCR output.
|
||||||
- If a document contains languages outside of those given in the
|
- If a document contains languages outside of those given in the
|
||||||
@@ -186,6 +186,11 @@ Ghostscript also imposes some limitations:
|
|||||||
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
||||||
one of the standard XMP metadata namespaces for PDFs. In particular,
|
one of the standard XMP metadata namespaces for PDFs. In particular,
|
||||||
PRISM Metdata is removed.
|
PRISM Metdata is removed.
|
||||||
|
- Ghostscript's PDF/A conversion seems to remove or deactivate
|
||||||
|
hyperlinks and other active content.
|
||||||
|
|
||||||
|
You can use ``--output-type pdf`` to disable PDF/A conversion and produce
|
||||||
|
a standard, non-archival PDF.
|
||||||
|
|
||||||
Regarding OCRmyPDF itself:
|
Regarding OCRmyPDF itself:
|
||||||
|
|
||||||
|
|||||||
+10
-7
@@ -12,9 +12,9 @@ languages <https://github.com/tesseract-ocr/tesseract/blob/master/doc/tesseract.
|
|||||||
Languages are identified by standardized three-letter codes (called ISO 639-2 Alpha-3).
|
Languages are identified by standardized three-letter codes (called ISO 639-2 Alpha-3).
|
||||||
Tesseract's documentation also lists the three-letter code for your language.
|
Tesseract's documentation also lists the three-letter code for your language.
|
||||||
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
||||||
are not, e.g. German is ``deu``.
|
are not, e.g. German is ``deu`` and French is ``fra``.
|
||||||
|
|
||||||
After you have installed a language pack, you can use it ``ocrmypdf -l <language>``,
|
After you have installed a language pack, you can use it with ``ocrmypdf -l <language>``,
|
||||||
for example ``ocrmypdf -l spa``. For multilingual documents, you can specify
|
for example ``ocrmypdf -l spa``. For multilingual documents, you can specify
|
||||||
all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French.
|
all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French.
|
||||||
English is assumed by default unless other language(s) are specified.
|
English is assumed by default unless other language(s) are specified.
|
||||||
@@ -35,8 +35,8 @@ Debian and Ubuntu users
|
|||||||
|
|
||||||
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||||
to what languages it should search for. Multiple languages can be
|
to what languages it should search for. Multiple languages can be
|
||||||
requested using either ``-l eng+fre`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fre``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
Fedora users
|
Fedora users
|
||||||
============
|
============
|
||||||
@@ -51,8 +51,8 @@ Fedora users
|
|||||||
|
|
||||||
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||||
to what languages it should search for. Multiple languages can be
|
to what languages it should search for. Multiple languages can be
|
||||||
requested using either ``-l eng+fre`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fre``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
macOS users
|
macOS users
|
||||||
===========
|
===========
|
||||||
@@ -70,4 +70,7 @@ derived Docker image as
|
|||||||
Windows users
|
Windows users
|
||||||
=============
|
=============
|
||||||
|
|
||||||
The Tesseract installer provided by Chocolatey already includes 100 languages.
|
The Tesseract installer provided by Chocolatey currently includes only English language.
|
||||||
|
To install other languages, download the respective language pack (``.traineddata`` file)
|
||||||
|
from https://github.com/tesseract-ocr/tessdata/ and place it in
|
||||||
|
``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed).
|
||||||
|
|||||||
@@ -64,7 +64,7 @@ malicious user could upload a chosen PDF. In particular, it is not
|
|||||||
necessarily secure against PDF malware or PDFs that cause denial of
|
necessarily secure against PDF malware or PDFs that cause denial of
|
||||||
service. OCRmyPDF relies on Ghostscript, and therefore, if deployed
|
service. OCRmyPDF relies on Ghostscript, and therefore, if deployed
|
||||||
online one should be prepared to comply with Ghostscript's Affero GPL
|
online one should be prepared to comply with Ghostscript's Affero GPL
|
||||||
license, OCRmyPDF's GPL license, and any other licenses.
|
license, and any other licenses.
|
||||||
|
|
||||||
Setting aside these concerns, a side effect of OCRmyPDF is it may
|
Setting aside these concerns, a side effect of OCRmyPDF is it may
|
||||||
incidentally sanitize PDFs that contain certain types of malware. It
|
incidentally sanitize PDFs that contain certain types of malware. It
|
||||||
@@ -128,8 +128,9 @@ Commercial alternatives
|
|||||||
The author also provides professional services that include OCR and
|
The author also provides professional services that include OCR and
|
||||||
building databases around PDFs, and is happy to provide consultation.
|
building databases around PDFs, and is happy to provide consultation.
|
||||||
|
|
||||||
Abbyy Cloud OCR is a viable commercial alternative with a web services
|
Abbyy Cloud OCR is viable commercial alternative with a web services
|
||||||
API.
|
API. Amazon Textract, Google Cloud Vision, and Microsoft Azure
|
||||||
|
Computer Vision provide advanced OCR but have less PDF rendering capability.
|
||||||
|
|
||||||
Password protection, digital signatures and certification
|
Password protection, digital signatures and certification
|
||||||
=========================================================
|
=========================================================
|
||||||
|
|||||||
@@ -65,6 +65,27 @@ similar to ``pytest`` packages such as ``pytest-cov`` (the package) and
|
|||||||
``ocrmypdf-`` (for the package name on PyPI) and ``ocrmypdf_`` (for the
|
``ocrmypdf-`` (for the package name on PyPI) and ``ocrmypdf_`` (for the
|
||||||
module), just like pytest plugins.
|
module), just like pytest plugins.
|
||||||
|
|
||||||
|
Setuptools plugins
|
||||||
|
==================
|
||||||
|
|
||||||
|
You can also create a plugin that OCRmyPDF will always automatically load if both are
|
||||||
|
installed in the same virtual environment, using a setuptools entrypoint.
|
||||||
|
|
||||||
|
Your package's ``setup.py`` would need to contain the following, for a plugin
|
||||||
|
named ``ocrmypdf-exampleplugin``:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
# sample ./setup.py file
|
||||||
|
from setuptools import setup
|
||||||
|
|
||||||
|
setup(
|
||||||
|
name="ocrmypdf-exampleplugin",
|
||||||
|
packages=["exampleplugin"],
|
||||||
|
# the following makes a plugin available to pytest
|
||||||
|
entry_points={"ocrmypdf": ["exampleplugin = exampleplugin.pluginmodule"]},
|
||||||
|
)
|
||||||
|
|
||||||
Plugin requirements
|
Plugin requirements
|
||||||
===================
|
===================
|
||||||
|
|
||||||
@@ -130,6 +151,18 @@ Custom command line arguments
|
|||||||
|
|
||||||
.. autofunction:: ocrmypdf.pluginspec.check_options
|
.. autofunction:: ocrmypdf.pluginspec.check_options
|
||||||
|
|
||||||
|
Execution and progress reporting
|
||||||
|
--------------------------------
|
||||||
|
|
||||||
|
.. autoclass: ocrmypdf.pluginspec.Executor
|
||||||
|
:members:
|
||||||
|
|
||||||
|
.. autofunction:: ocrmypdf.pluginspec.get_logging_console
|
||||||
|
|
||||||
|
.. autofunction:: ocrmypdf.pluginspec.get_executor
|
||||||
|
|
||||||
|
.. autofunction:: ocrmypdf.pluginspec.get_progressbar_class
|
||||||
|
|
||||||
Applying special behavior before processing
|
Applying special behavior before processing
|
||||||
-------------------------------------------
|
-------------------------------------------
|
||||||
|
|
||||||
@@ -147,6 +180,8 @@ Modifying intermediate images
|
|||||||
|
|
||||||
.. autofunction:: ocrmypdf.pluginspec.filter_page_image
|
.. autofunction:: ocrmypdf.pluginspec.filter_page_image
|
||||||
|
|
||||||
|
.. autofunction:: ocrmypdf.pluginspec.filter_pdf_page
|
||||||
|
|
||||||
OCR engine
|
OCR engine
|
||||||
----------
|
----------
|
||||||
|
|
||||||
|
|||||||
+367
-7
@@ -12,14 +12,374 @@ may be unreliable. Use the API to depend on precise behavior.
|
|||||||
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||||
wish to use some of its features for working with PDFs.
|
wish to use some of its features for working with PDFs.
|
||||||
|
|
||||||
Note that it is licensed under GPLv3, so scripts that
|
v12.1.0
|
||||||
``import ocrmypdf`` and are released publicly should probably also be
|
=======
|
||||||
licensed under GPLv3.
|
|
||||||
|
- For security reasons we now require Pillow >= 8.2.x. (Older versions will continue
|
||||||
|
to work if upgrading is not an option.)
|
||||||
|
- The build system was reorganized to rely on ``setup.cfg`` instead of ``setup.py``.
|
||||||
|
All changes should work with previously supported versions of setuptools.
|
||||||
|
- The files in ``requirements/*`` are now considered deprecated but will be retained for v12.
|
||||||
|
Instead use ``pip install ocrmypdf[test]`` instead of ``requirements/test.txt``, etc.
|
||||||
|
These files will be removed in v13.
|
||||||
|
|
||||||
|
v12.0.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Expand the list of languages supported by the hocr PDF renderer.
|
||||||
|
Several languages were previously considered not supported, particularly those
|
||||||
|
non-European languages that use the Latin alphabet.
|
||||||
|
- Fixed a case where the exception stack trace was suppressed in verbose mode.
|
||||||
|
- Improved documentation around commercial OCR.
|
||||||
|
|
||||||
|
v12.0.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fix exception thrown when using ``--remove-background`` on files containing small
|
||||||
|
images (#769).
|
||||||
|
- Improve documentation for description of adding language packs to the Docker image
|
||||||
|
and corrected name of French language pack.
|
||||||
|
|
||||||
|
v12.0.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fix "invalid version number" for untagged tesseract versions (#770).
|
||||||
|
|
||||||
|
v12.0.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
**Breaking changes**
|
||||||
|
|
||||||
|
- Due to recent security issues in pikepdf, Pillow and reportlab, we now require
|
||||||
|
newer versions of these libraries and some of their dependencies. (If necessary,
|
||||||
|
package maintainers may override these versions at their discretion; lower
|
||||||
|
versions will often work.)
|
||||||
|
- We now use the "LeaveColorUnchanged" color conversion strategy when directing
|
||||||
|
Ghostscript to create a PDF/A. Generally this is faster than performing a
|
||||||
|
color conversion, which is not always necessary.
|
||||||
|
- OCR text is now packaged in a Form XObject. This makes it easier to isolate
|
||||||
|
OCR from other document content. However, some poorly implemented PDF text
|
||||||
|
extraction algorithms may fail to detect the text.
|
||||||
|
- Many API functions have stricter parameter checking or expect keyword arguments
|
||||||
|
were they previously did not.
|
||||||
|
- Some deprecated functions in ``ocrmypdf.optimize`` were removed.
|
||||||
|
- The ``ocrmypdf.leptonica`` module is now deprecated, due to difficulties with
|
||||||
|
the current strategy of ABI binding on newer platforms like Apple Silicon.
|
||||||
|
It will be removed and replaced, either by repackaging Leptonica as an
|
||||||
|
independent library using or using a different image processing library.
|
||||||
|
- Continuous integration moved to GitHub Actions.
|
||||||
|
- We no longer depend on ``pytest_helpers_namespace`` for testing.
|
||||||
|
|
||||||
|
**New features**
|
||||||
|
|
||||||
|
- New plugin hook: ``get_progressbar_class``, for progress reporting,
|
||||||
|
allowing developers to replace the standard console progress bar with some
|
||||||
|
other mechanism, such as updating a GUI progress bar.
|
||||||
|
- New plugin hook: ``get_executor``, for replacing the concurrency model.
|
||||||
|
This is primarily to support execution on AWS Lambda, which does not support
|
||||||
|
standard Python ``multiprocessing`` due to its lack of shared memory.
|
||||||
|
- New plugin hook: ``get_logging_console``, for replacing the standard
|
||||||
|
way OCRmyPDF outputs its messages.
|
||||||
|
- New plugin hook: ``filter_pdf_page``, for modifying individual PDF
|
||||||
|
pages produced by OCRmyPDF.
|
||||||
|
- OCRmyPDF now runs on nonstandard execution environments that do not have
|
||||||
|
interprocess semaphores, such as AWS Lambda and Android Termux. If the environment
|
||||||
|
does not have semaphores, OCRmyPDF will automatically select an alternate
|
||||||
|
process executor that does not use semaphores.
|
||||||
|
- Continuous integration moved to GitHub Actions.
|
||||||
|
- We now generate an ARM64-compatible Docker image alongside the x64 image.
|
||||||
|
Thanks to @andkrause for doing most of the work in a pull request several months
|
||||||
|
ago, which we were finally able to integrate now. Also thanks to @0x326 for
|
||||||
|
review comments.
|
||||||
|
|
||||||
|
**Fixes**
|
||||||
|
|
||||||
|
- Fixed a possible deadlock on attempting to flush ``sys.stderr`` when older
|
||||||
|
versions of Leptonica are in use.
|
||||||
|
- Some worker processes inherited resources from their parents such as log
|
||||||
|
handlers that may have also lead to deadlocks. These resources are now released.
|
||||||
|
- Improvements to test coverage.
|
||||||
|
- Removed vestiges of support for Tesseract versions older than 4.0.0-beta1 (
|
||||||
|
which ships with Ubuntu 18.04).
|
||||||
|
- OCRmyPDF can now parse all of Tesseract version numbers, since several
|
||||||
|
schemes have been in use.
|
||||||
|
- Fixed an issue with parsing PDFs that contain images drawn at a scale of 0. (#761)
|
||||||
|
- Removed a frequently repeated message about disabling mmap.
|
||||||
|
|
||||||
|
v11.7.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Exclude CCITT Group 3 images from being optimized. Some libraries
|
||||||
|
OCRmyPDF uses do not seem to handle this obscure compression format properly.
|
||||||
|
You may get errors or possible corrupted output images without this fix.
|
||||||
|
|
||||||
|
v11.7.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Updated pinned versions in main.txt, primarily to upgrade Pillow to 8.1.2, due
|
||||||
|
to recently disclosed security vulnerabilities in that software.
|
||||||
|
- The ``--sidecar`` parameter now causes an exception if set to the same file as
|
||||||
|
the input or output PDF.
|
||||||
|
|
||||||
|
v11.7.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Some exceptions while attempting image optimization were only logged at the debug
|
||||||
|
level, causing them to be suppressed. These errors are now logged appropriately.
|
||||||
|
- Improved the error message related to ``--unpaper-args``.
|
||||||
|
- Updated documentation to mention the new conda distribution.
|
||||||
|
|
||||||
|
v11.7.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- We now support using ``--sidecar`` in conjunction with ``--pages``; these arguments
|
||||||
|
used to be mutually exclusive. (#735)
|
||||||
|
- Fixed a possible issue with PDF/A-1b generation. Acrobat complained that our PDFs use
|
||||||
|
object streams. More robust PDF/A validators like veraPDF don't consider this a
|
||||||
|
problem, but we'll honor Acrobat's objection from here on. This may increase file
|
||||||
|
size of PDF/A-1b files. PDF/A-2b files will not be affected.
|
||||||
|
|
||||||
|
v11.6.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed a regression where the wrong page orientation would be produced when using
|
||||||
|
arguments such as ``--deskew --rotate-pages`` (#730).
|
||||||
|
|
||||||
|
v11.6.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an issue with attempting optimize unusually narrow-width images by excluding
|
||||||
|
these images from optimization (#732).
|
||||||
|
- Remove an obsolete compatibility shim for a version of pikepdf that is no longer
|
||||||
|
supported.
|
||||||
|
|
||||||
|
v11.6.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- OCRmyPDF will now automatically register plugins from the same virtual environment
|
||||||
|
with an appropriate setuptools entrypoint.
|
||||||
|
- Refactor the plugin manager to remove unnecessary complications and make plugin
|
||||||
|
registration more automatic.
|
||||||
|
- ``PageContext`` and ``PdfContext`` are now formally part of the API, as they
|
||||||
|
should have been, since they were part of ``ocrmypdf.pluginspec``.
|
||||||
|
|
||||||
|
v11.5.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an issue where the output page size might differ by a fractional amount
|
||||||
|
due to rounding, when ``--force-ocr`` was used and the page contained objects
|
||||||
|
with multiple resolutions.
|
||||||
|
- When determining the resolution at which to rasterize a page, we now consider
|
||||||
|
printed text on the page as requiring a higher resolution. This fixes issues
|
||||||
|
with certain pages being rendered with unacceptably low resolution text, but
|
||||||
|
may increase output file sizes in some workflows where low resolution text
|
||||||
|
is acceptable.
|
||||||
|
- Added a workaround to fix an exception that occurs when trying to
|
||||||
|
``import ocrmypdf.leptonica`` on Apple ARM silicon (or potentially, other
|
||||||
|
platforms that do not permit write+executable memory).
|
||||||
|
|
||||||
|
v11.4.5
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an issue where files may not be closed when the API is used.
|
||||||
|
- Improved ``setup.cfg`` with better settings for test coverage.
|
||||||
|
|
||||||
|
v11.4.4
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed ``AttributeError: 'NoneType' object has no attribute 'userunit'``, issue #700,
|
||||||
|
related to OCRmyPDF not properly forwarded an error message from pdfminer.six.
|
||||||
|
- Adjusted typing of some arguments.
|
||||||
|
- ``ocrmypdf.ocr`` now takes a ``threading.Lock`` for reasons outlined in the
|
||||||
|
documentation.
|
||||||
|
|
||||||
|
v11.4.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Removed a redundant debug message.
|
||||||
|
- Test suite now asserts that most patched functions are called when they should be.
|
||||||
|
- Test suite now skips a test that fails on two particular versions of piekpdf.
|
||||||
|
|
||||||
|
v11.4.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed support for Cygwin, hopefully.
|
||||||
|
- watcher.py: Fixed an issue with the OCR_LOGLEVEL not being interpreted.
|
||||||
|
|
||||||
|
v11.4.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an issue where invalid pages ranges passed using the ``pages`` argument,
|
||||||
|
such as "1-0" would cause unhandled exceptions.
|
||||||
|
- Accepted a user-contributed to the Synology demo script in misc/synology.py.
|
||||||
|
- Clarified documentation about change of temporary file location ``ocrmypdf.io``.
|
||||||
|
- Fixed Python wheel tag which was incorrectly set to py35 even though we long
|
||||||
|
since dropped support for Python 3.5.
|
||||||
|
|
||||||
|
v11.4.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- When looking for Tesseract and Ghostscript, we now check the Windows Registry to
|
||||||
|
see if their installers registered the location of their executables. This should
|
||||||
|
help Windows users who have installed these programs to non-standard
|
||||||
|
locations.
|
||||||
|
- We now report on the progress of PDF/A conversion, since this operation is
|
||||||
|
sometimes slow.
|
||||||
|
- Improved command line completions.
|
||||||
|
- The prefix of the temporary folder OCRmyPDF creates has been changed from
|
||||||
|
``com.github.ocrmypdf`` to ``ocrmypdf.io``. Scripts that chose to depend on this
|
||||||
|
prefix may need to be adjusted. (This has always been an implementation detail so is
|
||||||
|
not considered part of the semantic versioning "contract".)
|
||||||
|
- Fixed issue #692, where a particular file with malformed fonts would flood an
|
||||||
|
internal message cue by generating so many debug messages.
|
||||||
|
- Fixed an exception on processing hOCR files with no page record. Tesseract
|
||||||
|
is not known to generate such files.
|
||||||
|
|
||||||
|
v11.3.4
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an error message 'called readLinearizationData for file that is not
|
||||||
|
linearized' that may occur when pikepdf 2.1.0 is used. (Upgrading to pikepdf
|
||||||
|
2.1.1 also fixes the issue.)
|
||||||
|
- File watcher now automatically includes ``.PDF`` in addition to ``.pdf`` to
|
||||||
|
better support case sensitive file systems.
|
||||||
|
- Some documentation and comment improvements.
|
||||||
|
|
||||||
|
v11.3.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- If unpaper outputs non-UTF-8 data, quietly fix this rather than choke on the
|
||||||
|
conversion. (Possibly addresses #671.)
|
||||||
|
|
||||||
|
v11.3.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Explicitly require pikepdf 2.0.0 or newer when running on Python 3.9. (There are
|
||||||
|
concerns about the stability of pybind11 2.5.x with Python 3.9, which is used in
|
||||||
|
pikepdf 1.x.)
|
||||||
|
- Fixed another issue related to page rotation.
|
||||||
|
- Fixed an issue where image marked as image masks were not properly considered
|
||||||
|
as optimization candidates.
|
||||||
|
- On some systems, unpaper seems to be unable to process the PNGs we offer it
|
||||||
|
as input. We now convert the input to PNM format, which unpaper always accepts.
|
||||||
|
Fixes #665 and #667.
|
||||||
|
- DPI sent to unpaper is now rounded to a more reasonable number of decimal digits.
|
||||||
|
- Debug and error messages from unpaper were being suppressed.
|
||||||
|
- Some documentation tweaks.
|
||||||
|
|
||||||
|
v11.3.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Declare support for new versions: pdfminer.six 20201018 and pikepdf 2.x
|
||||||
|
- Fix warning related to ``--pdfa-image-compression`` that appears at the wrong
|
||||||
|
time.
|
||||||
|
|
||||||
|
v11.3.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- The "OCR" step is describing as "Image processing" in the output messages when
|
||||||
|
OCR is disabled, to better explain the application's behavior.
|
||||||
|
- Debug logs are now only created when run as a command line, and not when OCR
|
||||||
|
is performed for an API call. It is the calling application's responsibility
|
||||||
|
to set up logging.
|
||||||
|
- For PDFs with a low number of pages, we gathered information about the input PDF
|
||||||
|
in a thread rather than process (when there are more pages). When run as a
|
||||||
|
thread, we did not close the file handle to the working PDF, leaking one file
|
||||||
|
handle per call of ``ocrmypdf.ocr``.
|
||||||
|
- Fixed an issue where debug messages send by child worker processes did not match
|
||||||
|
the log settings of parent process, causing messages to be dropped. This affected
|
||||||
|
macOS and Windows only where the parent process is not forked.
|
||||||
|
- Fixed the hookspec of rasterize_pdf_page to remove default parameters that
|
||||||
|
were not handled in an expected way by pluggy.
|
||||||
|
- Fixed another issue with automatic page rotation (#658) due to the issue above.
|
||||||
|
|
||||||
|
v11.2.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an issue where optimization of a 1-bit image with a color palette or
|
||||||
|
associated ICC that was optimized to JBIG2 could have its colors inverted.
|
||||||
|
|
||||||
|
v11.2.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an issue with optimizing PNG-type images that had soft masks or image masks.
|
||||||
|
This is a regression introduced in (or about) v11.1.0.
|
||||||
|
- Improved type checking of the ``plugins`` parameter for the ``ocrmypdf.ocr``
|
||||||
|
API call.
|
||||||
|
|
||||||
|
v11.1.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed hOCR renderer writing the text in roughly reverse order. This should not
|
||||||
|
affect reasonably smart PDF readers that properly locate the position of all
|
||||||
|
text, but may confuse those that rely on the order of objects in the content
|
||||||
|
stream. (#642)
|
||||||
|
|
||||||
|
v11.1.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- We now avoid using named temporary files when using pngquant allowing containerized
|
||||||
|
pngquant installs to be used.
|
||||||
|
- Clarified an error message.
|
||||||
|
- Highest number of 1's in a release ever!
|
||||||
|
|
||||||
|
v11.1.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed page rotation issues: #634, #589.
|
||||||
|
- Fixed some cases where optimization created an invalid image such as a
|
||||||
|
1-bit "RGB" image: #629, #620.
|
||||||
|
- Page numbers are now displayed in debug logs when pages are being grafted.
|
||||||
|
- ocrmypdf.optimize.rewrite_png and ocrmypdf.optimize.rewrite_png_as_g4 were
|
||||||
|
marked deprecated. Strictly speaking these should have been internal APIs,
|
||||||
|
but they were never hidden.
|
||||||
|
- As a precaution, pikepdf mmap-based file access has been disabled due to a
|
||||||
|
rare race condition that causes a crash when certain objects are deallocated.
|
||||||
|
The problem is likely in pikepdf's dependency pybind11.
|
||||||
|
- Extended the example plugin to demonstrate conversion to mono.
|
||||||
|
|
||||||
|
v11.0.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed issue #612, TypeError exception. Fixed by eliminating unnecessary repair of
|
||||||
|
input PDF metadata in memory.
|
||||||
|
|
||||||
|
v11.0.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Blacklist pdfminer.six 20200720, which has a regression fixed in 20200726.
|
||||||
|
- Approve img2pdf 0.4 as it passes tests.
|
||||||
|
- Clarify that the GPL-3 portion of pdfa.py was removed with the changes in v11.0.0;
|
||||||
|
the debian/copyright file did not properly annotate this change.
|
||||||
|
|
||||||
|
v11.0.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Project license changed to Mozilla Public License 2.0. Some miscellaneous
|
||||||
|
code is now under MIT license and non-code content/media remains under
|
||||||
|
CC-BY-SA 4.0. License changed with approval of all people who were found
|
||||||
|
to have contributed to GPLv3 licensed sections of the project. (#600)
|
||||||
|
- Because the license changed, this is being treated as a major version number
|
||||||
|
change; however, there are no known breaking changes in functional behavior
|
||||||
|
or API compared to v10.x.
|
||||||
|
|
||||||
|
v10.3.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed a "KeyError: 'dpi'" error message when using ``--threshold`` on an image.
|
||||||
|
(#607)
|
||||||
|
|
||||||
|
v10.3.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed a case where we reported "no reason" for a file size increase, when we
|
||||||
|
could determine the reason.
|
||||||
|
- Enabled support for pdfminer.six 20200726.
|
||||||
|
|
||||||
v10.3.1
|
v10.3.1
|
||||||
=======
|
=======
|
||||||
|
|
||||||
- Fixed a number of test suite failures with pdfminer.six older than veresion 20200420.
|
- Fixed a number of test suite failures with pdfminer.six older than veresion 20200402.
|
||||||
- Enabled support for pdfminer.six 20200720.
|
- Enabled support for pdfminer.six 20200720.
|
||||||
|
|
||||||
v10.3.0
|
v10.3.0
|
||||||
@@ -362,7 +722,7 @@ v9.0.0
|
|||||||
|
|
||||||
- Added a high level API for applications that want to integrate OCRmyPDF.
|
- Added a high level API for applications that want to integrate OCRmyPDF.
|
||||||
Special thanks to Martin Wind (@mawi1988) whose made significant contributions
|
Special thanks to Martin Wind (@mawi1988) whose made significant contributions
|
||||||
to this effort. OCRmyPDF is GPLv3-licensed.
|
to this effort.
|
||||||
- Added progress bars for long-running steps. ■■■■■■■□□
|
- Added progress bars for long-running steps. ■■■■■■■□□
|
||||||
- We now create linearized ("fast web view") PDFs by default. The new parameter
|
- We now create linearized ("fast web view") PDFs by default. The new parameter
|
||||||
``--fast-web-view`` provides control over when this feature is applied.
|
``--fast-web-view`` provides control over when this feature is applied.
|
||||||
@@ -989,8 +1349,8 @@ v6.1.0
|
|||||||
v6.0.0
|
v6.0.0
|
||||||
======
|
======
|
||||||
|
|
||||||
- The software license has been changed to GPLv3. Test resource files
|
- The software license has been changed to GPLv3 [it has since changed again].
|
||||||
and some individual sources may have other licenses.
|
Test resource files and some individual sources may have other licenses.
|
||||||
- OCRmyPDF now depends on
|
- OCRmyPDF now depends on
|
||||||
`PyMuPDF <https://pymupdf.readthedocs.io/en/latest/installation/>`__.
|
`PyMuPDF <https://pymupdf.readthedocs.io/en/latest/installation/>`__.
|
||||||
Including PyMuPDF is the primary reason for the change to GPLv3.
|
Including PyMuPDF is the primary reason for the change to GPLv3.
|
||||||
|
|||||||
+19
-1
@@ -1,5 +1,23 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# Original version by DeliciousPickle@github; modified
|
# Copyright 2016 findingorder: https://github.com/findingorder
|
||||||
|
#
|
||||||
|
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
# of this software and associated documentation files (the "Software"), to deal
|
||||||
|
# in the Software without restriction, including without limitation the rights
|
||||||
|
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
# copies of the Software, and to permit persons to whom the Software is
|
||||||
|
# furnished to do so, subject to the following conditions:
|
||||||
|
#
|
||||||
|
# The above copyright notice and this permission notice shall be included in all
|
||||||
|
# copies or substantial portions of the Software.
|
||||||
|
#
|
||||||
|
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
# SOFTWARE.
|
||||||
|
|
||||||
# This script must be edited to meet your needs.
|
# This script must be edited to meet your needs.
|
||||||
|
|
||||||
|
|||||||
@@ -1,5 +1,26 @@
|
|||||||
# ocrmypdf completion -*- shell-script -*-
|
# ocrmypdf completion -*- shell-script -*-
|
||||||
|
|
||||||
|
# Copyright 2019 Frank Pille
|
||||||
|
# Copyright 2020 Alex Willner
|
||||||
|
#
|
||||||
|
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
# of this software and associated documentation files (the "Software"), to deal
|
||||||
|
# in the Software without restriction, including without limitation the rights
|
||||||
|
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
# copies of the Software, and to permit persons to whom the Software is
|
||||||
|
# furnished to do so, subject to the following conditions:
|
||||||
|
#
|
||||||
|
# The above copyright notice and this permission notice shall be included in all
|
||||||
|
# copies or substantial portions of the Software.
|
||||||
|
#
|
||||||
|
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
# SOFTWARE.
|
||||||
|
|
||||||
set -o errexit
|
set -o errexit
|
||||||
|
|
||||||
_ocrmypdf()
|
_ocrmypdf()
|
||||||
|
|||||||
@@ -1,3 +1,23 @@
|
|||||||
|
# Copyright 2020 James R. Barlow
|
||||||
|
#
|
||||||
|
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
# of this software and associated documentation files (the "Software"), to deal
|
||||||
|
# in the Software without restriction, including without limitation the rights
|
||||||
|
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
# copies of the Software, and to permit persons to whom the Software is
|
||||||
|
# furnished to do so, subject to the following conditions:
|
||||||
|
#
|
||||||
|
# The above copyright notice and this permission notice shall be included in all
|
||||||
|
# copies or substantial portions of the Software.
|
||||||
|
#
|
||||||
|
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
# SOFTWARE.
|
||||||
|
|
||||||
complete -c ocrmypdf -x -n '__fish_is_first_arg' -l version
|
complete -c ocrmypdf -x -n '__fish_is_first_arg' -l version
|
||||||
complete -c ocrmypdf -x -n '__fish_is_first_arg' -s h -s "?" -l help
|
complete -c ocrmypdf -x -n '__fish_is_first_arg' -s h -s "?" -l help
|
||||||
|
|
||||||
@@ -39,7 +59,8 @@ complete -c ocrmypdf -x -l output-type -a '(__fish_ocrmypdf_output_type)' -d "se
|
|||||||
|
|
||||||
function __fish_ocrmypdf_pdf_renderer
|
function __fish_ocrmypdf_pdf_renderer
|
||||||
echo -e "auto\t"(_ "auto select PDF renderer")
|
echo -e "auto\t"(_ "auto select PDF renderer")
|
||||||
echo -e "hocr\t"(_ "use hocr renderer")
|
echo -e "hocr\t"(_ "use hOCR renderer")
|
||||||
|
echo -e "hocrdebug\t"(_ "uses hOCR renderer in debug mode, showing recognized text")
|
||||||
echo -e "sandwich\t"(_ "use sandwich renderer")
|
echo -e "sandwich\t"(_ "use sandwich renderer")
|
||||||
end
|
end
|
||||||
complete -c ocrmypdf -x -l pdf-renderer -a '(__fish_ocrmypdf_pdf_renderer)' -d "select PDF renderer options"
|
complete -c ocrmypdf -x -l pdf-renderer -a '(__fish_ocrmypdf_pdf_renderer)' -d "select PDF renderer options"
|
||||||
@@ -115,4 +136,4 @@ complete -c ocrmypdf -r -l user-words -d "specify location of user words file"
|
|||||||
complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file"
|
complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file"
|
||||||
complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF"
|
complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF"
|
||||||
|
|
||||||
complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf)"
|
complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf; __fish_complete_suffix .PDF; __fish_complete_suffix .jpg; __fish_complete_suffix .png)"
|
||||||
|
|||||||
@@ -9,7 +9,7 @@ services:
|
|||||||
- "/media/scan:/input"
|
- "/media/scan:/input"
|
||||||
- "/mnt/scan:/output"
|
- "/mnt/scan:/output"
|
||||||
environment:
|
environment:
|
||||||
- OCR_OUTPUT_DIRECTORY_YEAR_MONT=0
|
- OCR_OUTPUT_DIRECTORY_YEAR_MONTH=0
|
||||||
user: "<SET TO YOUR USER ID>:<SET TO YOUR GROUP ID>"
|
user: "<SET TO YOUR USER ID>:<SET TO YOUR GROUP ID>"
|
||||||
entrypoint: python3
|
entrypoint: python3
|
||||||
command: watcher.py
|
command: watcher.py
|
||||||
|
|||||||
+45
-14
@@ -1,17 +1,41 @@
|
|||||||
# © 2020 James R Barlow: https://github.com/jbarlow83
|
# © 2020 James R Barlow: https://github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This program is free software: you can redistribute it and/or modify
|
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
# it under the terms of the GNU General Public License as published by
|
# of this software and associated documentation files (the "Software"), to deal
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
# in the Software without restriction, including without limitation the rights
|
||||||
# (at your option) any later version.
|
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
# copies of the Software, and to permit persons to whom the Software is
|
||||||
|
# furnished to do so, subject to the following conditions:
|
||||||
#
|
#
|
||||||
# This program is distributed in the hope that it will be useful,
|
# The above copyright notice and this permission notice shall be included in all
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
# copies or substantial portions of the Software.
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
#
|
||||||
# You should have received a copy of the GNU General Public License
|
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
# SOFTWARE.
|
||||||
|
|
||||||
|
"""
|
||||||
|
An example of an OCRmyPDF plugin.
|
||||||
|
|
||||||
|
This plugin adds two new command line arguments
|
||||||
|
--grayscale-ocr: converts the image to grayscale before performing OCR on it
|
||||||
|
(This is occasionally useful for images whose color confounds OCR. It only
|
||||||
|
affects the image shown to OCR. The image is not saved.)
|
||||||
|
--mono-page: converts pages all pages in the output file to black and white
|
||||||
|
|
||||||
|
To use this from the command line:
|
||||||
|
ocrmypdf --plugin path/to/example_plugin.py --mono-page input.pdf output.pdf
|
||||||
|
|
||||||
|
To use this as an API:
|
||||||
|
import ocrmypdf
|
||||||
|
ocrmypdf.ocr('input.pdf', 'output.pdf',
|
||||||
|
plugins=['path/to/example_plugin.py'], mono_page=True
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
@@ -25,6 +49,7 @@ log = logging.getLogger(__name__)
|
|||||||
@hookimpl
|
@hookimpl
|
||||||
def add_options(parser):
|
def add_options(parser):
|
||||||
parser.add_argument('--grayscale-ocr', action='store_true')
|
parser.add_argument('--grayscale-ocr', action='store_true')
|
||||||
|
parser.add_argument('--mono-page', action='store_true')
|
||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
@@ -47,7 +72,13 @@ def filter_ocr_image(page, image):
|
|||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def filter_page_image(page, image_filename):
|
def filter_page_image(page, image_filename):
|
||||||
output = image_filename.with_suffix('.jpg')
|
if page.options.mono_page:
|
||||||
with Image.open(image_filename) as im:
|
with Image.open(image_filename) as im:
|
||||||
im.save(output)
|
im = im.convert('1')
|
||||||
return output
|
im.save(image_filename)
|
||||||
|
return image_filename
|
||||||
|
else:
|
||||||
|
output = image_filename.with_suffix('.jpg')
|
||||||
|
with Image.open(image_filename) as im:
|
||||||
|
im.save(output)
|
||||||
|
return output
|
||||||
|
|||||||
+22
-2
@@ -1,5 +1,23 @@
|
|||||||
#!/bin/env python3
|
#!/bin/env python3
|
||||||
# Contributed by github.com/Enantiomerie
|
# Copyright 2017 github.com/Enantiomerie
|
||||||
|
#
|
||||||
|
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
|
# of this software and associated documentation files (the "Software"), to deal
|
||||||
|
# in the Software without restriction, including without limitation the rights
|
||||||
|
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
# copies of the Software, and to permit persons to whom the Software is
|
||||||
|
# furnished to do so, subject to the following conditions:
|
||||||
|
#
|
||||||
|
# The above copyright notice and this permission notice shall be included in all
|
||||||
|
# copies or substantial portions of the Software.
|
||||||
|
#
|
||||||
|
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
|
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
# SOFTWARE.
|
||||||
|
|
||||||
# This script must be edited to meet your needs.
|
# This script must be edited to meet your needs.
|
||||||
|
|
||||||
@@ -61,8 +79,10 @@ for dir_name, subdirs, file_list in os.walk(start_dir):
|
|||||||
stdout=output_file,
|
stdout=output_file,
|
||||||
stderr=subprocess.PIPE,
|
stderr=subprocess.PIPE,
|
||||||
check=False,
|
check=False,
|
||||||
|
text=True,
|
||||||
|
errors='ignore',
|
||||||
)
|
)
|
||||||
logging.info(proc.stderr.read())
|
logging.info(proc.stderr)
|
||||||
os.chmod(full_path_ocr, 0o664)
|
os.chmod(full_path_ocr, 0o664)
|
||||||
os.chmod(full_path, 0o664)
|
os.chmod(full_path, 0o664)
|
||||||
full_path_ocr_archive = sys.argv[2]
|
full_path_ocr_archive = sys.argv[2]
|
||||||
|
|||||||
+24
-14
@@ -1,18 +1,23 @@
|
|||||||
# Copyright (C) 2019 Ian Alexander: https://github.com/ianalexander
|
# Copyright (C) 2019 Ian Alexander: https://github.com/ianalexander
|
||||||
# Copyright (C) 2020 James R Barlow: https://github.com/jbarlow83
|
# Copyright (C) 2020 James R Barlow: https://github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This program is free software: you can redistribute it and/or modify
|
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
# it under the terms of the GNU General Public License as published by
|
# of this software and associated documentation files (the "Software"), to deal
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
# in the Software without restriction, including without limitation the rights
|
||||||
# (at your option) any later version.
|
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||||
|
# copies of the Software, and to permit persons to whom the Software is
|
||||||
|
# furnished to do so, subject to the following conditions:
|
||||||
#
|
#
|
||||||
# This program is distributed in the hope that it will be useful,
|
# The above copyright notice and this permission notice shall be included in all
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
# copies or substantial portions of the Software.
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
#
|
||||||
# You should have received a copy of the GNU General Public License
|
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||||
|
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||||
|
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||||
|
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||||
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
|
# SOFTWARE.
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
@@ -39,8 +44,8 @@ DESKEW = bool(os.getenv('OCR_DESKEW', ''))
|
|||||||
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
||||||
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
||||||
USE_POLLING = bool(os.getenv('OCR_USE_POLLING', ''))
|
USE_POLLING = bool(os.getenv('OCR_USE_POLLING', ''))
|
||||||
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper()
|
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
|
||||||
PATTERNS = ['*.pdf']
|
PATTERNS = ['*.pdf', '*.PDF']
|
||||||
|
|
||||||
log = logging.getLogger('ocrmypdf-watcher')
|
log = logging.getLogger('ocrmypdf-watcher')
|
||||||
|
|
||||||
@@ -112,7 +117,12 @@ class HandleObserverEvent(PatternMatchingEventHandler):
|
|||||||
|
|
||||||
def main():
|
def main():
|
||||||
ocrmypdf.configure_logging(
|
ocrmypdf.configure_logging(
|
||||||
verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True
|
verbosity=(
|
||||||
|
ocrmypdf.Verbosity.default
|
||||||
|
if LOGLEVEL != 'DEBUG'
|
||||||
|
else ocrmypdf.Verbosity.debug
|
||||||
|
),
|
||||||
|
manage_root_logger=True,
|
||||||
)
|
)
|
||||||
log.setLevel(LOGLEVEL)
|
log.setLevel(LOGLEVEL)
|
||||||
log.info(
|
log.info(
|
||||||
@@ -130,7 +140,7 @@ def main():
|
|||||||
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
||||||
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
||||||
f"USE_POLLING: {USE_POLLING}\n"
|
f"USE_POLLING: {USE_POLLING}\n"
|
||||||
f"LOGLEVEL: {LOGLEVEL}\n"
|
f"LOGLEVEL: {LOGLEVEL}"
|
||||||
)
|
)
|
||||||
|
|
||||||
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS:
|
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS:
|
||||||
|
|||||||
+4
-1
@@ -3,11 +3,14 @@ requires = [
|
|||||||
"setuptools >= 30.3.0",
|
"setuptools >= 30.3.0",
|
||||||
"wheel",
|
"wheel",
|
||||||
"cffi",
|
"cffi",
|
||||||
"setuptools_scm",
|
"setuptools_scm[toml] >= 3.4",
|
||||||
"setuptools_scm_git_archive"
|
"setuptools_scm_git_archive"
|
||||||
]
|
]
|
||||||
build-backend = "setuptools.build_meta"
|
build-backend = "setuptools.build_meta"
|
||||||
|
|
||||||
|
[tool.setuptools_scm]
|
||||||
|
version_scheme = "post-release"
|
||||||
|
|
||||||
[tool.black]
|
[tool.black]
|
||||||
line-length = 88
|
line-length = 88
|
||||||
target-version = ["py36", "py37", "py38"]
|
target-version = ["py36", "py37", "py38"]
|
||||||
|
|||||||
+9
-11
@@ -1,12 +1,10 @@
|
|||||||
# requirements.txt can be used to replicate the developer's build environment
|
# Deprecated and not maintained; use "pip install ocrmypdf" instead
|
||||||
# setup.py lists a separate set of requirements that are looser to simplify
|
cffi == 1.14.5
|
||||||
# installation
|
coloredlogs == 15.0 # technically optional
|
||||||
cffi == 1.14.0
|
img2pdf == 0.4.0
|
||||||
coloredlogs == 14.0 # technically optional
|
pdfminer.six == 20201018
|
||||||
img2pdf == 0.3.6
|
pikepdf == 2.10.0
|
||||||
pdfminer.six == 20200517
|
|
||||||
pikepdf == 1.16.1
|
|
||||||
pluggy == 0.13.1
|
pluggy == 0.13.1
|
||||||
Pillow == 7.1.2
|
Pillow == 8.2.0
|
||||||
reportlab == 3.5.42
|
reportlab == 3.5.66
|
||||||
tqdm == 4.46.1
|
tqdm == 4.59.0
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
pytest >= 5.0.0
|
# Deprecated and not maintained; use "pip install ocrmypdf[test]" instead
|
||||||
pytest-helpers-namespace >= 2019.1.8
|
pytest >= 6.0.0
|
||||||
pytest-xdist >= 1.31.0
|
pytest-xdist >= 2.2.0
|
||||||
pytest-cov >= 2.10.0
|
pytest-cov >= 2.11.1
|
||||||
python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3
|
python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3
|
||||||
# or brew install exempi
|
# or brew install exempi
|
||||||
#PyMuPDF == 1.13.4 # optional
|
#PyMuPDF == 1.13.4 # optional
|
||||||
|
|||||||
@@ -1 +1,2 @@
|
|||||||
watchdog == 0.10.2
|
# Deprecated and not maintained; use "pip install ocrmypdf[watcher]" instead
|
||||||
|
watchdog == 1.0.2
|
||||||
|
|||||||
@@ -1 +1,2 @@
|
|||||||
|
# Deprecated and not maintained; use "pip install ocrmypdf[webservice]" instead
|
||||||
Flask >= 1, < 2
|
Flask >= 1, < 2
|
||||||
|
|||||||
@@ -1,8 +1,101 @@
|
|||||||
|
[metadata]
|
||||||
|
name = ocrmypdf
|
||||||
|
description = OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched
|
||||||
|
long_description = file: README.md
|
||||||
|
long_description_content_type = text/markdown; charset=UTF-8
|
||||||
|
url = https://github.com/jbarlow83/OCRmyPDF
|
||||||
|
author = James R. Barlow
|
||||||
|
author_email = james@purplerock.ca
|
||||||
|
license_files =
|
||||||
|
LICENSE
|
||||||
|
keywords =
|
||||||
|
PDF
|
||||||
|
OCR
|
||||||
|
optical character recognition
|
||||||
|
PDF/A
|
||||||
|
scanning
|
||||||
|
classifiers =
|
||||||
|
Programming Language :: Python :: 3.6
|
||||||
|
Programming Language :: Python :: 3.7
|
||||||
|
Programming Language :: Python :: 3.8
|
||||||
|
Programming Language :: Python :: 3.9
|
||||||
|
Development Status :: 5 - Production/Stable
|
||||||
|
Environment :: Console
|
||||||
|
Intended Audience :: End Users/Desktop
|
||||||
|
Intended Audience :: Science/Research
|
||||||
|
Intended Audience :: System Administrators
|
||||||
|
License :: OSI Approved :: Mozilla Public License 2.0 (MPL 2.0)
|
||||||
|
Operating System :: MacOS :: MacOS X
|
||||||
|
Operating System :: Microsoft :: Windows :: Windows 10
|
||||||
|
Operating System :: POSIX
|
||||||
|
Operating System :: POSIX :: BSD
|
||||||
|
Operating System :: POSIX :: Linux
|
||||||
|
Topic :: Scientific/Engineering :: Image Recognition
|
||||||
|
Topic :: Text Processing :: Indexing
|
||||||
|
Topic :: Text Processing :: Linguistic
|
||||||
|
project_urls =
|
||||||
|
Documentation = https://ocrmypdf.readthedocs.io/
|
||||||
|
Source = https://github.com/jbarlow83/ocrmypdf
|
||||||
|
Tracker = https://github.com/jbarlow83/ocrmypdf/issues
|
||||||
|
|
||||||
|
[options]
|
||||||
|
zip_safe = False
|
||||||
|
packages = find:
|
||||||
|
package_dir =
|
||||||
|
=src
|
||||||
|
platforms = any
|
||||||
|
include_package_data=True
|
||||||
|
install_requires =
|
||||||
|
cffi >= 1.9.1 # must be a setup and install requirement
|
||||||
|
coloredlogs >= 14.0 # strictly optional
|
||||||
|
img2pdf >= 0.3.0, < 0.5 # pure Python, so track HEAD closely
|
||||||
|
pdfminer.six >= 20191110, != 20200720, <= 20201018
|
||||||
|
pikepdf >= 2.10.0
|
||||||
|
Pillow >= 8.2.0
|
||||||
|
pluggy >= 0.13.0, < 1.0
|
||||||
|
reportlab >= 3.5.66
|
||||||
|
setuptools
|
||||||
|
tqdm >= 4
|
||||||
|
python_requires = >= 3.6
|
||||||
|
setup_requires = # can be removed whenever we can drop pip 9 support
|
||||||
|
cffi >= 1.9.1 # to build the leptonica module
|
||||||
|
setuptools_scm # so that version will work
|
||||||
|
setuptools_scm_git_archive # enable version from github tarballs
|
||||||
|
|
||||||
|
[options.package_data]
|
||||||
|
ocrmypdf =
|
||||||
|
data/sRGB.icc
|
||||||
|
py.typed
|
||||||
|
|
||||||
|
[options.packages.find]
|
||||||
|
where = src
|
||||||
|
|
||||||
|
[options.extras_require]
|
||||||
|
test =
|
||||||
|
pytest >= 6.0.0
|
||||||
|
pytest-xdist >= 2.2.0
|
||||||
|
pytest-cov >= 2.11.1
|
||||||
|
python-xmp-toolkit == 2.0.1 # also requires apt-get install libexempi3
|
||||||
|
# or brew install exempi
|
||||||
|
docs =
|
||||||
|
sphinx
|
||||||
|
sphinx_rtd_theme
|
||||||
|
extended_test =
|
||||||
|
PyMuPDF == 1.13.4
|
||||||
|
watcher =
|
||||||
|
watchdog >= 1.0.2, < 2
|
||||||
|
webservice =
|
||||||
|
Flask >= 1, < 2
|
||||||
|
|
||||||
|
[options.entry_points]
|
||||||
|
console_scripts =
|
||||||
|
ocrmypdf = ocrmypdf.__main__:run
|
||||||
|
|
||||||
[bdist_wheel]
|
[bdist_wheel]
|
||||||
python-tag = py35
|
python-tag = py36
|
||||||
|
|
||||||
[aliases]
|
[aliases]
|
||||||
test=pytest
|
test = pytest
|
||||||
|
|
||||||
[check-manifest]
|
[check-manifest]
|
||||||
ignore =
|
ignore =
|
||||||
@@ -15,15 +108,39 @@ filterwarnings =
|
|||||||
ignore:.*XMLParser.*:DeprecationWarning
|
ignore:.*XMLParser.*:DeprecationWarning
|
||||||
markers =
|
markers =
|
||||||
slow
|
slow
|
||||||
|
addopts =
|
||||||
|
-n auto
|
||||||
|
|
||||||
[isort]
|
[isort]
|
||||||
multi_line_output=3
|
multi_line_output = 3
|
||||||
include_trailing_comma=True
|
include_trailing_comma = True
|
||||||
force_grid_wrap=0
|
force_grid_wrap = 0
|
||||||
use_parentheses=True
|
use_parentheses = True
|
||||||
line_length=88
|
line_length = 88
|
||||||
known_first_party = ocrmypdf
|
known_first_party = ocrmypdf
|
||||||
known_third_party = PIL,_cffi_backend,cffi,flask,img2pdf,pdfminer,pikepdf,pkg_resources,pluggy,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug
|
known_third_party = PIL,_cffi_backend,cffi,flask,img2pdf,pdfminer,pikepdf,pkg_resources,pluggy,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug
|
||||||
|
|
||||||
[metadata]
|
[coverage:paths]
|
||||||
license_file = LICENSE
|
source =
|
||||||
|
src/ocrmypdf
|
||||||
|
|
||||||
|
[coverage:run]
|
||||||
|
branch = true
|
||||||
|
parallel = true
|
||||||
|
concurrency = multiprocessing
|
||||||
|
|
||||||
|
[coverage:report]
|
||||||
|
# Regexes for lines to exclude from consideration
|
||||||
|
exclude_lines =
|
||||||
|
# Have to re-enable the standard pragma
|
||||||
|
pragma: no cover
|
||||||
|
|
||||||
|
# Don't complain if tests don't hit defensive assertion code:
|
||||||
|
raise AssertionError
|
||||||
|
raise NotImplementedError
|
||||||
|
|
||||||
|
# Don't complain if non-runnable code isn't run:
|
||||||
|
if 0:
|
||||||
|
if False:
|
||||||
|
if __name__ == .__main__.:
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
|||||||
@@ -1,103 +1,21 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# -*- coding: utf-8 -*-
|
# -*- coding: utf-8 -*-
|
||||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
# © 2021 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
from __future__ import print_function, unicode_literals
|
|
||||||
|
|
||||||
import sys
|
|
||||||
|
|
||||||
from setuptools import find_packages, setup
|
|
||||||
|
|
||||||
if sys.version_info < (3, 6):
|
|
||||||
print("Python 3.6 or newer is required", file=sys.stderr)
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
if 'upload' in sys.argv[1:]:
|
|
||||||
print('Use twine to upload the package - setup.py upload is insecure')
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
tests_require = open('requirements/test.txt', encoding='utf-8').read().splitlines()
|
|
||||||
|
|
||||||
|
|
||||||
def readme():
|
from setuptools import setup
|
||||||
with open('README.md', encoding='utf-8') as f:
|
|
||||||
return f.read()
|
|
||||||
|
|
||||||
|
|
||||||
|
# Minimal setup to support older setuptools/setuptools_scm
|
||||||
setup(
|
setup(
|
||||||
name='ocrmypdf',
|
|
||||||
description='OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched',
|
|
||||||
long_description=readme(),
|
|
||||||
long_description_content_type='text/markdown',
|
|
||||||
url='https://github.com/jbarlow83/OCRmyPDF',
|
|
||||||
author='James R. Barlow',
|
|
||||||
author_email='james@purplerock.ca',
|
|
||||||
packages=find_packages('src', exclude=["tests", "tests.*"]),
|
|
||||||
package_dir={'': 'src'},
|
|
||||||
keywords=['PDF', 'OCR', 'optical character recognition', 'PDF/A', 'scanning'],
|
|
||||||
classifiers=[
|
|
||||||
"Programming Language :: Python :: 3.6",
|
|
||||||
"Programming Language :: Python :: 3.7",
|
|
||||||
"Programming Language :: Python :: 3.8",
|
|
||||||
"Programming Language :: Python :: 3.9",
|
|
||||||
"Development Status :: 5 - Production/Stable",
|
|
||||||
"Environment :: Console",
|
|
||||||
"Intended Audience :: End Users/Desktop",
|
|
||||||
"Intended Audience :: Science/Research",
|
|
||||||
"Intended Audience :: System Administrators",
|
|
||||||
"License :: OSI Approved :: GNU General Public License v3 (GPLv3)",
|
|
||||||
"Operating System :: MacOS :: MacOS X",
|
|
||||||
"Operating System :: Microsoft :: Windows :: Windows 10",
|
|
||||||
"Operating System :: POSIX",
|
|
||||||
"Operating System :: POSIX :: BSD",
|
|
||||||
"Operating System :: POSIX :: Linux",
|
|
||||||
"Topic :: Scientific/Engineering :: Image Recognition",
|
|
||||||
"Topic :: Text Processing :: Indexing",
|
|
||||||
"Topic :: Text Processing :: Linguistic",
|
|
||||||
],
|
|
||||||
python_requires=' >= 3.6',
|
|
||||||
setup_requires=[ # can be removed whenever we can drop pip 9 support
|
setup_requires=[ # can be removed whenever we can drop pip 9 support
|
||||||
'cffi >= 1.9.1', # to build the leptonica module
|
'cffi >= 1.9.1', # to build the leptonica module
|
||||||
'pytest-runner', # to enable python setup.py test
|
|
||||||
'setuptools_scm', # so that version will work
|
'setuptools_scm', # so that version will work
|
||||||
'setuptools_scm_git_archive', # enable version from github tarballs
|
'setuptools_scm_git_archive', # enable version from github tarballs
|
||||||
],
|
],
|
||||||
use_scm_version={'version_scheme': 'post-release'},
|
use_scm_version={'version_scheme': 'post-release'},
|
||||||
cffi_modules=['src/ocrmypdf/lib/compile_leptonica.py:ffibuilder'],
|
cffi_modules=['src/ocrmypdf/lib/compile_leptonica.py:ffibuilder'],
|
||||||
install_requires=[
|
|
||||||
'cffi >= 1.9.1', # must be a setup and install requirement
|
|
||||||
'coloredlogs >= 14.0', # strictly optional
|
|
||||||
'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely
|
|
||||||
'pdfminer.six >= 20191110, <= 20200720',
|
|
||||||
'pikepdf >= 1.14.0, < 2',
|
|
||||||
'Pillow >= 7.0.0',
|
|
||||||
'pluggy >= 0.13.0',
|
|
||||||
'reportlab >= 3.3.0', # oldest released version with sane image handling
|
|
||||||
'tqdm >= 4',
|
|
||||||
],
|
|
||||||
tests_require=tests_require,
|
|
||||||
entry_points={'console_scripts': ['ocrmypdf = ocrmypdf.__main__:run']},
|
|
||||||
package_data={'ocrmypdf': ['data/sRGB.icc']},
|
|
||||||
include_package_data=True,
|
|
||||||
zip_safe=False,
|
|
||||||
project_urls={
|
|
||||||
'Documentation': 'https://ocrmypdf.readthedocs.io/',
|
|
||||||
'Source': 'https://github.com/jbarlow83/ocrmypdf',
|
|
||||||
'Tracker': 'https://github.com/jbarlow83/ocrmypdf/issues',
|
|
||||||
},
|
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
# Release checklist
|
||||||
|
|
||||||
|
## Patch release
|
||||||
|
|
||||||
|
- Check `pytest`
|
||||||
|
|
||||||
|
- Update release notes
|
||||||
|
|
||||||
|
## Minor release
|
||||||
|
|
||||||
|
## Major release
|
||||||
|
|
||||||
|
- Run `pre-commit autoupdate`
|
||||||
|
|
||||||
|
- Check README.md
|
||||||
|
|
||||||
|
- Check setup.py
|
||||||
|
|
||||||
|
- Are classifiers up to date?
|
||||||
|
- Is `python_requires` correct?
|
||||||
|
- Python 3.6 is EOL on December 2021-12. Could drop support then.
|
||||||
|
- Can we tighten any `install_requires` dependencies?
|
||||||
|
|
||||||
|
- Search for old version shims we can remove
|
||||||
|
|
||||||
|
- "shim"
|
||||||
|
- ` pikepdf.__version__`
|
||||||
|
|
||||||
|
- Search for deprecation: search all files for deprec*, etc.
|
||||||
|
|
||||||
|
- Check requirements in setup.cfg
|
||||||
|
|
||||||
|
- Delete `tests/cache`, do `pytest --runslow`, and update cache.
|
||||||
|
|
||||||
|
- Do `pytest --cov-report html`
|
||||||
@@ -1,24 +1,15 @@
|
|||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
# © 2017 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
|
|
||||||
from pluggy import HookimplMarker as _HookimplMarker
|
from pluggy import HookimplMarker as _HookimplMarker
|
||||||
|
|
||||||
from ocrmypdf import helpers, hocrtransform, leptonica, pdfa, pdfinfo
|
from ocrmypdf import helpers, hocrtransform, leptonica, pdfa, pdfinfo
|
||||||
|
from ocrmypdf._concurrent import Executor
|
||||||
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._version import PROGRAM_NAME, __version__
|
from ocrmypdf._version import PROGRAM_NAME, __version__
|
||||||
from ocrmypdf.api import Verbosity, configure_logging, ocr
|
from ocrmypdf.api import Verbosity, configure_logging, ocr
|
||||||
from ocrmypdf.exceptions import (
|
from ocrmypdf.exceptions import (
|
||||||
|
|||||||
@@ -1,20 +1,10 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# © 2015-19 James R. Barlow: github.com/jbarlow83
|
# © 2015-19 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -57,7 +47,10 @@ def run(args=None):
|
|||||||
verbosity = Verbosity.quiet
|
verbosity = Verbosity.quiet
|
||||||
options.progress_bar = False
|
options.progress_bar = False
|
||||||
configure_logging(
|
configure_logging(
|
||||||
verbosity, progress_bar_friendly=options.progress_bar, manage_root_logger=True
|
verbosity,
|
||||||
|
progress_bar_friendly=options.progress_bar,
|
||||||
|
manage_root_logger=True,
|
||||||
|
plugin_manager=plugin_manager,
|
||||||
)
|
)
|
||||||
log.debug('ocrmypdf %s', __version__)
|
log.debug('ocrmypdf %s', __version__)
|
||||||
try:
|
try:
|
||||||
|
|||||||
+120
-135
@@ -1,148 +1,133 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import logging.handlers
|
|
||||||
import multiprocessing
|
|
||||||
import os
|
|
||||||
import signal
|
|
||||||
import sys
|
|
||||||
import threading
|
import threading
|
||||||
from multiprocessing import Pool as ProcessPool
|
from abc import ABC, abstractmethod
|
||||||
from multiprocessing.dummy import Pool as ThreadPool
|
|
||||||
from typing import Callable, Iterable, Optional
|
from typing import Callable, Iterable, Optional
|
||||||
|
|
||||||
from tqdm import tqdm
|
|
||||||
|
|
||||||
from ocrmypdf.exceptions import InputFileError
|
def _task_noop(*_args, **_kwargs):
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
def log_listener(queue):
|
class NullProgressBar:
|
||||||
"""Listen to the worker processes and forward the messages to logging
|
def __init__(self, **kwargs):
|
||||||
|
pass
|
||||||
|
|
||||||
For simplicity this is a thread rather than a process. Only one process
|
def __enter__(self):
|
||||||
should actually write to sys.stderr or whatever we're using, so if this is
|
return self
|
||||||
made into a process the main application needs to be directed to it.
|
|
||||||
|
|
||||||
See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes
|
def __exit__(self, exc_type, exc_value, traceback):
|
||||||
|
return False
|
||||||
|
|
||||||
|
def update(self, _arg=None):
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
|
class Executor(ABC):
|
||||||
|
pool_lock = threading.Lock()
|
||||||
|
pbar_class = NullProgressBar
|
||||||
|
|
||||||
|
def __init__(self, *, pbar_class=None):
|
||||||
|
if pbar_class:
|
||||||
|
self.pbar_class = pbar_class
|
||||||
|
|
||||||
|
def __call__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
use_threads: bool,
|
||||||
|
max_workers: int,
|
||||||
|
tqdm_kwargs: dict,
|
||||||
|
worker_initializer: Optional[Callable] = None,
|
||||||
|
task: Optional[Callable] = None,
|
||||||
|
task_arguments: Optional[Iterable] = None,
|
||||||
|
task_finished: Optional[Callable] = None,
|
||||||
|
) -> None:
|
||||||
|
"""
|
||||||
|
Set up parallel execution and progress reporting.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
use_threads: If ``False``, the workload is the sort that will benefit from
|
||||||
|
running in a multiprocessing context (for example, it uses Python
|
||||||
|
heavily, and parallelizing it with threads is not expected to be
|
||||||
|
performant).
|
||||||
|
max_workers: The maximum number of workers that should be run.
|
||||||
|
tdqm_kwargs: Arguments to set up the progress bar.
|
||||||
|
worker_initializer: Called when a worker is initialized, in the worker's
|
||||||
|
execution context. If the child workers are processes, it must be
|
||||||
|
possible to marshall/pickle the worker initializer.
|
||||||
|
``functools.partial`` can be used to bind parameters.
|
||||||
|
task: Called when the worker starts a new task, in the worker's execution
|
||||||
|
context. Must be possible to marshall to the worker.
|
||||||
|
task_finished: Called when a worker finishes a task, in the parent's
|
||||||
|
context.
|
||||||
|
task_arguments: An iterable that generates a group of parameters for each
|
||||||
|
task. This runs in the parent's context, but the parameters must be
|
||||||
|
marshallable to the worker.
|
||||||
|
"""
|
||||||
|
|
||||||
|
if not task_arguments:
|
||||||
|
return # Nothing to do!
|
||||||
|
if not worker_initializer:
|
||||||
|
worker_initializer = _task_noop
|
||||||
|
if not task_finished:
|
||||||
|
task_finished = _task_noop
|
||||||
|
if not task:
|
||||||
|
task = _task_noop
|
||||||
|
|
||||||
|
with self.pool_lock:
|
||||||
|
self._execute(
|
||||||
|
use_threads=use_threads,
|
||||||
|
max_workers=max_workers,
|
||||||
|
tqdm_kwargs=tqdm_kwargs,
|
||||||
|
worker_initializer=worker_initializer,
|
||||||
|
task=task,
|
||||||
|
task_arguments=task_arguments,
|
||||||
|
task_finished=task_finished,
|
||||||
|
)
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
def _execute(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
use_threads: bool,
|
||||||
|
max_workers: int,
|
||||||
|
tqdm_kwargs: dict,
|
||||||
|
worker_initializer: Callable,
|
||||||
|
task: Callable,
|
||||||
|
task_arguments: Iterable,
|
||||||
|
task_finished: Callable,
|
||||||
|
):
|
||||||
|
"""Custom executors should override this method."""
|
||||||
|
|
||||||
|
|
||||||
|
def setup_executor(plugin_manager) -> Executor:
|
||||||
|
pbar_class = plugin_manager.hook.get_progressbar_class()
|
||||||
|
return plugin_manager.hook.get_executor(progressbar_class=pbar_class)
|
||||||
|
|
||||||
|
|
||||||
|
class SerialExecutor(Executor):
|
||||||
|
"""Implements a purely sequential executor using the parallel protocol.
|
||||||
|
|
||||||
|
The current process/thread will be the worker that executes all tasks
|
||||||
|
in order. As such, ``worker_initializer`` will never be called.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
while True:
|
def _execute(
|
||||||
try:
|
self,
|
||||||
record = queue.get()
|
*,
|
||||||
if record is None:
|
use_threads: bool,
|
||||||
break
|
max_workers: int,
|
||||||
logger = logging.getLogger(record.name)
|
tqdm_kwargs: dict,
|
||||||
logger.handle(record)
|
worker_initializer: Callable,
|
||||||
except Exception: # pylint: disable=broad-except
|
task: Callable,
|
||||||
import traceback # pylint: disable=import-outside-toplevel
|
task_arguments: Iterable,
|
||||||
|
task_finished: Callable,
|
||||||
print("Logging problem", file=sys.stderr)
|
): # pylint: disable=unused-argument
|
||||||
traceback.print_exc(file=sys.stderr)
|
with self.pbar_class(**tqdm_kwargs) as pbar:
|
||||||
|
for args in task_arguments:
|
||||||
|
result = task(args)
|
||||||
def process_sigbus(*args):
|
task_finished(result, pbar)
|
||||||
raise InputFileError("A worker process lost access to an input file")
|
|
||||||
|
|
||||||
|
|
||||||
def process_init(queue, user_init):
|
|
||||||
"""Initialize a process pool worker"""
|
|
||||||
|
|
||||||
# Ignore SIGINT (our parent process will kill us gracefully)
|
|
||||||
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
|
||||||
|
|
||||||
# Install SIGBUS handler (so our parent process can abort somewhat gracefully)
|
|
||||||
if hasattr(signal, 'SIGBUS'):
|
|
||||||
signal.signal(signal.SIGBUS, process_sigbus)
|
|
||||||
|
|
||||||
# Reconfigure the root logger for this process to send all messages to a queue
|
|
||||||
h = logging.handlers.QueueHandler(queue)
|
|
||||||
root = logging.getLogger()
|
|
||||||
root.handlers = []
|
|
||||||
root.addHandler(h)
|
|
||||||
|
|
||||||
if user_init:
|
|
||||||
user_init()
|
|
||||||
|
|
||||||
|
|
||||||
def thread_init(_queue, user_init):
|
|
||||||
# As a thread, block SIGBUS so the main thread deals with it...
|
|
||||||
if hasattr(signal, 'SIGBUS'):
|
|
||||||
signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS})
|
|
||||||
if user_init:
|
|
||||||
user_init()
|
|
||||||
|
|
||||||
|
|
||||||
def exec_progress_pool(
|
|
||||||
*,
|
|
||||||
use_threads: bool,
|
|
||||||
max_workers: int,
|
|
||||||
tqdm_kwargs: dict,
|
|
||||||
task_initializer: Optional[Callable] = None,
|
|
||||||
task: Optional[Callable] = None,
|
|
||||||
task_arguments: Optional[Iterable] = None,
|
|
||||||
task_finished: Optional[Callable] = None,
|
|
||||||
):
|
|
||||||
log_queue: multiprocessing.Queue = multiprocessing.Queue(-1)
|
|
||||||
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
|
||||||
|
|
||||||
if use_threads:
|
|
||||||
pool_class = ThreadPool
|
|
||||||
initializer = thread_init
|
|
||||||
else:
|
|
||||||
pool_class = ProcessPool
|
|
||||||
initializer = process_init
|
|
||||||
listener.start()
|
|
||||||
|
|
||||||
with tqdm(**tqdm_kwargs) as pbar:
|
|
||||||
pool = pool_class(
|
|
||||||
processes=max_workers,
|
|
||||||
initializer=initializer,
|
|
||||||
initargs=(log_queue, task_initializer),
|
|
||||||
)
|
|
||||||
try:
|
|
||||||
results = pool.imap_unordered(task, task_arguments)
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
result = results.next()
|
|
||||||
if task_finished:
|
|
||||||
task_finished(result, pbar)
|
|
||||||
else:
|
|
||||||
pbar.update()
|
|
||||||
except StopIteration:
|
|
||||||
break
|
|
||||||
except KeyboardInterrupt:
|
|
||||||
# Terminate pool so we exit instantly
|
|
||||||
pool.terminate()
|
|
||||||
# Don't try listener.join() here, will deadlock
|
|
||||||
raise
|
|
||||||
except Exception:
|
|
||||||
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
|
||||||
# Unless inside pytest, exit immediately because no one wants
|
|
||||||
# to wait for child processes to finalize results that will be
|
|
||||||
# thrown away. Inside pytest, we want child processes to exit
|
|
||||||
# cleanly so that they output an error messages or coverage data
|
|
||||||
# we need from them.
|
|
||||||
pool.terminate()
|
|
||||||
raise
|
|
||||||
finally:
|
|
||||||
# Terminate log listener
|
|
||||||
log_queue.put_nowait(None)
|
|
||||||
pool.close()
|
|
||||||
pool.join()
|
|
||||||
|
|
||||||
listener.join()
|
|
||||||
|
|||||||
@@ -1,18 +1,8 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Manage third party executables"""
|
"""Manage third party executables"""
|
||||||
|
|||||||
@@ -1,19 +1,9 @@
|
|||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
# © 2017 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Interface to Ghostscript executable"""
|
"""Interface to Ghostscript executable"""
|
||||||
|
|
||||||
@@ -25,34 +15,34 @@ from os import fspath
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import which
|
from shutil import which
|
||||||
from subprocess import PIPE, CalledProcessError
|
from subprocess import PIPE, CalledProcessError
|
||||||
from typing import Optional, cast
|
from typing import Optional
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
||||||
from ocrmypdf.helpers import Resolution
|
from ocrmypdf.helpers import Resolution
|
||||||
from ocrmypdf.subprocess import get_version, run
|
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
missing_gs_error = """
|
||||||
|
---------------------------------------------------------------------
|
||||||
|
This error normally occurs when ocrmypdf find can't Ghostscript.
|
||||||
|
Please ensure Ghostscript is installed and its location is added to
|
||||||
|
the system PATH environment variable.
|
||||||
|
|
||||||
|
For details see:
|
||||||
|
https://ocrmypdf.readthedocs.io/en/latest/installation.html
|
||||||
|
---------------------------------------------------------------------
|
||||||
|
"""
|
||||||
|
|
||||||
_gswin = None
|
_gswin = None
|
||||||
if os.name == 'nt':
|
if os.name == 'nt':
|
||||||
_gswin = which('gswin64c')
|
_gswin = which('gswin64c')
|
||||||
if not _gswin:
|
if not _gswin:
|
||||||
_gswin = which('gswin32c')
|
_gswin = which('gswin32c')
|
||||||
if not _gswin:
|
if not _gswin:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(missing_gs_error)
|
||||||
"""
|
|
||||||
---------------------------------------------------------------------
|
|
||||||
This error normally occurs when ocrmypdf can't Ghostscript. Please
|
|
||||||
ensure Ghostscript is installed and its location is added to the
|
|
||||||
system PATH environment variable.
|
|
||||||
|
|
||||||
For details see:
|
|
||||||
https://ocrmypdf.readthedocs.io/en/latest/installation.html
|
|
||||||
---------------------------------------------------------------------
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
_gswin = Path(_gswin).stem
|
_gswin = Path(_gswin).stem
|
||||||
|
|
||||||
GS = _gswin if _gswin else 'gs'
|
GS = _gswin if _gswin else 'gs'
|
||||||
@@ -66,16 +56,14 @@ def version():
|
|||||||
def jpeg_passthrough_available() -> bool:
|
def jpeg_passthrough_available() -> bool:
|
||||||
"""Returns True if the installed version of Ghostscript supports JPEG passthru
|
"""Returns True if the installed version of Ghostscript supports JPEG passthru
|
||||||
|
|
||||||
Prior to 9.23, Ghostscript decode and re-encoded JPEGs internally. In 9.23
|
Prior to 9.23, Ghostscript decoded and re-encoded JPEGs internally. In 9.23
|
||||||
it gained the ability to keep JPEGs unmodified. However, the 9.23
|
it gained the ability to keep JPEGs unmodified. However, the 9.23
|
||||||
implementation was buggy and would deletes the last two bytes of images in
|
implementation was buggy and would deletes the last two bytes of images in
|
||||||
some cases, as reported here.
|
some cases, as reported here.
|
||||||
https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
||||||
|
|
||||||
The issue was fixed for 9.24, hence that is the first version we consider
|
The issue was fixed for 9.24, hence that is the first version we consider
|
||||||
the feature available. (However, we don't use 9.24 at all, so the first
|
the feature available. (Ghostscript 9.24 has its own problems is blacklisted.)
|
||||||
version that allows JPEG passthrough is 9.25.
|
|
||||||
|
|
||||||
"""
|
"""
|
||||||
return version() >= '9.24'
|
return version() >= '9.24'
|
||||||
|
|
||||||
@@ -91,8 +79,8 @@ def rasterize_pdf(
|
|||||||
raster_device: str,
|
raster_device: str,
|
||||||
raster_dpi: Resolution,
|
raster_dpi: Resolution,
|
||||||
pageno: int = 1,
|
pageno: int = 1,
|
||||||
page_dpi: Resolution = None,
|
page_dpi: Optional[Resolution] = None,
|
||||||
rotation: int = None,
|
rotation: Optional[int] = None,
|
||||||
filter_vector: bool = False,
|
filter_vector: bool = False,
|
||||||
):
|
):
|
||||||
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units."""
|
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units."""
|
||||||
@@ -132,8 +120,6 @@ def rasterize_pdf(
|
|||||||
stderr = p.stderr.decode(errors='replace')
|
stderr = p.stderr.decode(errors='replace')
|
||||||
if _gs_error_reported(stderr):
|
if _gs_error_reported(stderr):
|
||||||
log.error(stderr)
|
log.error(stderr)
|
||||||
elif stderr:
|
|
||||||
log.debug(stderr)
|
|
||||||
|
|
||||||
with Image.open(BytesIO(p.stdout)) as im:
|
with Image.open(BytesIO(p.stdout)) as im:
|
||||||
if rotation is not None:
|
if rotation is not None:
|
||||||
@@ -151,13 +137,44 @@ def rasterize_pdf(
|
|||||||
im.save(fspath(output_file), dpi=page_dpi)
|
im.save(fspath(output_file), dpi=page_dpi)
|
||||||
|
|
||||||
|
|
||||||
|
class GhostscriptFollower:
|
||||||
|
re_process = re.compile(r"Processing pages \d+ through (\d+).")
|
||||||
|
re_page = re.compile(r"Page (\d+)")
|
||||||
|
|
||||||
|
def __init__(self, progressbar_class):
|
||||||
|
self.count = 0
|
||||||
|
self.progressbar_class = progressbar_class
|
||||||
|
self.progressbar = None
|
||||||
|
|
||||||
|
def __call__(self, line):
|
||||||
|
if not self.progressbar_class:
|
||||||
|
return
|
||||||
|
if not self.progressbar:
|
||||||
|
m = self.re_process.match(line.strip())
|
||||||
|
if m:
|
||||||
|
self.count = int(m.group(1))
|
||||||
|
self.progressbar = self.progressbar_class(
|
||||||
|
total=self.count, desc="PDF/A conversion", unit='page'
|
||||||
|
)
|
||||||
|
return
|
||||||
|
else:
|
||||||
|
m = self.re_page.match(line.strip())
|
||||||
|
if m:
|
||||||
|
self.progressbar.update()
|
||||||
|
|
||||||
|
|
||||||
def generate_pdfa(
|
def generate_pdfa(
|
||||||
pdf_pages,
|
pdf_pages,
|
||||||
output_file: os.PathLike,
|
output_file: os.PathLike,
|
||||||
|
*,
|
||||||
compression: str,
|
compression: str,
|
||||||
pdf_version: str = '1.5',
|
pdf_version: str = '1.5',
|
||||||
pdfa_part: str = '2',
|
pdfa_part: str = '2',
|
||||||
|
progressbar_class=None,
|
||||||
):
|
):
|
||||||
|
# Ghostscript's compression is all or nothing. We can either force all images
|
||||||
|
# to JPEG, force all to Flate/PNG, or let it decide how to encode the images.
|
||||||
|
# In most case it's best to let it decide.
|
||||||
compression_args = []
|
compression_args = []
|
||||||
if compression == 'jpeg':
|
if compression == 'jpeg':
|
||||||
compression_args = [
|
compression_args = [
|
||||||
@@ -179,14 +196,16 @@ def generate_pdfa(
|
|||||||
"-dAutoFilterGrayImages=true",
|
"-dAutoFilterGrayImages=true",
|
||||||
]
|
]
|
||||||
|
|
||||||
|
strategy = 'LeaveColorUnchanged'
|
||||||
# Older versions of Ghostscript expect a leading slash in
|
# Older versions of Ghostscript expect a leading slash in
|
||||||
# sColorConversionStrategy, newer ones should not have it. See Ghostscript
|
# sColorConversionStrategy, newer ones should not have it. See Ghostscript
|
||||||
# git commit fe1c025d.
|
# git commit fe1c025d.
|
||||||
strategy = 'RGB' if version() >= '9.19' else '/RGB'
|
strategy = ('/' + strategy) if version() < '9.19' else strategy
|
||||||
|
|
||||||
if version() == '9.23':
|
if version() == '9.23':
|
||||||
# 9.23: new feature JPEG passthrough is broken in some cases, best to
|
# 9.23: added JPEG passthrough as a new feature, but with a bug that
|
||||||
# disable it always
|
# incorrectly formats some images. Fixed as of 9.24. So we disable this
|
||||||
|
# feature for 9.23.
|
||||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
# https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
||||||
compression_args.append('-dPassThroughJPEGImages=false')
|
compression_args.append('-dPassThroughJPEGImages=false')
|
||||||
|
|
||||||
@@ -196,7 +215,6 @@ def generate_pdfa(
|
|||||||
args_gs = (
|
args_gs = (
|
||||||
[
|
[
|
||||||
GS,
|
GS,
|
||||||
"-dQUIET",
|
|
||||||
"-dBATCH",
|
"-dBATCH",
|
||||||
"-dNOPAUSE",
|
"-dNOPAUSE",
|
||||||
"-dSAFER",
|
"-dSAFER",
|
||||||
@@ -216,16 +234,28 @@ def generate_pdfa(
|
|||||||
]
|
]
|
||||||
)
|
)
|
||||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||||
|
|
||||||
try:
|
try:
|
||||||
with Path(output_file).open('wb') as output:
|
with Path(output_file).open('wb') as output:
|
||||||
p = run(args_gs, stdout=output, stderr=PIPE, check=True)
|
p = run_polling_stderr(
|
||||||
|
args_gs,
|
||||||
|
stdout=output,
|
||||||
|
stderr=PIPE,
|
||||||
|
check=True,
|
||||||
|
text=True,
|
||||||
|
encoding='utf-8',
|
||||||
|
errors='replace',
|
||||||
|
callback=GhostscriptFollower(progressbar_class),
|
||||||
|
)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
# Ghostscript does not change return code when it fails to create
|
# Ghostscript does not change return code when it fails to create
|
||||||
# PDF/A - check PDF/A status elsewhere
|
# PDF/A - check PDF/A status elsewhere
|
||||||
log.error(e.stderr.decode(errors='replace'))
|
log.error(e.stderr)
|
||||||
raise SubprocessOutputError('Ghostscript PDF/A rendering failed')
|
raise SubprocessOutputError('Ghostscript PDF/A rendering failed') from e
|
||||||
else:
|
else:
|
||||||
stderr = p.stderr.decode('utf-8', errors='replace')
|
stderr = p.stderr
|
||||||
|
# If there is an error we log the whole stderr, except for filtering
|
||||||
|
# duplicates.
|
||||||
if _gs_error_reported(stderr):
|
if _gs_error_reported(stderr):
|
||||||
last_part = None
|
last_part = None
|
||||||
repcount = 0
|
repcount = 0
|
||||||
@@ -238,11 +268,3 @@ def generate_pdfa(
|
|||||||
else:
|
else:
|
||||||
repcount += 1
|
repcount += 1
|
||||||
last_part = part
|
last_part = part
|
||||||
elif 'overprint mode not set' in stderr:
|
|
||||||
# Unless someone is going to print PDF/A documents on a
|
|
||||||
# magical sRGB printer I can't see the removal of overprinting
|
|
||||||
# being a problem....
|
|
||||||
log.debug(
|
|
||||||
"Ghostscript had to remove PDF 'overprinting' from the "
|
|
||||||
"input file to complete PDF/A conversion. "
|
|
||||||
)
|
|
||||||
|
|||||||
@@ -1,19 +1,9 @@
|
|||||||
# © 2018 James R. Barlow: github.com/jbarlow83
|
# © 2018 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Interface to jbig2 executable"""
|
"""Interface to jbig2 executable"""
|
||||||
|
|
||||||
@@ -51,9 +41,17 @@ def convert_group(*, cwd, infiles, out_prefix):
|
|||||||
return proc
|
return proc
|
||||||
|
|
||||||
|
|
||||||
|
def convert_group_mp(args):
|
||||||
|
return convert_group(cwd=args[0], infiles=args[1], out_prefix=args[2])
|
||||||
|
|
||||||
|
|
||||||
def convert_single(*, cwd, infile, outfile):
|
def convert_single(*, cwd, infile, outfile):
|
||||||
args = ['jbig2', '-p', infile]
|
args = ['jbig2', '-p', infile]
|
||||||
with open(outfile, 'wb') as fstdout:
|
with open(outfile, 'wb') as fstdout:
|
||||||
proc = run(args, cwd=cwd, stdout=fstdout, stderr=PIPE)
|
proc = run(args, cwd=cwd, stdout=fstdout, stderr=PIPE)
|
||||||
proc.check_returncode()
|
proc.check_returncode()
|
||||||
return proc
|
return proc
|
||||||
|
|
||||||
|
|
||||||
|
def convert_single_mp(args):
|
||||||
|
return convert_single(cwd=args[0], infile=args[1], outfile=args[2])
|
||||||
|
|||||||
@@ -1,24 +1,16 @@
|
|||||||
# © 2018 James R. Barlow: github.com/jbarlow83
|
# © 2018 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Interface to pngquant executable"""
|
"""Interface to pngquant executable"""
|
||||||
|
|
||||||
from os import fspath
|
from contextlib import contextmanager
|
||||||
from tempfile import NamedTemporaryFile
|
from io import BytesIO
|
||||||
|
from pathlib import Path
|
||||||
|
from subprocess import PIPE
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
@@ -38,34 +30,36 @@ def available():
|
|||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
def quantize(input_file, output_file, quality_min, quality_max):
|
@contextmanager
|
||||||
input_file = fspath(input_file)
|
def input_as_png(input_file: Path):
|
||||||
output_file = fspath(output_file)
|
if not input_file.name.endswith('.png'):
|
||||||
if input_file.endswith('.jpg'):
|
with Image.open(input_file) as im:
|
||||||
with Image.open(input_file) as im, NamedTemporaryFile(suffix='.png') as tmp:
|
bio = BytesIO()
|
||||||
im.save(tmp)
|
im.save(bio, format='png')
|
||||||
args = [
|
bio.seek(0)
|
||||||
'pngquant',
|
yield bio
|
||||||
'--force',
|
|
||||||
'--skip-if-larger',
|
|
||||||
'--output',
|
|
||||||
output_file,
|
|
||||||
'--quality',
|
|
||||||
f'{quality_min}-{quality_max}',
|
|
||||||
'--',
|
|
||||||
tmp.name,
|
|
||||||
]
|
|
||||||
run(args)
|
|
||||||
else:
|
else:
|
||||||
|
with open(input_file, 'rb') as f:
|
||||||
|
yield f
|
||||||
|
|
||||||
|
|
||||||
|
def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max: int):
|
||||||
|
with input_as_png(input_file) as input_stream:
|
||||||
args = [
|
args = [
|
||||||
'pngquant',
|
'pngquant',
|
||||||
'--force',
|
'--force',
|
||||||
'--skip-if-larger',
|
'--skip-if-larger',
|
||||||
'--output',
|
|
||||||
output_file,
|
|
||||||
'--quality',
|
'--quality',
|
||||||
f'{quality_min}-{quality_max}',
|
f'{quality_min}-{quality_max}',
|
||||||
'--',
|
'--', # pngquant: stop processing arguments
|
||||||
input_file,
|
'-', # pngquant: stream input and output
|
||||||
]
|
]
|
||||||
run(args)
|
result = run(args, stdin=input_stream, stdout=PIPE, stderr=PIPE, check=False)
|
||||||
|
|
||||||
|
if result.returncode == 0:
|
||||||
|
# input_file could be the same as output_file, so we defer the write
|
||||||
|
output_file.write_bytes(result.stdout)
|
||||||
|
|
||||||
|
|
||||||
|
def quantize_mp(args):
|
||||||
|
return quantize(*args)
|
||||||
|
|||||||
@@ -1,30 +1,22 @@
|
|||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
# © 2017 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Interface to Tesseract executable"""
|
"""Interface to Tesseract executable"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
import shutil
|
import shutil
|
||||||
from collections import namedtuple
|
from collections import namedtuple
|
||||||
|
from distutils.version import StrictVersion
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
||||||
from typing import List
|
from typing import List, Optional
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
@@ -63,31 +55,30 @@ class TesseractLoggerAdapter(logging.LoggerAdapter):
|
|||||||
return '[tesseract] %s' % (msg), kwargs
|
return '[tesseract] %s' % (msg), kwargs
|
||||||
|
|
||||||
|
|
||||||
|
class TesseractVersion(StrictVersion):
|
||||||
|
version_re = re.compile(
|
||||||
|
r'''
|
||||||
|
^(\d+) \. (\d+) (\. (\d+))? # groups: 1/major, 2/minor, 3/[skip], 4/patch
|
||||||
|
[-]? # optional hyphen separator
|
||||||
|
(?:(alpha|beta|rc|dev)[.\-\ ]?(\d+)?)? # 5/prerelease, 6/prerelease_num
|
||||||
|
(?:-(\d+)-g[0-9a-f]+)? # untagged git version
|
||||||
|
$
|
||||||
|
''',
|
||||||
|
re.VERBOSE | re.ASCII,
|
||||||
|
)
|
||||||
|
|
||||||
|
def parse(self, vstring):
|
||||||
|
try:
|
||||||
|
super().parse(vstring)
|
||||||
|
except TypeError as e:
|
||||||
|
if 'int() argument must be a string' in str(e):
|
||||||
|
super().parse(vstring + '0')
|
||||||
|
|
||||||
|
|
||||||
def version():
|
def version():
|
||||||
return get_version('tesseract', regex=r'tesseract\s(.+)')
|
return get_version('tesseract', regex=r'tesseract\s(.+)')
|
||||||
|
|
||||||
|
|
||||||
def has_textonly_pdf(langs=None):
|
|
||||||
"""Does Tesseract have textonly_pdf capability?
|
|
||||||
|
|
||||||
Available in v4.00.00alpha since January 2017. Best to
|
|
||||||
parse the parameter list.
|
|
||||||
"""
|
|
||||||
args_tess = tess_base_args(langs, engine_mode=None) + ['--print-parameters', 'pdf']
|
|
||||||
params = ''
|
|
||||||
try:
|
|
||||||
proc = run(args_tess, check=True, stdout=PIPE, stderr=STDOUT)
|
|
||||||
params = proc.stdout
|
|
||||||
except CalledProcessError as e:
|
|
||||||
raise MissingDependencyError(
|
|
||||||
"Could not --print-parameters from tesseract. This can happen if the "
|
|
||||||
"TESSDATA_PREFIX environment is not set to a valid tessdata folder. "
|
|
||||||
) from e
|
|
||||||
if b'textonly_pdf' in params:
|
|
||||||
return True
|
|
||||||
return False
|
|
||||||
|
|
||||||
|
|
||||||
def has_user_words():
|
def has_user_words():
|
||||||
"""Does Tesseract have --user-words capability?
|
"""Does Tesseract have --user-words capability?
|
||||||
|
|
||||||
@@ -110,7 +101,12 @@ def get_languages():
|
|||||||
args_tess = ['tesseract', '--list-langs']
|
args_tess = ['tesseract', '--list-langs']
|
||||||
try:
|
try:
|
||||||
proc = run(
|
proc = run(
|
||||||
args_tess, universal_newlines=True, stdout=PIPE, stderr=STDOUT, check=True
|
args_tess,
|
||||||
|
text=True,
|
||||||
|
stdout=PIPE,
|
||||||
|
stderr=STDOUT,
|
||||||
|
logs_errors_to_stdout=True,
|
||||||
|
check=True,
|
||||||
)
|
)
|
||||||
output = proc.stdout
|
output = proc.stdout
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
@@ -123,7 +119,7 @@ def get_languages():
|
|||||||
return set(lang.strip() for lang in rest)
|
return set(lang.strip() for lang in rest)
|
||||||
|
|
||||||
|
|
||||||
def tess_base_args(langs: List[str], engine_mode: int) -> List[str]:
|
def tess_base_args(langs: List[str], engine_mode: Optional[int]) -> List[str]:
|
||||||
args = ['tesseract']
|
args = ['tesseract']
|
||||||
if langs:
|
if langs:
|
||||||
args.extend(['-l', '+'.join(langs)])
|
args.extend(['-l', '+'.join(langs)])
|
||||||
@@ -132,7 +128,7 @@ def tess_base_args(langs: List[str], engine_mode: int) -> List[str]:
|
|||||||
return args
|
return args
|
||||||
|
|
||||||
|
|
||||||
def get_orientation(input_file: Path, engine_mode: int, timeout: float):
|
def get_orientation(input_file: Path, engine_mode: Optional[int], timeout: float):
|
||||||
args_tesseract = tess_base_args(['osd'], engine_mode) + [
|
args_tesseract = tess_base_args(['osd'], engine_mode) + [
|
||||||
'--psm',
|
'--psm',
|
||||||
'0',
|
'0',
|
||||||
@@ -226,6 +222,7 @@ def _generate_null_hocr(output_hocr, output_text, image):
|
|||||||
|
|
||||||
|
|
||||||
def generate_hocr(
|
def generate_hocr(
|
||||||
|
*,
|
||||||
input_file: Path,
|
input_file: Path,
|
||||||
output_hocr: Path,
|
output_hocr: Path,
|
||||||
output_text: Path,
|
output_text: Path,
|
||||||
|
|||||||
@@ -1,19 +1,9 @@
|
|||||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
# unpaper documentation:
|
# unpaper documentation:
|
||||||
# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md
|
# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md
|
||||||
@@ -23,10 +13,11 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import shlex
|
import shlex
|
||||||
|
from decimal import Decimal
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
from subprocess import PIPE, STDOUT
|
||||||
from tempfile import TemporaryDirectory
|
from tempfile import TemporaryDirectory
|
||||||
from typing import Tuple
|
from typing import List, Optional, Tuple, Union
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
@@ -34,10 +25,12 @@ from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
|||||||
from ocrmypdf.subprocess import get_version
|
from ocrmypdf.subprocess import get_version
|
||||||
from ocrmypdf.subprocess import run as external_run
|
from ocrmypdf.subprocess import run as external_run
|
||||||
|
|
||||||
|
DecFloat = Union[Decimal, float]
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def version():
|
def version() -> str:
|
||||||
return get_version('unpaper')
|
return get_version('unpaper')
|
||||||
|
|
||||||
|
|
||||||
@@ -63,23 +56,25 @@ def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]:
|
|||||||
except KeyError:
|
except KeyError:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
"Failed to convert image to a supported format."
|
"Failed to convert image to a supported format."
|
||||||
) from e
|
) from None
|
||||||
|
|
||||||
if im_modified or input_file.suffix != '.png':
|
if im_modified or input_file.suffix != '.pnm':
|
||||||
input_png = tmpdir / 'input.png'
|
input_pnm = tmpdir / 'input.pnm'
|
||||||
im.save(input_png, format='PNG', compress_level=1)
|
im.save(input_pnm, format='PPM')
|
||||||
else:
|
else:
|
||||||
# No changes, PNG input, just use the file we already have
|
# No changes, PNG input, just use the file we already have
|
||||||
input_png = input_file
|
input_pnm = input_file
|
||||||
output_pnm = tmpdir / f'output{suffix}'
|
output_pnm = tmpdir / f'output{suffix}'
|
||||||
return input_png, output_pnm
|
return input_pnm, output_pnm
|
||||||
|
|
||||||
|
|
||||||
def run(input_file, output_file, dpi, mode_args):
|
def run(
|
||||||
args_unpaper = ['unpaper', '-v', '--dpi', str(dpi)] + mode_args
|
input_file: Path, output_file: Path, *, dpi: DecFloat, mode_args: List[str]
|
||||||
|
) -> None:
|
||||||
|
args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args
|
||||||
|
|
||||||
with TemporaryDirectory() as tmpdir:
|
with TemporaryDirectory() as tmpdir:
|
||||||
input_png, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file)
|
input_pnm, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file)
|
||||||
|
|
||||||
# To prevent any shenanigans from accepting arbitrary parameters in
|
# To prevent any shenanigans from accepting arbitrary parameters in
|
||||||
# --unpaper-args, we:
|
# --unpaper-args, we:
|
||||||
@@ -88,41 +83,41 @@ def run(input_file, output_file, dpi, mode_args):
|
|||||||
# 3) append absolute paths for the input and output file
|
# 3) append absolute paths for the input and output file
|
||||||
# This should ensure that a user cannot clobber some other file with
|
# This should ensure that a user cannot clobber some other file with
|
||||||
# their unpaper arguments (whether intentionally or otherwise)
|
# their unpaper arguments (whether intentionally or otherwise)
|
||||||
args_unpaper.extend([os.fspath(input_png), os.fspath(output_pnm)])
|
args_unpaper.extend([os.fspath(input_pnm), os.fspath(output_pnm)])
|
||||||
|
external_run(
|
||||||
|
args_unpaper,
|
||||||
|
close_fds=True,
|
||||||
|
check=True,
|
||||||
|
stderr=STDOUT, # unpaper writes logging output to stdout and stderr
|
||||||
|
stdout=PIPE, # and cannot send file output to stdout
|
||||||
|
cwd=tmpdir,
|
||||||
|
logs_errors_to_stdout=True,
|
||||||
|
)
|
||||||
try:
|
try:
|
||||||
proc = external_run(
|
with Image.open(output_pnm) as imout:
|
||||||
args_unpaper,
|
imout.save(output_file, dpi=(dpi, dpi))
|
||||||
check=True,
|
except (FileNotFoundError, OSError):
|
||||||
close_fds=True,
|
raise SubprocessOutputError(
|
||||||
universal_newlines=True,
|
"unpaper: failed to produce the expected output file. "
|
||||||
stderr=STDOUT, # unpaper writes logging output to stdout and stderr
|
+ " Called with: "
|
||||||
cwd=tmpdir, # and cannot send file output to stdout
|
+ str(args_unpaper)
|
||||||
stdout=PIPE,
|
) from None
|
||||||
)
|
|
||||||
except CalledProcessError as e:
|
|
||||||
log.debug(e.stderr)
|
|
||||||
raise e from e
|
|
||||||
else:
|
|
||||||
log.debug(proc.stderr)
|
|
||||||
try:
|
|
||||||
with Image.open(output_pnm) as imout:
|
|
||||||
imout.save(output_file, dpi=(dpi, dpi))
|
|
||||||
except (FileNotFoundError, OSError):
|
|
||||||
raise SubprocessOutputError(
|
|
||||||
"unpaper: failed to produce the expected output file. "
|
|
||||||
+ " Called with: "
|
|
||||||
+ str(args_unpaper)
|
|
||||||
) from None
|
|
||||||
|
|
||||||
|
|
||||||
def validate_custom_args(args: str):
|
def validate_custom_args(args: str) -> List[str]:
|
||||||
unpaper_args = shlex.split(args)
|
unpaper_args = shlex.split(args)
|
||||||
if any('/' in arg for arg in unpaper_args):
|
if any(('/' in arg or arg == '.' or arg == '..') for arg in unpaper_args):
|
||||||
raise ValueError('No filenames allowed in --unpaper-args')
|
raise ValueError('No filenames allowed in --unpaper-args')
|
||||||
return unpaper_args
|
return unpaper_args
|
||||||
|
|
||||||
|
|
||||||
def clean(input_file, output_file, dpi, unpaper_args=None):
|
def clean(
|
||||||
|
input_file: Path,
|
||||||
|
output_file: Path,
|
||||||
|
*,
|
||||||
|
dpi: DecFloat,
|
||||||
|
unpaper_args: Optional[List[str]] = None,
|
||||||
|
):
|
||||||
default_args = [
|
default_args = [
|
||||||
'--layout',
|
'--layout',
|
||||||
'none',
|
'none',
|
||||||
@@ -136,4 +131,4 @@ def clean(input_file, output_file, dpi, unpaper_args=None):
|
|||||||
]
|
]
|
||||||
if not unpaper_args:
|
if not unpaper_args:
|
||||||
unpaper_args = default_args
|
unpaper_args = default_args
|
||||||
run(input_file, output_file, dpi, unpaper_args)
|
run(input_file, output_file, dpi=dpi, mode_args=unpaper_args)
|
||||||
|
|||||||
+60
-47
@@ -1,48 +1,44 @@
|
|||||||
# © 2018 James R. Barlow: github.com/jbarlow83
|
# © 2018 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
import uuid
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
from typing import Optional
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
|
from pikepdf.objects import Dictionary, Name
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
MAX_REPLACE_PAGES = 100
|
MAX_REPLACE_PAGES = 100
|
||||||
|
|
||||||
|
|
||||||
def _update_page_resources(*, page, font, font_key, procset):
|
def _ensure_dictionary(obj, name):
|
||||||
"""Update this page's fonts with a reference to the Glyphless font"""
|
if name not in obj:
|
||||||
|
obj[name] = Dictionary({})
|
||||||
|
return obj[name]
|
||||||
|
|
||||||
if '/Resources' not in page:
|
|
||||||
page['/Resources'] = pikepdf.Dictionary({})
|
def _update_resources(*, obj, font, font_key, procset):
|
||||||
resources = page['/Resources']
|
"""Update this obj's fonts with a reference to the Glyphless font.
|
||||||
try:
|
|
||||||
fonts = resources['/Font']
|
obj can be a page or Form XObject.
|
||||||
except KeyError:
|
"""
|
||||||
fonts = pikepdf.Dictionary({})
|
|
||||||
|
resources = _ensure_dictionary(obj, Name.Resources)
|
||||||
|
fonts = _ensure_dictionary(resources, Name.Font)
|
||||||
if font_key is not None and font_key not in fonts:
|
if font_key is not None and font_key not in fonts:
|
||||||
fonts[font_key] = font
|
fonts[font_key] = font
|
||||||
resources['/Font'] = fonts
|
|
||||||
|
|
||||||
# Reassign /ProcSet to one that just lists everything - ProcSet is
|
# Reassign /ProcSet to one that just lists everything - ProcSet is
|
||||||
# obsolete and doesn't matter but recommended for old viewer support
|
# obsolete and doesn't matter but recommended for old viewer support
|
||||||
resources['/ProcSet'] = procset
|
if procset:
|
||||||
|
resources['/ProcSet'] = procset
|
||||||
|
|
||||||
|
|
||||||
def strip_invisible_text(pdf, page):
|
def strip_invisible_text(pdf, page):
|
||||||
@@ -135,34 +131,38 @@ class OcrGrafter:
|
|||||||
del self.pdf_base.pages[-1]
|
del self.pdf_base.pages[-1]
|
||||||
emplaced_page = True
|
emplaced_page = True
|
||||||
|
|
||||||
|
# Calculate if the text is misaligned compared to the content
|
||||||
if emplaced_page:
|
if emplaced_page:
|
||||||
content_rotation = autorotate_correction
|
content_rotation = autorotate_correction
|
||||||
text_rotation = autorotate_correction
|
text_rotation = autorotate_correction
|
||||||
text_misaligned = (text_rotation - content_rotation) % 360
|
text_misaligned = (text_rotation - content_rotation) % 360
|
||||||
log.debug(
|
log.debug(
|
||||||
f"Rotations for page {pageno}: [text, auto, misalign, content] = "
|
f"Text rotation: (text, autorotate, content) -> text misalignment = "
|
||||||
f"{text_rotation}, {autorotate_correction}, "
|
f"({text_rotation}, {autorotate_correction}, {content_rotation}) -> {text_misaligned}"
|
||||||
f"{text_misaligned}, {content_rotation}"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
if textpdf and self.font:
|
if textpdf and self.font:
|
||||||
# Graft the text layer onto this page, whether new or old
|
# Graft the text layer onto this page, whether new or old, possibly
|
||||||
|
# rotating the text layer by the amount is misaligned.
|
||||||
strip_old = self.context.options.redo_ocr
|
strip_old = self.context.options.redo_ocr
|
||||||
self._graft_text_layer(
|
self._graft_text_layer(
|
||||||
page_num=pageno + 1,
|
page_num=pageno + 1,
|
||||||
textpdf=textpdf,
|
textpdf=textpdf,
|
||||||
font=self.font,
|
font=self.font,
|
||||||
font_key=self.font_key,
|
font_key=self.font_key,
|
||||||
rotation=text_misaligned,
|
text_rotation=text_misaligned,
|
||||||
procset=self.procset,
|
procset=self.procset,
|
||||||
strip_old_text=strip_old,
|
strip_old_text=strip_old,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Correct the rotation if applicable
|
# Correct the overall page rotation if needed, now that the text and content
|
||||||
self.pdf_base.pages[pageno].Rotate = (
|
# are aligned
|
||||||
content_rotation - autorotate_correction
|
page_rotation = (content_rotation - autorotate_correction) % 360
|
||||||
) % 360
|
self.pdf_base.pages[pageno].Rotate = page_rotation
|
||||||
|
log.debug(
|
||||||
|
f"Page rotation: (content, auto) -> page = "
|
||||||
|
f"({content_rotation}, {autorotate_correction}) -> {page_rotation}"
|
||||||
|
)
|
||||||
if self.emplacements % MAX_REPLACE_PAGES == 0:
|
if self.emplacements % MAX_REPLACE_PAGES == 0:
|
||||||
self.save_and_reload()
|
self.save_and_reload()
|
||||||
|
|
||||||
@@ -175,13 +175,13 @@ class OcrGrafter:
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
page0 = self.pdf_base.pages[0]
|
page0 = self.pdf_base.pages[0]
|
||||||
_update_page_resources(
|
_update_resources(
|
||||||
page=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
obj=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
||||||
)
|
)
|
||||||
|
|
||||||
# We cannot read and write the same file, that will corrupt it
|
# We cannot read and write the same file, that will corrupt it
|
||||||
# but we don't to keep more copies than we need to. Delete intermediates.
|
# but we don't to keep more copies than we need to. Delete intermediates.
|
||||||
# {interim_count} is the opened file we were updateing
|
# {interim_count} is the opened file we were updating
|
||||||
# {interim_count - 1} can be deleted
|
# {interim_count - 1} can be deleted
|
||||||
# {interim_count + 1} is the new file will produce and open
|
# {interim_count + 1} is the new file will produce and open
|
||||||
old_file = self.output_file.with_suffix(f'.working{self.interim_count - 1}.pdf')
|
old_file = self.output_file.with_suffix(f'.working{self.interim_count - 1}.pdf')
|
||||||
@@ -216,6 +216,7 @@ class OcrGrafter:
|
|||||||
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
||||||
except (AttributeError, IndexError, KeyError):
|
except (AttributeError, IndexError, KeyError):
|
||||||
return None, None
|
return None, None
|
||||||
|
pdf_text_font = None
|
||||||
for f in possible_font_names:
|
for f in possible_font_names:
|
||||||
pdf_text_font = pdf_text_fonts.get(f, None)
|
pdf_text_font = pdf_text_fonts.get(f, None)
|
||||||
if pdf_text_font is not None:
|
if pdf_text_font is not None:
|
||||||
@@ -236,7 +237,7 @@ class OcrGrafter:
|
|||||||
font: pikepdf.Object,
|
font: pikepdf.Object,
|
||||||
font_key: pikepdf.Object,
|
font_key: pikepdf.Object,
|
||||||
procset: pikepdf.Object,
|
procset: pikepdf.Object,
|
||||||
rotation: int,
|
text_rotation: int,
|
||||||
strip_old_text: bool,
|
strip_old_text: bool,
|
||||||
):
|
):
|
||||||
"""Insert the text layer from text page 0 on to pdf_base at page_num"""
|
"""Insert the text layer from text page 0 on to pdf_base at page_num"""
|
||||||
@@ -266,13 +267,13 @@ class OcrGrafter:
|
|||||||
corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1])
|
corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1])
|
||||||
# -rotation because the input is a clockwise angle and this formula
|
# -rotation because the input is a clockwise angle and this formula
|
||||||
# uses CCW
|
# uses CCW
|
||||||
rotation = -rotation % 360
|
text_rotation = -text_rotation % 360
|
||||||
rotate = pikepdf.PdfMatrix().rotated(rotation)
|
rotate = pikepdf.PdfMatrix().rotated(text_rotation)
|
||||||
|
|
||||||
# Because of rounding of DPI, we might get a text layer that is not
|
# Because of rounding of DPI, we might get a text layer that is not
|
||||||
# identically sized to the target page. Scale to adjust. Normally this
|
# identically sized to the target page. Scale to adjust. Normally this
|
||||||
# is within 0.998.
|
# is within 0.998.
|
||||||
if rotation in (90, 270):
|
if text_rotation in (90, 270):
|
||||||
wt, ht = ht, wt
|
wt, ht = ht, wt
|
||||||
scale_x = wp / wt
|
scale_x = wp / wt
|
||||||
scale_y = hp / ht
|
scale_y = hp / ht
|
||||||
@@ -285,17 +286,29 @@ class OcrGrafter:
|
|||||||
# finally move the lower left corner to match the mediabox
|
# finally move the lower left corner to match the mediabox
|
||||||
ctm = translate @ rotate @ scale @ untranslate @ corner
|
ctm = translate @ rotate @ scale @ untranslate @ corner
|
||||||
|
|
||||||
pdf_text_contents = (
|
base_resources = _ensure_dictionary(base_page, Name.Resources)
|
||||||
b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n'
|
base_xobjs = _ensure_dictionary(base_resources, Name.XObject)
|
||||||
|
text_xobj_name = Name('/' + str(uuid.uuid4()))
|
||||||
|
xobj = self.pdf_base.make_stream(pdf_text_contents)
|
||||||
|
base_xobjs[text_xobj_name] = xobj
|
||||||
|
xobj.Type = Name.XObject
|
||||||
|
xobj.Subtype = Name.Form
|
||||||
|
xobj.FormType = 1
|
||||||
|
xobj.BBox = mediabox
|
||||||
|
_update_resources(
|
||||||
|
obj=xobj, font=font, font_key=font_key, procset=[Name.PDF]
|
||||||
)
|
)
|
||||||
|
|
||||||
new_text_layer = pikepdf.Stream(self.pdf_base, pdf_text_contents)
|
pdf_draw_xobj = (
|
||||||
|
(b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
||||||
|
)
|
||||||
|
new_text_layer = pikepdf.Stream(self.pdf_base, pdf_draw_xobj)
|
||||||
|
|
||||||
if strip_old_text:
|
if strip_old_text:
|
||||||
strip_invisible_text(self.pdf_base, base_page)
|
strip_invisible_text(self.pdf_base, base_page)
|
||||||
|
|
||||||
base_page.page_contents_add(new_text_layer, prepend=True)
|
base_page.page_contents_add(new_text_layer, prepend=True)
|
||||||
|
|
||||||
_update_page_resources(
|
_update_resources(
|
||||||
page=base_page, font=font, font_key=font_key, procset=procset
|
obj=base_page, font=font, font_key=font_key, procset=procset
|
||||||
)
|
)
|
||||||
|
|||||||
+32
-18
@@ -1,34 +1,31 @@
|
|||||||
# © 2018 James R. Barlow: github.com/jbarlow83
|
# © 2018 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
from argparse import Namespace
|
from argparse import Namespace
|
||||||
from copy import copy
|
from copy import copy
|
||||||
from io import IOBase
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Iterator
|
from typing import Iterator
|
||||||
|
|
||||||
|
from pluggy import PluginManager
|
||||||
|
|
||||||
from ocrmypdf.pdfinfo import PdfInfo
|
from ocrmypdf.pdfinfo import PdfInfo
|
||||||
|
from ocrmypdf.pdfinfo.info import PageInfo
|
||||||
|
|
||||||
|
|
||||||
class PdfContext:
|
class PdfContext:
|
||||||
"""Holds our context for a particular run of the pipeline"""
|
"""Holds the context for a particular run of the pipeline."""
|
||||||
|
|
||||||
|
options: Namespace #: The specified options for processing this PDF.
|
||||||
|
origin: Path #: The filename of the original input file.
|
||||||
|
pdfinfo: PdfInfo #: Detailed data for this PDF.
|
||||||
|
plugin_manager: PluginManager #: PluginManager for processing the current PDF.
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
@@ -45,21 +42,33 @@ class PdfContext:
|
|||||||
self.plugin_manager = plugin_manager
|
self.plugin_manager = plugin_manager
|
||||||
|
|
||||||
def get_path(self, name: str) -> Path:
|
def get_path(self, name: str) -> Path:
|
||||||
|
"""Generate a ``Path`` for an intermediate file involved in processing.
|
||||||
|
|
||||||
|
The path will be in a temporary folder that is common for all processing
|
||||||
|
of this particular PDF.
|
||||||
|
"""
|
||||||
return self.work_folder / name
|
return self.work_folder / name
|
||||||
|
|
||||||
def get_page_contexts(self) -> Iterator['PageContext']:
|
def get_page_contexts(self) -> Iterator['PageContext']:
|
||||||
|
"""Get all ``PageContext`` for this PDF."""
|
||||||
npages = len(self.pdfinfo)
|
npages = len(self.pdfinfo)
|
||||||
for n in range(npages):
|
for n in range(npages):
|
||||||
yield PageContext(self, n)
|
yield PageContext(self, n)
|
||||||
|
|
||||||
|
|
||||||
class PageContext:
|
class PageContext:
|
||||||
"""Holds our context for a page
|
"""Holds our context for a page.
|
||||||
|
|
||||||
Must be pickable, so stores only intrinsic/simple data elements or those
|
Must be pickable, so stores only intrinsic/simple data elements or those
|
||||||
capable of their serializing themselves via __getstate__.
|
capable of their serializing themselves via ``__getstate__``.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
options: Namespace #: The specified options for processing this PDF.
|
||||||
|
origin: Path #: The filename of the original input file.
|
||||||
|
pageno: int #: This page number (zero-based).
|
||||||
|
pageinfo: PageInfo #: Information on this page.
|
||||||
|
plugin_manager: PluginManager #: PluginManager for processing the current PDF.
|
||||||
|
|
||||||
def __init__(self, pdf_context: PdfContext, pageno):
|
def __init__(self, pdf_context: PdfContext, pageno):
|
||||||
self.work_folder = pdf_context.work_folder
|
self.work_folder = pdf_context.work_folder
|
||||||
self.origin = pdf_context.origin
|
self.origin = pdf_context.origin
|
||||||
@@ -69,6 +78,11 @@ class PageContext:
|
|||||||
self.plugin_manager = pdf_context.plugin_manager
|
self.plugin_manager = pdf_context.plugin_manager
|
||||||
|
|
||||||
def get_path(self, name: str) -> Path:
|
def get_path(self, name: str) -> Path:
|
||||||
|
"""Generate a ``Path`` for a file that is part of processing this page.
|
||||||
|
|
||||||
|
The path will be based in a common temporary folder and have a prefix based
|
||||||
|
on the page number.
|
||||||
|
"""
|
||||||
return self.work_folder / ("%06d_%s" % (self.pageno + 1, name))
|
return self.work_folder / ("%06d_%s" % (self.pageno + 1, name))
|
||||||
|
|
||||||
def __getstate__(self):
|
def __getstate__(self):
|
||||||
|
|||||||
@@ -1,19 +1,9 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import sys
|
import sys
|
||||||
|
|||||||
+119
-62
@@ -1,19 +1,9 @@
|
|||||||
# © 2016 James R. Barlow: github.com/jbarlow83
|
# © 2016 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -23,7 +13,7 @@ from contextlib import suppress
|
|||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj
|
from shutil import copyfileobj
|
||||||
from typing import BinaryIO, Dict, Iterable, Optional, Union, cast
|
from typing import Dict, Iterable, Optional
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
import pikepdf
|
import pikepdf
|
||||||
@@ -31,6 +21,7 @@ from pikepdf.models.metadata import encode_pdf_date
|
|||||||
from PIL import Image, ImageColor, ImageDraw
|
from PIL import Image, ImageColor, ImageDraw
|
||||||
|
|
||||||
from ocrmypdf import leptonica
|
from ocrmypdf import leptonica
|
||||||
|
from ocrmypdf._concurrent import Executor
|
||||||
from ocrmypdf._exec import unpaper
|
from ocrmypdf._exec import unpaper
|
||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._version import PROGRAM_NAME
|
from ocrmypdf._version import PROGRAM_NAME
|
||||||
@@ -155,11 +146,13 @@ def triage(original_filename, input_file, output_file, options):
|
|||||||
|
|
||||||
def get_pdfinfo(
|
def get_pdfinfo(
|
||||||
input_file,
|
input_file,
|
||||||
|
*,
|
||||||
|
executor: Executor,
|
||||||
detailed_analysis=False,
|
detailed_analysis=False,
|
||||||
progbar=False,
|
progbar=False,
|
||||||
max_workers=None,
|
max_workers=None,
|
||||||
check_pages=None,
|
check_pages=None,
|
||||||
):
|
) -> PdfInfo:
|
||||||
try:
|
try:
|
||||||
return PdfInfo(
|
return PdfInfo(
|
||||||
input_file,
|
input_file,
|
||||||
@@ -167,11 +160,12 @@ def get_pdfinfo(
|
|||||||
progbar=progbar,
|
progbar=progbar,
|
||||||
max_workers=max_workers,
|
max_workers=max_workers,
|
||||||
check_pages=check_pages,
|
check_pages=check_pages,
|
||||||
|
executor=executor,
|
||||||
)
|
)
|
||||||
except pikepdf.PasswordError:
|
except pikepdf.PasswordError as e:
|
||||||
raise EncryptedPdfError()
|
raise EncryptedPdfError() from e
|
||||||
except pikepdf.PdfError:
|
except pikepdf.PdfError as e:
|
||||||
raise InputFileError()
|
raise InputFileError() from e
|
||||||
|
|
||||||
|
|
||||||
def validate_pdfinfo_options(context: PdfContext):
|
def validate_pdfinfo_options(context: PdfContext):
|
||||||
@@ -215,17 +209,21 @@ def validate_pdfinfo_options(context: PdfContext):
|
|||||||
context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options)
|
context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options)
|
||||||
|
|
||||||
|
|
||||||
|
def _vector_page_dpi(pageinfo):
|
||||||
|
return VECTOR_PAGE_DPI if pageinfo.has_vector or pageinfo.has_text else 0.0
|
||||||
|
|
||||||
|
|
||||||
def get_page_dpi(pageinfo, options):
|
def get_page_dpi(pageinfo, options):
|
||||||
"Get the DPI when nonsquare DPI is tolerable"
|
"Get the DPI when nonsquare DPI is tolerable"
|
||||||
xres = max(
|
xres = max(
|
||||||
pageinfo.dpi.x or VECTOR_PAGE_DPI,
|
pageinfo.dpi.x or VECTOR_PAGE_DPI,
|
||||||
options.oversample or 0.0,
|
options.oversample or 0.0,
|
||||||
VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0,
|
_vector_page_dpi(pageinfo),
|
||||||
)
|
)
|
||||||
yres = max(
|
yres = max(
|
||||||
pageinfo.dpi.y or VECTOR_PAGE_DPI,
|
pageinfo.dpi.y or VECTOR_PAGE_DPI,
|
||||||
options.oversample or 0,
|
options.oversample or 0,
|
||||||
VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0,
|
_vector_page_dpi(pageinfo),
|
||||||
)
|
)
|
||||||
return Resolution(float(xres), float(yres))
|
return Resolution(float(xres), float(yres))
|
||||||
|
|
||||||
@@ -239,7 +237,7 @@ def get_page_square_dpi(pageinfo, options) -> Resolution:
|
|||||||
max(
|
max(
|
||||||
(xres * userunit) or VECTOR_PAGE_DPI,
|
(xres * userunit) or VECTOR_PAGE_DPI,
|
||||||
(yres * userunit) or VECTOR_PAGE_DPI,
|
(yres * userunit) or VECTOR_PAGE_DPI,
|
||||||
VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0,
|
_vector_page_dpi(pageinfo),
|
||||||
options.oversample or 0.0,
|
options.oversample or 0.0,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
@@ -252,7 +250,7 @@ def get_canvas_square_dpi(pageinfo, options) -> Resolution:
|
|||||||
max(
|
max(
|
||||||
(pageinfo.dpi.x) or VECTOR_PAGE_DPI,
|
(pageinfo.dpi.x) or VECTOR_PAGE_DPI,
|
||||||
(pageinfo.dpi.y) or VECTOR_PAGE_DPI,
|
(pageinfo.dpi.y) or VECTOR_PAGE_DPI,
|
||||||
VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0,
|
_vector_page_dpi(pageinfo),
|
||||||
options.oversample or 0.0,
|
options.oversample or 0.0,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
@@ -342,8 +340,10 @@ def rasterize_preview(input_file: Path, page_context: PageContext):
|
|||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
raster_device='jpeggray',
|
raster_device='jpeggray',
|
||||||
raster_dpi=canvas_dpi,
|
raster_dpi=canvas_dpi,
|
||||||
page_dpi=page_dpi,
|
|
||||||
pageno=page_context.pageinfo.pageno + 1,
|
pageno=page_context.pageinfo.pageno + 1,
|
||||||
|
page_dpi=page_dpi,
|
||||||
|
rotation=0,
|
||||||
|
filter_vector=False,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
@@ -443,7 +443,7 @@ def rasterize(
|
|||||||
|
|
||||||
device = colorspaces[device_idx]
|
device = colorspaces[device_idx]
|
||||||
|
|
||||||
log.debug(f"Rasterize with {device}")
|
log.debug(f"Rasterize with {device}, rotation {correction}")
|
||||||
|
|
||||||
# Produce the page image with square resolution or else deskew and OCR
|
# Produce the page image with square resolution or else deskew and OCR
|
||||||
# will not work properly.
|
# will not work properly.
|
||||||
@@ -483,7 +483,12 @@ def preprocess_deskew(input_file: Path, page_context: PageContext):
|
|||||||
def preprocess_clean(input_file: Path, page_context: PageContext):
|
def preprocess_clean(input_file: Path, page_context: PageContext):
|
||||||
output_file = page_context.get_path('pp_clean.png')
|
output_file = page_context.get_path('pp_clean.png')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
unpaper.clean(input_file, output_file, dpi.x, page_context.options.unpaper_args)
|
unpaper.clean(
|
||||||
|
input_file,
|
||||||
|
output_file,
|
||||||
|
dpi=dpi.x,
|
||||||
|
unpaper_args=page_context.options.unpaper_args,
|
||||||
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
@@ -530,7 +535,9 @@ def create_ocr_image(image: Path, page_context: PageContext):
|
|||||||
if options.threshold:
|
if options.threshold:
|
||||||
pix = leptonica.Pix.frompil(im)
|
pix = leptonica.Pix.frompil(im)
|
||||||
pix = pix.masked_threshold_on_background_norm()
|
pix = pix.masked_threshold_on_background_norm()
|
||||||
im = pix.topil()
|
im_pix = pix.topil()
|
||||||
|
im_pix.info['dpi'] = im.info['dpi']
|
||||||
|
im = im_pix
|
||||||
|
|
||||||
del draw
|
del draw
|
||||||
|
|
||||||
@@ -585,7 +592,9 @@ def create_visible_page_jpg(image: Path, page_context: PageContext) -> Path:
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def create_pdf_page_from_image(image: Path, page_context: PageContext):
|
def create_pdf_page_from_image(
|
||||||
|
image: Path, page_context: PageContext, orientation_correction
|
||||||
|
):
|
||||||
# We rasterize a square DPI version of each page because most image
|
# We rasterize a square DPI version of each page because most image
|
||||||
# processing tools don't support rectangular DPI. Use the square DPI as it
|
# processing tools don't support rectangular DPI. Use the square DPI as it
|
||||||
# accurately describes the image. It would be possible to resample the image
|
# accurately describes the image. It would be possible to resample the image
|
||||||
@@ -593,28 +602,42 @@ def create_pdf_page_from_image(image: Path, page_context: PageContext):
|
|||||||
# except that the hocr renderer does not understand non-square DPI. The
|
# except that the hocr renderer does not understand non-square DPI. The
|
||||||
# sandwich renderer would be fine.
|
# sandwich renderer would be fine.
|
||||||
output_file = page_context.get_path('visible.pdf')
|
output_file = page_context.get_path('visible.pdf')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
|
||||||
layout_fun = img2pdf.get_fixed_dpi_layout_fun(dpi)
|
pageinfo = page_context.pageinfo
|
||||||
|
pagesize = 72.0 * float(pageinfo.width_inches), 72.0 * float(pageinfo.height_inches)
|
||||||
|
effective_rotation = (pageinfo.rotation - orientation_correction) % 360
|
||||||
|
if effective_rotation % 180 == 90:
|
||||||
|
pagesize = pagesize[1], pagesize[0]
|
||||||
|
|
||||||
# This create a single page PDF
|
# This create a single page PDF
|
||||||
with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf:
|
with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf:
|
||||||
log.debug('convert')
|
log.debug('convert')
|
||||||
|
|
||||||
|
layout_fun = img2pdf.get_layout_fun(pagesize)
|
||||||
img2pdf.convert(
|
img2pdf.convert(
|
||||||
imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf
|
imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf
|
||||||
)
|
)
|
||||||
log.debug('convert done')
|
log.debug('convert done')
|
||||||
|
|
||||||
|
output_file = page_context.plugin_manager.hook.filter_pdf_page(
|
||||||
|
page=page_context, image_filename=image, output_pdf=output_file
|
||||||
|
)
|
||||||
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def render_hocr_page(hocr: Path, page_context: PageContext):
|
def render_hocr_page(hocr: Path, page_context: PageContext):
|
||||||
|
options = page_context.options
|
||||||
output_file = page_context.get_path('ocr_hocr.pdf')
|
output_file = page_context.get_path('ocr_hocr.pdf')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context.pageinfo, options)
|
||||||
hocrtransform = HocrTransform(hocr, dpi.x) # square
|
debug_mode = options.pdf_renderer == 'hocrdebug'
|
||||||
|
|
||||||
|
hocrtransform = HocrTransform(hocr_filename=hocr, dpi=dpi.x) # square
|
||||||
hocrtransform.to_pdf(
|
hocrtransform.to_pdf(
|
||||||
output_file,
|
out_filename=output_file,
|
||||||
image_filename=None,
|
image_filename=None,
|
||||||
show_bounding_boxes=False,
|
show_bounding_boxes=False if not debug_mode else True,
|
||||||
invisible_text=True,
|
invisible_text=True if not debug_mode else False,
|
||||||
interword_spaces=True,
|
interword_spaces=True,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
@@ -663,11 +686,6 @@ def get_docinfo(base_pdf: pikepdf.Pdf, context: PdfContext) -> Dict[str, str]:
|
|||||||
|
|
||||||
pdfmark['/Creator'] = f'{PROGRAM_NAME} {VERSION} / {creator_tag}'
|
pdfmark['/Creator'] = f'{PROGRAM_NAME} {VERSION} / {creator_tag}'
|
||||||
pdfmark['/Producer'] = f'pikepdf {pikepdf.__version__}'
|
pdfmark['/Producer'] = f'pikepdf {pikepdf.__version__}'
|
||||||
if 'OCRMYPDF_CREATOR' in os.environ:
|
|
||||||
pdfmark['/Creator'] = os.environ['OCRMYPDF_CREATOR']
|
|
||||||
if 'OCRMYPDF_PRODUCER' in os.environ:
|
|
||||||
pdfmark['/Producer'] = os.environ['OCRMYPDF_PRODUCER']
|
|
||||||
|
|
||||||
pdfmark['/ModDate'] = encode_pdf_date(datetime.now(timezone.utc))
|
pdfmark['/ModDate'] = encode_pdf_date(datetime.now(timezone.utc))
|
||||||
return pdfmark
|
return pdfmark
|
||||||
|
|
||||||
@@ -715,6 +733,11 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext):
|
|||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
compression=options.pdfa_image_compression,
|
compression=options.pdfa_image_compression,
|
||||||
pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3
|
pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3
|
||||||
|
progressbar_class=(
|
||||||
|
context.plugin_manager.hook.get_progressbar_class()
|
||||||
|
if options.progress_bar
|
||||||
|
else None
|
||||||
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
return output_file
|
return output_file
|
||||||
@@ -727,6 +750,24 @@ def should_linearize(working_file: Path, context: PdfContext):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def get_pdf_save_settings(output_type: str):
|
||||||
|
if output_type == 'pdfa-1':
|
||||||
|
# Trigger recompression to ensure object streams are removed, because
|
||||||
|
# Acrobat complains about them in PDF/A-1b validation.
|
||||||
|
return dict(
|
||||||
|
preserve_pdfa=True,
|
||||||
|
compress_streams=True,
|
||||||
|
stream_decode_level=pikepdf.StreamDecodeLevel.generalized,
|
||||||
|
object_stream_mode=pikepdf.ObjectStreamMode.disable,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
return dict(
|
||||||
|
preserve_pdfa=True,
|
||||||
|
compress_streams=True,
|
||||||
|
object_stream_mode=(pikepdf.ObjectStreamMode.generate),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def metadata_fixup(working_file: Path, context: PdfContext):
|
def metadata_fixup(working_file: Path, context: PdfContext):
|
||||||
output_file = context.get_path('metafix.pdf')
|
output_file = context.get_path('metafix.pdf')
|
||||||
options = context.options
|
options = context.options
|
||||||
@@ -757,23 +798,21 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
|||||||
if 'xmp:CreateDate' not in meta:
|
if 'xmp:CreateDate' not in meta:
|
||||||
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
||||||
|
|
||||||
# Ghostscript likes to set title to Untitled if omitted from input.
|
with original.open_metadata(
|
||||||
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
||||||
# and the XMP Spec do not make this recommendation.
|
) as meta_original:
|
||||||
if meta.get('dc:title') == 'Untitled':
|
if meta.get('dc:title') == 'Untitled':
|
||||||
with original.open_metadata() as original_meta:
|
# Ghostscript likes to set title to Untitled if omitted from input.
|
||||||
if 'dc:title' not in original_meta:
|
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
||||||
|
# and the XMP Spec do not make this recommendation.
|
||||||
|
if 'dc:title' not in meta_original:
|
||||||
del meta['dc:title']
|
del meta['dc:title']
|
||||||
|
missing = set(meta_original.keys()) - set(meta.keys())
|
||||||
meta_original = original.open_metadata()
|
report_on_metadata(missing)
|
||||||
missing = set(meta_original.keys()) - set(meta.keys())
|
|
||||||
report_on_metadata(missing)
|
|
||||||
|
|
||||||
pdf.save(
|
pdf.save(
|
||||||
output_file,
|
output_file,
|
||||||
compress_streams=True,
|
**get_pdf_save_settings(options.output_type),
|
||||||
preserve_pdfa=True,
|
|
||||||
object_stream_mode=pikepdf.ObjectStreamMode.generate,
|
|
||||||
linearize=( # Don't linearize if optimize() will be linearizing too
|
linearize=( # Don't linearize if optimize() will be linearizing too
|
||||||
should_linearize(working_file, context)
|
should_linearize(working_file, context)
|
||||||
if options.optimize == 0
|
if options.optimize == 0
|
||||||
@@ -784,23 +823,37 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def optimize_pdf(input_file: Path, context: PdfContext):
|
def optimize_pdf(input_file: Path, context: PdfContext, executor: Executor):
|
||||||
output_file = context.get_path('optimize.pdf')
|
output_file = context.get_path('optimize.pdf')
|
||||||
save_settings = dict(
|
save_settings = dict(
|
||||||
compress_streams=True,
|
|
||||||
preserve_pdfa=True,
|
|
||||||
object_stream_mode=pikepdf.ObjectStreamMode.generate,
|
|
||||||
linearize=should_linearize(input_file, context),
|
linearize=should_linearize(input_file, context),
|
||||||
|
**get_pdf_save_settings(context.options.output_type),
|
||||||
)
|
)
|
||||||
optimize(input_file, output_file, context, save_settings)
|
optimize(input_file, output_file, context, save_settings, executor)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
|
def enumerate_compress_ranges(iterable):
|
||||||
|
skipped_from, index = None, None
|
||||||
|
for index, txt_file in enumerate(iterable):
|
||||||
|
index += 1
|
||||||
|
if txt_file:
|
||||||
|
if skipped_from is not None:
|
||||||
|
yield (skipped_from, index - 1), None
|
||||||
|
skipped_from = None
|
||||||
|
yield (index, index), txt_file
|
||||||
|
else:
|
||||||
|
if skipped_from is None:
|
||||||
|
skipped_from = index
|
||||||
|
if skipped_from is not None:
|
||||||
|
yield (skipped_from, index), None
|
||||||
|
|
||||||
|
|
||||||
def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
||||||
output_file = context.get_path('sidecar.txt')
|
output_file = context.get_path('sidecar.txt')
|
||||||
with open(output_file, 'w', encoding="utf-8") as stream:
|
with open(output_file, 'w', encoding="utf-8") as stream:
|
||||||
for page_num, txt_file in enumerate(txt_files):
|
for (frm, to), txt_file in enumerate_compress_ranges(txt_files):
|
||||||
if page_num != 0:
|
if frm != 1:
|
||||||
stream.write('\f') # Form feed between pages
|
stream.write('\f') # Form feed between pages
|
||||||
if txt_file:
|
if txt_file:
|
||||||
with open(txt_file, 'r', encoding="utf-8") as in_:
|
with open(txt_file, 'r', encoding="utf-8") as in_:
|
||||||
@@ -813,7 +866,11 @@ def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
|||||||
else:
|
else:
|
||||||
stream.write(txt)
|
stream.write(txt)
|
||||||
else:
|
else:
|
||||||
stream.write(f'[OCR skipped on page {(page_num + 1)}]')
|
if frm != to:
|
||||||
|
pages = f'{frm}-{to}'
|
||||||
|
else:
|
||||||
|
pages = f'{frm}'
|
||||||
|
stream.write(f'[OCR skipped on page(s) {pages}]')
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,30 +1,21 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import importlib
|
import importlib
|
||||||
import importlib.util
|
import importlib.util
|
||||||
|
import pkgutil
|
||||||
import sys
|
import sys
|
||||||
from functools import partial
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Callable, List, Tuple, Union
|
from typing import List, Tuple, Union
|
||||||
|
|
||||||
import pluggy
|
import pluggy
|
||||||
|
|
||||||
|
import ocrmypdf.builtin_plugins
|
||||||
from ocrmypdf import pluginspec
|
from ocrmypdf import pluginspec
|
||||||
from ocrmypdf.cli import get_parser, plugins_only_parser
|
from ocrmypdf.cli import get_parser, plugins_only_parser
|
||||||
|
|
||||||
@@ -40,62 +31,80 @@ class OcrmypdfPluginManager(pluggy.PluginManager):
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self, *args, setup_func: Callable[[pluggy.PluginManager], None], **kwargs
|
self,
|
||||||
|
*args,
|
||||||
|
plugins: List[Union[str, Path]],
|
||||||
|
builtins: bool = True,
|
||||||
|
**kwargs,
|
||||||
):
|
):
|
||||||
self._init_args = args
|
self.__init_args = args
|
||||||
self._setup_func = setup_func
|
self.__init_kwargs = kwargs
|
||||||
self._init_kwargs = kwargs
|
self.__plugins = plugins
|
||||||
|
self.__builtins = builtins
|
||||||
super().__init__(*args, **kwargs)
|
super().__init__(*args, **kwargs)
|
||||||
setup_func(self)
|
self.setup_plugins()
|
||||||
|
|
||||||
def __getstate__(self):
|
def __getstate__(self):
|
||||||
state = dict(
|
state = dict(
|
||||||
_init_args=self._init_args,
|
init_args=self.__init_args,
|
||||||
_setup_func=self._setup_func,
|
plugins=self.__plugins,
|
||||||
_init_kwargs=self._init_kwargs,
|
builtins=self.__builtins,
|
||||||
|
init_kwargs=self.__init_kwargs,
|
||||||
)
|
)
|
||||||
return state
|
return state
|
||||||
|
|
||||||
def __setstate__(self, state):
|
def __setstate__(self, state):
|
||||||
self.__init__(
|
self.__init__(
|
||||||
*state['_init_args'],
|
*state['init_args'],
|
||||||
setup_func=state['_setup_func'],
|
plugins=state['plugins'],
|
||||||
**state['_init_kwargs'],
|
builtins=state['builtins'],
|
||||||
|
**state['init_kwargs'],
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def setup_plugins(self):
|
||||||
|
self.add_hookspecs(pluginspec)
|
||||||
|
|
||||||
def _setup_plugins(
|
# 1. Register builtins
|
||||||
pm: pluggy.PluginManager, plugins: List[Union[str, Path]], builtins: bool = True
|
if self.__builtins:
|
||||||
):
|
for module in sorted(
|
||||||
pm.add_hookspecs(pluginspec)
|
pkgutil.iter_modules(ocrmypdf.builtin_plugins.__path__)
|
||||||
|
):
|
||||||
|
name = f'ocrmypdf.builtin_plugins.{module.name}'
|
||||||
|
module = importlib.import_module(name)
|
||||||
|
self.register(module)
|
||||||
|
|
||||||
all_plugins: List[Union[str, Path]] = []
|
# 2. Install semfree if needed
|
||||||
if builtins:
|
try:
|
||||||
all_plugins.extend(
|
# pylint: disable=import-outside-toplevel
|
||||||
[
|
from multiprocessing.synchronize import SemLock
|
||||||
'ocrmypdf.builtin_plugins.ghostscript',
|
|
||||||
'ocrmypdf.builtin_plugins.tesseract_ocr',
|
del SemLock
|
||||||
]
|
except ImportError:
|
||||||
)
|
self.register(importlib.import_module('ocrmypdf.extra_plugins.semfree'))
|
||||||
all_plugins.extend(plugins)
|
|
||||||
for name in all_plugins:
|
# 3. Register setuptools plugins
|
||||||
if isinstance(name, Path) or name.endswith('.py'):
|
self.load_setuptools_entrypoints('ocrmypdf')
|
||||||
# Import by filename
|
|
||||||
module_name = Path(name).stem
|
# 4. Register plugins specified on command line
|
||||||
spec = importlib.util.spec_from_file_location(module_name, name)
|
for name in self.__plugins:
|
||||||
module = importlib.util.module_from_spec(spec)
|
if isinstance(name, Path) or name.endswith('.py'):
|
||||||
sys.modules[module_name] = module
|
# Import by filename
|
||||||
spec.loader.exec_module(module)
|
module_name = Path(name).stem
|
||||||
else:
|
spec = importlib.util.spec_from_file_location(module_name, name)
|
||||||
# Import by dotted module name
|
module = importlib.util.module_from_spec(spec)
|
||||||
module = importlib.import_module(name)
|
sys.modules[module_name] = module
|
||||||
pm.register(module)
|
spec.loader.exec_module(module)
|
||||||
|
else:
|
||||||
|
# Import by dotted module name
|
||||||
|
module = importlib.import_module(name)
|
||||||
|
self.register(module)
|
||||||
|
|
||||||
|
|
||||||
def get_plugin_manager(plugins: List[str], builtins=True):
|
def get_plugin_manager(plugins: List[Union[str, Path]], builtins=True):
|
||||||
pm = OcrmypdfPluginManager(
|
pm = OcrmypdfPluginManager(
|
||||||
project_name='ocrmypdf',
|
project_name='ocrmypdf',
|
||||||
setup_func=partial(_setup_plugins, plugins=plugins, builtins=builtins),
|
plugins=plugins,
|
||||||
|
builtins=builtins,
|
||||||
)
|
)
|
||||||
return pm
|
return pm
|
||||||
|
|
||||||
|
|||||||
+64
-55
@@ -1,19 +1,9 @@
|
|||||||
# © 2016 James R. Barlow: github.com/jbarlow83
|
# © 2016 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import logging.handlers
|
import logging.handlers
|
||||||
@@ -25,10 +15,9 @@ from pathlib import Path
|
|||||||
from tempfile import mkdtemp
|
from tempfile import mkdtemp
|
||||||
from typing import List, NamedTuple, Optional, Tuple
|
from typing import List, NamedTuple, Optional, Tuple
|
||||||
|
|
||||||
import pikepdf
|
|
||||||
import PIL
|
import PIL
|
||||||
|
|
||||||
from ocrmypdf._concurrent import exec_progress_pool
|
from ocrmypdf._concurrent import Executor, setup_executor
|
||||||
from ocrmypdf._graft import OcrGrafter
|
from ocrmypdf._graft import OcrGrafter
|
||||||
from ocrmypdf._jobcontext import PageContext, PdfContext, cleanup_working_files
|
from ocrmypdf._jobcontext import PageContext, PdfContext, cleanup_working_files
|
||||||
from ocrmypdf._logging import PageNumberFilter
|
from ocrmypdf._logging import PageNumberFilter
|
||||||
@@ -65,6 +54,7 @@ from ocrmypdf._validation import (
|
|||||||
)
|
)
|
||||||
from ocrmypdf.exceptions import ExitCode, ExitCodeException
|
from ocrmypdf.exceptions import ExitCode, ExitCodeException
|
||||||
from ocrmypdf.helpers import (
|
from ocrmypdf.helpers import (
|
||||||
|
NeverRaise,
|
||||||
available_cpu_count,
|
available_cpu_count,
|
||||||
check_pdf,
|
check_pdf,
|
||||||
pikepdf_enable_mmap,
|
pikepdf_enable_mmap,
|
||||||
@@ -75,7 +65,7 @@ from ocrmypdf.pdfa import file_claims_pdfa
|
|||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
class PageResult(NamedTuple):
|
class PageResult(NamedTuple): # pylint: disable=inherit-non-class
|
||||||
pageno: int
|
pageno: int
|
||||||
pdf_page_from_image: Optional[Path]
|
pdf_page_from_image: Optional[Path]
|
||||||
ocr: Optional[Path]
|
ocr: Optional[Path]
|
||||||
@@ -213,15 +203,16 @@ def exec_page_sync(page_context: PageContext):
|
|||||||
if filtered_image:
|
if filtered_image:
|
||||||
visible_image_out = filtered_image
|
visible_image_out = filtered_image
|
||||||
pdf_page_from_image_out = create_pdf_page_from_image(
|
pdf_page_from_image_out = create_pdf_page_from_image(
|
||||||
visible_image_out, page_context
|
visible_image_out, page_context, orientation_correction
|
||||||
)
|
)
|
||||||
|
|
||||||
if options.pdf_renderer == 'hocr':
|
if options.pdf_renderer.startswith('hocr'):
|
||||||
(hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context)
|
(hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context)
|
||||||
ocr_out = render_hocr_page(hocr_out, page_context)
|
ocr_out = render_hocr_page(hocr_out, page_context)
|
||||||
|
elif options.pdf_renderer == 'sandwich':
|
||||||
if options.pdf_renderer == 'sandwich':
|
|
||||||
(ocr_out, text_out) = ocr_engine_textonly_pdf(ocr_image_out, page_context)
|
(ocr_out, text_out) = ocr_engine_textonly_pdf(ocr_image_out, page_context)
|
||||||
|
else:
|
||||||
|
raise NotImplementedError(f"pdf_renderer {options.pdf_renderer}")
|
||||||
|
|
||||||
return PageResult(
|
return PageResult(
|
||||||
pageno=page_context.pageno,
|
pageno=page_context.pageno,
|
||||||
@@ -232,14 +223,14 @@ def exec_page_sync(page_context: PageContext):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def post_process(pdf_file, context: PdfContext):
|
def post_process(pdf_file, context: PdfContext, executor: Executor):
|
||||||
pdf_out = pdf_file
|
pdf_out = pdf_file
|
||||||
if context.options.output_type.startswith('pdfa'):
|
if context.options.output_type.startswith('pdfa'):
|
||||||
ps_stub_out = generate_postscript_stub(context)
|
ps_stub_out = generate_postscript_stub(context)
|
||||||
pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context)
|
pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context)
|
||||||
|
|
||||||
pdf_out = metadata_fixup(pdf_out, context)
|
pdf_out = metadata_fixup(pdf_out, context)
|
||||||
return optimize_pdf(pdf_out, context)
|
return optimize_pdf(pdf_out, context, executor)
|
||||||
|
|
||||||
|
|
||||||
def worker_init(max_pixels: int):
|
def worker_init(max_pixels: int):
|
||||||
@@ -250,11 +241,12 @@ def worker_init(max_pixels: int):
|
|||||||
pikepdf_enable_mmap()
|
pikepdf_enable_mmap()
|
||||||
|
|
||||||
|
|
||||||
def exec_concurrent(context: PdfContext):
|
def exec_concurrent(context: PdfContext, executor: Executor):
|
||||||
"""Execute the pipeline concurrently"""
|
"""Execute the pipeline concurrently"""
|
||||||
|
|
||||||
# Run exec_page_sync on every page context
|
# Run exec_page_sync on every page context
|
||||||
max_workers = min(len(context.pdfinfo), context.options.jobs)
|
options = context.options
|
||||||
|
max_workers = min(len(context.pdfinfo), options.jobs)
|
||||||
if max_workers > 1:
|
if max_workers > 1:
|
||||||
log.info("Start processing %d pages concurrently", max_workers)
|
log.info("Start processing %d pages concurrently", max_workers)
|
||||||
|
|
||||||
@@ -262,55 +254,61 @@ def exec_concurrent(context: PdfContext):
|
|||||||
ocrgraft = OcrGrafter(context)
|
ocrgraft = OcrGrafter(context)
|
||||||
|
|
||||||
def update_page(result: PageResult, pbar):
|
def update_page(result: PageResult, pbar):
|
||||||
sidecars[result.pageno] = result.text
|
try:
|
||||||
pbar.update()
|
tls.pageno = result.pageno + 1
|
||||||
ocrgraft.graft_page(
|
sidecars[result.pageno] = result.text
|
||||||
pageno=result.pageno,
|
pbar.update()
|
||||||
image=result.pdf_page_from_image,
|
ocrgraft.graft_page(
|
||||||
textpdf=result.ocr,
|
pageno=result.pageno,
|
||||||
autorotate_correction=result.orientation_correction,
|
image=result.pdf_page_from_image,
|
||||||
)
|
textpdf=result.ocr,
|
||||||
pbar.update()
|
autorotate_correction=result.orientation_correction,
|
||||||
|
)
|
||||||
|
pbar.update()
|
||||||
|
finally:
|
||||||
|
tls.pageno = None
|
||||||
|
|
||||||
exec_progress_pool(
|
executor(
|
||||||
use_threads=context.options.use_threads,
|
use_threads=options.use_threads,
|
||||||
max_workers=max_workers,
|
max_workers=max_workers,
|
||||||
tqdm_kwargs=dict(
|
tqdm_kwargs=dict(
|
||||||
total=(2 * len(context.pdfinfo)),
|
total=(2 * len(context.pdfinfo)),
|
||||||
desc='OCR',
|
desc='OCR' if options.tesseract_timeout > 0 else 'Image processing',
|
||||||
unit='page',
|
unit='page',
|
||||||
unit_scale=0.5,
|
unit_scale=0.5,
|
||||||
disable=not context.options.progress_bar,
|
disable=not options.progress_bar,
|
||||||
),
|
),
|
||||||
task_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS),
|
worker_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS),
|
||||||
task=exec_page_sync,
|
task=exec_page_sync,
|
||||||
task_arguments=context.get_page_contexts(),
|
task_arguments=context.get_page_contexts(),
|
||||||
task_finished=update_page,
|
task_finished=update_page,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Output sidecar text
|
# Output sidecar text
|
||||||
if context.options.sidecar:
|
if options.sidecar:
|
||||||
text = merge_sidecars(sidecars, context)
|
text = merge_sidecars(sidecars, context)
|
||||||
# Copy text file to destination
|
# Copy text file to destination
|
||||||
copy_final(text, context.options.sidecar, context)
|
copy_final(text, options.sidecar, context)
|
||||||
|
|
||||||
# Merge layers to one single pdf
|
# Merge layers to one single pdf
|
||||||
pdf = ocrgraft.finalize()
|
pdf = ocrgraft.finalize()
|
||||||
|
|
||||||
# PDF/A and metadata
|
# PDF/A and metadata
|
||||||
pdf = post_process(pdf, context)
|
log.info("Postprocessing...")
|
||||||
|
pdf = post_process(pdf, context, executor)
|
||||||
|
|
||||||
# Copy PDF file to destination
|
# Copy PDF file to destination
|
||||||
copy_final(pdf, context.options.output_file, context)
|
copy_final(pdf, options.output_file, context)
|
||||||
|
|
||||||
|
|
||||||
class NeverRaise(Exception):
|
def configure_debug_logging(log_filename: Path, prefix: str = ''):
|
||||||
"""An exception that is never raised"""
|
"""
|
||||||
|
Create a debug log file at a specified location.
|
||||||
|
|
||||||
pass # pylint: disable=unnecessary-pass
|
Arguments:
|
||||||
|
log_filename: Where to the put the log file.
|
||||||
|
prefix: The logging domain prefix that should be sent to the log.
|
||||||
def configure_debug_logging(log_filename, prefix=''):
|
"""
|
||||||
log_file_handler = logging.FileHandler(log_filename, delay=True)
|
log_file_handler = logging.FileHandler(log_filename, delay=True)
|
||||||
log_file_handler.setLevel(logging.DEBUG)
|
log_file_handler.setLevel(logging.DEBUG)
|
||||||
formatter = logging.Formatter(
|
formatter = logging.Formatter(
|
||||||
@@ -331,15 +329,23 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
if not plugin_manager:
|
if not plugin_manager:
|
||||||
plugin_manager = get_plugin_manager(options.plugins)
|
plugin_manager = get_plugin_manager(options.plugins)
|
||||||
|
|
||||||
work_folder = Path(mkdtemp(prefix="com.github.ocrmypdf."))
|
work_folder = Path(mkdtemp(prefix="ocrmypdf.io."))
|
||||||
debug_log_handler = None
|
debug_log_handler = None
|
||||||
if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get(
|
if (
|
||||||
'PYTEST_CURRENT_TEST', ''
|
(options.keep_temporary_files or options.verbose >= 1)
|
||||||
|
and not os.environ.get('PYTEST_CURRENT_TEST', '')
|
||||||
|
and not api
|
||||||
):
|
):
|
||||||
debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log")
|
# Debug log for command line interface only with verbose output
|
||||||
|
# See https://github.com/pytest-dev/pytest/issues/5502 for why we skip this
|
||||||
|
# when pytest is running
|
||||||
|
debug_log_handler = configure_debug_logging(
|
||||||
|
Path(work_folder) / "debug.log"
|
||||||
|
) # pragma: no cover
|
||||||
|
|
||||||
pikepdf_enable_mmap()
|
pikepdf_enable_mmap()
|
||||||
|
|
||||||
|
executor = setup_executor(plugin_manager)
|
||||||
try:
|
try:
|
||||||
check_requested_output_file(options)
|
check_requested_output_file(options)
|
||||||
start_input_file, original_filename = create_input_file(options, work_folder)
|
start_input_file, original_filename = create_input_file(options, work_folder)
|
||||||
@@ -352,6 +358,7 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
# Gather pdfinfo and create context
|
# Gather pdfinfo and create context
|
||||||
pdfinfo = get_pdfinfo(
|
pdfinfo = get_pdfinfo(
|
||||||
origin_pdf,
|
origin_pdf,
|
||||||
|
executor=executor,
|
||||||
detailed_analysis=options.redo_ocr,
|
detailed_analysis=options.redo_ocr,
|
||||||
progbar=options.progress_bar,
|
progbar=options.progress_bar,
|
||||||
max_workers=options.jobs if not options.use_threads else 1, # To help debug
|
max_workers=options.jobs if not options.use_threads else 1, # To help debug
|
||||||
@@ -364,7 +371,7 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
validate_pdfinfo_options(context)
|
validate_pdfinfo_options(context)
|
||||||
|
|
||||||
# Execute the pipeline
|
# Execute the pipeline
|
||||||
exec_concurrent(context)
|
exec_concurrent(context, executor)
|
||||||
|
|
||||||
if options.output_file == '-':
|
if options.output_file == '-':
|
||||||
log.info("Output sent to stdout")
|
log.info("Output sent to stdout")
|
||||||
@@ -399,7 +406,9 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
log.error("KeyboardInterrupt")
|
log.error("KeyboardInterrupt")
|
||||||
return ExitCode.ctrl_c
|
return ExitCode.ctrl_c
|
||||||
except (ExitCodeException if not api else NeverRaise) as e:
|
except (ExitCodeException if not api else NeverRaise) as e:
|
||||||
if str(e):
|
if options.verbose >= 1:
|
||||||
|
log.exception("ExitCodeException")
|
||||||
|
elif str(e):
|
||||||
log.error("%s: %s", type(e).__name__, str(e))
|
log.error("%s: %s", type(e).__name__, str(e))
|
||||||
else:
|
else:
|
||||||
log.error(type(e).__name__)
|
log.error(type(e).__name__)
|
||||||
|
|||||||
+43
-40
@@ -1,20 +1,9 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# © 2015-17 James R. Barlow: github.com/jbarlow83
|
# © 2015-17 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
|
|
||||||
import locale
|
import locale
|
||||||
@@ -24,7 +13,7 @@ import sys
|
|||||||
import unicodedata
|
import unicodedata
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj
|
from shutil import copyfileobj
|
||||||
from typing import Tuple
|
from typing import List, Set, Tuple, Union
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
import PIL
|
import PIL
|
||||||
@@ -43,12 +32,12 @@ from ocrmypdf.helpers import (
|
|||||||
monotonic,
|
monotonic,
|
||||||
safe_symlink,
|
safe_symlink,
|
||||||
)
|
)
|
||||||
|
from ocrmypdf.hocrtransform import HOCR_OK_LANGS
|
||||||
from ocrmypdf.subprocess import check_external_program
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
# -------------
|
# -------------
|
||||||
# External dependencies
|
# External dependencies
|
||||||
|
|
||||||
HOCR_OK_LANGS = frozenset(['eng', 'deu', 'spa', 'ita', 'por'])
|
|
||||||
DEFAULT_LANGUAGE = 'eng' # Enforce English hegemony
|
DEFAULT_LANGUAGE = 'eng' # Enforce English hegemony
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
@@ -89,7 +78,7 @@ def check_options_languages(options, ocr_engine_languages):
|
|||||||
def check_options_output(options):
|
def check_options_output(options):
|
||||||
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
||||||
|
|
||||||
if options.pdf_renderer == 'hocr' and not is_latin:
|
if options.pdf_renderer.startswith('hocr') and not is_latin:
|
||||||
msg = (
|
msg = (
|
||||||
"The 'hocr' PDF renderer is known to cause problems with one "
|
"The 'hocr' PDF renderer is known to cause problems with one "
|
||||||
"or more of the languages in your document. Use "
|
"or more of the languages in your document. Use "
|
||||||
@@ -123,6 +112,10 @@ def check_options_sidecar(options):
|
|||||||
"--sidecar filename must be specified when output file is stdout."
|
"--sidecar filename must be specified when output file is stdout."
|
||||||
)
|
)
|
||||||
options.sidecar = options.output_file + '.txt'
|
options.sidecar = options.output_file + '.txt'
|
||||||
|
if options.sidecar == options.input_file or options.sidecar == options.output_file:
|
||||||
|
raise BadArgsError(
|
||||||
|
"--sidecar file must be different from the input and output files"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def check_options_preprocessing(options):
|
def check_options_preprocessing(options):
|
||||||
@@ -144,13 +137,13 @@ def check_options_preprocessing(options):
|
|||||||
options.unpaper_args
|
options.unpaper_args
|
||||||
)
|
)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
raise BadArgsError(str(e))
|
raise BadArgsError("--unpaper-args: " + str(e)) from e
|
||||||
|
|
||||||
|
|
||||||
def _pages_from_ranges(ranges):
|
def _pages_from_ranges(ranges: str) -> Set[int]:
|
||||||
if is_iterable_notstr(ranges):
|
if is_iterable_notstr(ranges):
|
||||||
return set(ranges)
|
return set(ranges)
|
||||||
pages = []
|
pages: List[int] = []
|
||||||
page_groups = ranges.replace(' ', '').split(',')
|
page_groups = ranges.replace(' ', '').split(',')
|
||||||
for g in page_groups:
|
for g in page_groups:
|
||||||
if not g:
|
if not g:
|
||||||
@@ -161,9 +154,18 @@ def _pages_from_ranges(ranges):
|
|||||||
pages.append(int(g) - 1)
|
pages.append(int(g) - 1)
|
||||||
else:
|
else:
|
||||||
try:
|
try:
|
||||||
pages.extend(range(int(start) - 1, int(end)))
|
new_pages = list(range(int(start) - 1, int(end)))
|
||||||
|
if not new_pages:
|
||||||
|
raise BadArgsError(f"invalid page subrange '{start}-{end}'")
|
||||||
|
pages.extend(new_pages)
|
||||||
except ValueError:
|
except ValueError:
|
||||||
raise BadArgsError("invalid page range")
|
raise BadArgsError("invalid page range") from None
|
||||||
|
|
||||||
|
if not pages:
|
||||||
|
raise BadArgsError(
|
||||||
|
f"The string of page ranges '{ranges}' did not contain any recognizable "
|
||||||
|
f"page ranges."
|
||||||
|
)
|
||||||
|
|
||||||
if not monotonic(pages):
|
if not monotonic(pages):
|
||||||
log.warning(
|
log.warning(
|
||||||
@@ -186,8 +188,6 @@ def check_options_ocr_behavior(options):
|
|||||||
)
|
)
|
||||||
if exclusive_options >= 2:
|
if exclusive_options >= 2:
|
||||||
raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.")
|
raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.")
|
||||||
if options.pages and options.sidecar:
|
|
||||||
raise BadArgsError("--pages and --sidecar are mutually exclusive")
|
|
||||||
if options.pages:
|
if options.pages:
|
||||||
options.pages = _pages_from_ranges(options.pages)
|
options.pages = _pages_from_ranges(options.pages)
|
||||||
|
|
||||||
@@ -224,12 +224,12 @@ def check_options_optimizing(options):
|
|||||||
|
|
||||||
|
|
||||||
def check_options_advanced(options):
|
def check_options_advanced(options):
|
||||||
if options.pdfa_image_compression != 'auto' and options.output_type.startswith(
|
if options.pdfa_image_compression != 'auto' and not options.output_type.startswith(
|
||||||
'pdfa'
|
'pdfa'
|
||||||
):
|
):
|
||||||
log.warning(
|
log.warning(
|
||||||
"--pdfa-image-compression argument has no effect when "
|
"--pdfa-image-compression argument only applies when "
|
||||||
"--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'"
|
"--output-type is one of 'pdfa', 'pdfa-1', or 'pdfa-2'"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -279,7 +279,7 @@ def check_closed_streams(options): # pragma: no cover
|
|||||||
Attempting to a fork/exec a new Python process when any of std{in,out,err}
|
Attempting to a fork/exec a new Python process when any of std{in,out,err}
|
||||||
are closed or not flushable for some reason may raise an exception.
|
are closed or not flushable for some reason may raise an exception.
|
||||||
Fix this by opening devnull if the handle seems to be closed. Do this
|
Fix this by opening devnull if the handle seems to be closed. Do this
|
||||||
globally to avoid tracking places all places that fork.
|
globally to avoid tracking all places that fork.
|
||||||
|
|
||||||
Seems to be specific to multiprocessing.Process not all Python process
|
Seems to be specific to multiprocessing.Process not all Python process
|
||||||
forkers.
|
forkers.
|
||||||
@@ -319,17 +319,6 @@ def check_closed_streams(options): # pragma: no cover
|
|||||||
return True
|
return True
|
||||||
|
|
||||||
|
|
||||||
def log_page_orientations(pdfinfo):
|
|
||||||
direction = {0: 'n', 90: 'e', 180: 's', 270: 'w'}
|
|
||||||
orientations = []
|
|
||||||
for n, page in enumerate(pdfinfo):
|
|
||||||
angle = page.rotation or 0
|
|
||||||
if angle != 0:
|
|
||||||
orientations.append('{0}{1}'.format(n + 1, direction.get(angle, '')))
|
|
||||||
if orientations:
|
|
||||||
log.info('Page orientations detected: %s', ' '.join(orientations))
|
|
||||||
|
|
||||||
|
|
||||||
def create_input_file(options, work_folder: Path) -> Tuple[Path, str]:
|
def create_input_file(options, work_folder: Path) -> Tuple[Path, str]:
|
||||||
if options.input_file == '-':
|
if options.input_file == '-':
|
||||||
# stdin
|
# stdin
|
||||||
@@ -352,7 +341,17 @@ def create_input_file(options, work_folder: Path) -> Tuple[Path, str]:
|
|||||||
safe_symlink(options.input_file, target)
|
safe_symlink(options.input_file, target)
|
||||||
return target, os.fspath(options.input_file)
|
return target, os.fspath(options.input_file)
|
||||||
except FileNotFoundError:
|
except FileNotFoundError:
|
||||||
raise InputFileError(f"File not found - {options.input_file}")
|
msg = f"File not found - {options.input_file}"
|
||||||
|
if Path('/.dockerenv').exists(): # pragma: no cover
|
||||||
|
msg += (
|
||||||
|
"\nDocker cannot your working directory unless you "
|
||||||
|
"explicitly share it with the Docker container and set up"
|
||||||
|
"permissions correctly.\n"
|
||||||
|
"You may find it easier to use stdin/stdout:"
|
||||||
|
"\n"
|
||||||
|
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf\n"
|
||||||
|
)
|
||||||
|
raise InputFileError(msg)
|
||||||
|
|
||||||
|
|
||||||
def check_requested_output_file(options):
|
def check_requested_output_file(options):
|
||||||
@@ -416,6 +415,10 @@ def report_output_file_size(options, input_file, output_file):
|
|||||||
f"The optional dependency '{name}' was not found, so some image "
|
f"The optional dependency '{name}' was not found, so some image "
|
||||||
f"optimizations could not be attempted."
|
f"optimizations could not be attempted."
|
||||||
)
|
)
|
||||||
|
if options.output_type.startswith('pdfa'):
|
||||||
|
reasons.append("PDF/A conversion was enabled. (Try `--output-type pdf`.)")
|
||||||
|
if options.plugins:
|
||||||
|
reasons.append("Plugins were used.")
|
||||||
|
|
||||||
if reasons:
|
if reasons:
|
||||||
explanation = "Possible reasons for this include:\n" + '\n'.join(reasons) + "\n"
|
explanation = "Possible reasons for this include:\n" + '\n'.join(reasons) + "\n"
|
||||||
|
|||||||
@@ -1,19 +1,8 @@
|
|||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
# © 2017 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
|
|
||||||
import pkg_resources
|
import pkg_resources
|
||||||
|
|||||||
+86
-53
@@ -1,28 +1,24 @@
|
|||||||
# © 2019 James R. Barlow: github.com/jbarlow83
|
# © 2019 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
import threading
|
||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
|
from io import IOBase
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import BinaryIO, Iterable, Union
|
from typing import AnyStr, BinaryIO, Iterable, Optional, Union
|
||||||
|
from warnings import warn
|
||||||
|
|
||||||
from ocrmypdf._logging import PageNumberFilter, TqdmConsole
|
from ocrmypdf._logging import ( # pylint: disable=unused-import
|
||||||
|
PageNumberFilter,
|
||||||
|
TqdmConsole,
|
||||||
|
)
|
||||||
from ocrmypdf._plugin_manager import get_plugin_manager
|
from ocrmypdf._plugin_manager import get_plugin_manager
|
||||||
from ocrmypdf._sync import run_pipeline
|
from ocrmypdf._sync import run_pipeline
|
||||||
from ocrmypdf._validation import check_options
|
from ocrmypdf._validation import check_options
|
||||||
@@ -35,7 +31,10 @@ except ModuleNotFoundError:
|
|||||||
coloredlogs = None
|
coloredlogs = None
|
||||||
|
|
||||||
|
|
||||||
PathOrIO = Union[BinaryIO, os.PathLike, str, bytes]
|
StrPath = Union[os.PathLike, AnyStr]
|
||||||
|
PathOrIO = Union[BinaryIO, StrPath]
|
||||||
|
|
||||||
|
_api_lock = threading.Lock()
|
||||||
|
|
||||||
|
|
||||||
class Verbosity(IntEnum):
|
class Verbosity(IntEnum):
|
||||||
@@ -49,29 +48,45 @@ class Verbosity(IntEnum):
|
|||||||
|
|
||||||
def configure_logging(
|
def configure_logging(
|
||||||
verbosity: Verbosity,
|
verbosity: Verbosity,
|
||||||
|
*,
|
||||||
progress_bar_friendly: bool = True,
|
progress_bar_friendly: bool = True,
|
||||||
manage_root_logger: bool = False,
|
manage_root_logger: bool = False,
|
||||||
|
plugin_manager=None,
|
||||||
):
|
):
|
||||||
"""Set up logging.
|
"""Set up logging.
|
||||||
|
|
||||||
Library users may wish to use this function if they want their log output to be
|
Before calling :func:`ocrmypdf.ocr()`, you can use this function to
|
||||||
similar to ocrmypdf command line interface. If not used, the external application
|
configure logging if you want ocrmypdf's output to look like the ocrmypdf
|
||||||
should configure logging on its own.
|
command line interface. It will register log handlers, log filters, and
|
||||||
|
formatters, configure color logging to standard error, and adjust the log
|
||||||
|
levels of third party libraries. Details of this are fine-tuned and subject
|
||||||
|
to change. The ``verbosity`` argument is equivalent to the argument
|
||||||
|
``--verbose`` and applies those settings. If you have a wrapper
|
||||||
|
script for ocrmypdf and you want it to be very similar to ocrmypdf, use this
|
||||||
|
function; if you are using ocrmypdf as part of an application that manages
|
||||||
|
its own logging, you probably do not want this function.
|
||||||
|
|
||||||
ocrmypdf will perform all of its logging under the ``"ocrmypdf"`` logging namespace.
|
If this function is not called, ocrmypdf will not configure logging, and it
|
||||||
In addition, ocrmypdf imports pdfminer, which logs under ``"pdfminer"``. A library
|
is up to the caller of ``ocrmypdf.ocr()`` to set up logging as it wishes using
|
||||||
user may wish to configure both; note that pdfminer is extremely chatty at the log
|
the Python standard library's logging module. If this function is called,
|
||||||
level ``logging.INFO``.
|
the caller may of course make further adjustments to logging.
|
||||||
|
|
||||||
Library users may perform additional configuration afterwards.
|
Regardless of whether this function is called, ocrmypdf will perform all of
|
||||||
|
its logging under the ``"ocrmypdf"`` logging namespace. In addition,
|
||||||
|
ocrmypdf imports pdfminer, which logs under ``"pdfminer"``. A library user
|
||||||
|
may wish to configure both; note that pdfminer is extremely chatty at the
|
||||||
|
log level ``logging.INFO``.
|
||||||
|
|
||||||
|
This function does not set up the ``debug.log`` log file that the command
|
||||||
|
line interface does at certain verbosity levels. Applications should configure
|
||||||
|
their own debug logging.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
verbosity (Verbosity): Verbosity level.
|
verbosity: Verbosity level.
|
||||||
progress_bar_friendly (bool): Install the TqdmConsole log handler, which is
|
progress_bar_friendly: If True (the default), install a custom log handler
|
||||||
compatible with the tqdm progress bar; without this log messages will
|
that is compatible with progress bars and colored output.
|
||||||
overwrite the progress bar
|
manage_root_logger: Configure the process's root logger.
|
||||||
manage_root_logger (bool): Configure the process's root logger, to ensure
|
plugin_manager: The plugin manager, used for obtaining the custom log handler.
|
||||||
all log output is sent through
|
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
The toplevel logger for ocrmypdf (or the root logger, if we are managing it).
|
The toplevel logger for ocrmypdf (or the root logger, if we are managing it).
|
||||||
@@ -82,9 +97,11 @@ def configure_logging(
|
|||||||
log = logging.getLogger(prefix)
|
log = logging.getLogger(prefix)
|
||||||
log.setLevel(logging.DEBUG)
|
log.setLevel(logging.DEBUG)
|
||||||
|
|
||||||
if progress_bar_friendly:
|
console = None
|
||||||
console = logging.StreamHandler(stream=TqdmConsole(sys.stderr))
|
if plugin_manager and progress_bar_friendly:
|
||||||
else:
|
console = plugin_manager.hook.get_logging_console()
|
||||||
|
|
||||||
|
if not console:
|
||||||
console = logging.StreamHandler(stream=sys.stderr)
|
console = logging.StreamHandler(stream=sys.stderr)
|
||||||
|
|
||||||
if verbosity < 0:
|
if verbosity < 0:
|
||||||
@@ -170,14 +187,14 @@ def create_options(
|
|||||||
else:
|
else:
|
||||||
raise TypeError(f"{arg}: {val} ({type(val)})")
|
raise TypeError(f"{arg}: {val} ({type(val)})")
|
||||||
|
|
||||||
try:
|
if isinstance(input_file, (BinaryIO, IOBase)):
|
||||||
cmdline.append(os.fspath(input_file))
|
|
||||||
except TypeError:
|
|
||||||
cmdline.append('stream://input_file')
|
cmdline.append('stream://input_file')
|
||||||
try:
|
else:
|
||||||
cmdline.append(os.fspath(output_file))
|
cmdline.append(os.fspath(input_file))
|
||||||
except TypeError:
|
if isinstance(output_file, (BinaryIO, IOBase)):
|
||||||
cmdline.append('stream://output_file')
|
cmdline.append('stream://output_file')
|
||||||
|
else:
|
||||||
|
cmdline.append(os.fspath(output_file))
|
||||||
|
|
||||||
parser._api_mode = True
|
parser._api_mode = True
|
||||||
options = parser.parse_args(cmdline)
|
options = parser.parse_args(cmdline)
|
||||||
@@ -199,7 +216,7 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
language: Iterable[str] = None,
|
language: Iterable[str] = None,
|
||||||
image_dpi: int = None,
|
image_dpi: int = None,
|
||||||
output_type=None,
|
output_type=None,
|
||||||
sidecar: os.PathLike = None,
|
sidecar: Optional[StrPath] = None,
|
||||||
jobs: int = None,
|
jobs: int = None,
|
||||||
use_threads: bool = None,
|
use_threads: bool = None,
|
||||||
title: str = None,
|
title: str = None,
|
||||||
@@ -236,7 +253,8 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
user_words: os.PathLike = None,
|
user_words: os.PathLike = None,
|
||||||
user_patterns: os.PathLike = None,
|
user_patterns: os.PathLike = None,
|
||||||
fast_web_view: float = None,
|
fast_web_view: float = None,
|
||||||
plugins: Iterable[str] = None,
|
plugins: Iterable[StrPath] = None,
|
||||||
|
plugin_manager=None,
|
||||||
keep_temporary_files: bool = None,
|
keep_temporary_files: bool = None,
|
||||||
progress_bar: bool = None,
|
progress_bar: bool = None,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
@@ -258,7 +276,7 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
read.
|
read.
|
||||||
output_file: If a :class:`pathlib.Path`, ``str`` or ``bytes``, this is
|
output_file: If a :class:`pathlib.Path`, ``str`` or ``bytes``, this is
|
||||||
interpreted as file system path to the output file. If the object
|
interpreted as file system path to the output file. If the object
|
||||||
appears to be a writable stream (with methods such as ``.read()`` and
|
appears to be a writable stream (with methods such as ``.write()`` and
|
||||||
``.seek()``), the output will be written to this stream. If
|
``.seek()``), the output will be written to this stream. If
|
||||||
``output_file`` is ``"-"``, the output will be written to ``sys.stdout``
|
``output_file`` is ``"-"``, the output will be written to ``sys.stdout``
|
||||||
(provided that standard output does not seem to be a terminal device).
|
(provided that standard output does not seem to be a terminal device).
|
||||||
@@ -288,20 +306,35 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
Returns:
|
Returns:
|
||||||
:class:`ocrmypdf.ExitCode`
|
:class:`ocrmypdf.ExitCode`
|
||||||
"""
|
"""
|
||||||
|
if plugins and plugin_manager:
|
||||||
|
raise ValueError("plugins= and plugin_manager are mutually exclusive")
|
||||||
|
|
||||||
if not plugins:
|
if not plugins:
|
||||||
plugins = []
|
plugins = []
|
||||||
|
elif isinstance(plugins, (str, Path)):
|
||||||
|
plugins = [plugins]
|
||||||
else:
|
else:
|
||||||
plugins = list(plugins)
|
plugins = list(plugins)
|
||||||
|
|
||||||
parser = get_parser()
|
# No new variable names should be assigned until these two steps are run
|
||||||
_plugin_manager = get_plugin_manager(plugins)
|
create_options_kwargs = {k: v for k, v in locals().items() if k != 'kwargs'}
|
||||||
_plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
|
|
||||||
|
|
||||||
create_options_kwargs = {
|
|
||||||
k: v for k, v in locals().items() if not k.startswith('_') and k != 'kwargs'
|
|
||||||
}
|
|
||||||
create_options_kwargs.update(kwargs)
|
create_options_kwargs.update(kwargs)
|
||||||
|
|
||||||
options = create_options(**create_options_kwargs)
|
parser = get_parser()
|
||||||
check_options(options, _plugin_manager)
|
create_options_kwargs['parser'] = parser
|
||||||
return run_pipeline(options=options, plugin_manager=_plugin_manager, api=True)
|
|
||||||
|
with _api_lock:
|
||||||
|
# We can't allow multiple ocrmypdf.ocr() threads to run in parallel, because
|
||||||
|
# they might install different plugins, and generally speaking we have areas
|
||||||
|
# of code that use global state.
|
||||||
|
|
||||||
|
if not plugin_manager:
|
||||||
|
plugin_manager = get_plugin_manager(plugins)
|
||||||
|
plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
|
||||||
|
|
||||||
|
if 'verbose' in kwargs:
|
||||||
|
warn("ocrmypdf.ocr(verbose=) is ignored. Use ocrmypdf.configure_logging().")
|
||||||
|
|
||||||
|
options = create_options(**create_options_kwargs)
|
||||||
|
check_options(options, plugin_manager)
|
||||||
|
return run_pipeline(options=options, plugin_manager=plugin_manager, api=True)
|
||||||
|
|||||||
@@ -1,16 +1,9 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
# This file exists only mark builtin_plugins as a package.
|
||||||
# (at your option) any later version.
|
# The plugin manager will not load it, so anything defined here may not be
|
||||||
#
|
# processed as a module.
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|||||||
@@ -0,0 +1,172 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import logging.handlers
|
||||||
|
import multiprocessing
|
||||||
|
import os
|
||||||
|
import queue
|
||||||
|
import signal
|
||||||
|
import sys
|
||||||
|
import threading
|
||||||
|
from contextlib import suppress
|
||||||
|
from multiprocessing import Pool as ProcessPool
|
||||||
|
from multiprocessing.pool import ThreadPool
|
||||||
|
from typing import Callable, Iterable, Union
|
||||||
|
|
||||||
|
from tqdm import tqdm
|
||||||
|
|
||||||
|
from ocrmypdf import Executor, hookimpl
|
||||||
|
from ocrmypdf._logging import TqdmConsole
|
||||||
|
from ocrmypdf.exceptions import InputFileError
|
||||||
|
from ocrmypdf.helpers import remove_all_log_handlers
|
||||||
|
|
||||||
|
Queue = Union[multiprocessing.Queue, queue.Queue]
|
||||||
|
|
||||||
|
|
||||||
|
def log_listener(q: Queue):
|
||||||
|
"""Listen to the worker processes and forward the messages to logging
|
||||||
|
|
||||||
|
For simplicity this is a thread rather than a process. Only one process
|
||||||
|
should actually write to sys.stderr or whatever we're using, so if this is
|
||||||
|
made into a process the main application needs to be directed to it.
|
||||||
|
|
||||||
|
See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes
|
||||||
|
"""
|
||||||
|
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
record = q.get()
|
||||||
|
if record is None:
|
||||||
|
break
|
||||||
|
logger = logging.getLogger(record.name)
|
||||||
|
logger.handle(record)
|
||||||
|
except Exception: # pylint: disable=broad-except
|
||||||
|
import traceback # pylint: disable=import-outside-toplevel
|
||||||
|
|
||||||
|
print("Logging problem", file=sys.stderr)
|
||||||
|
traceback.print_exc(file=sys.stderr)
|
||||||
|
|
||||||
|
|
||||||
|
def process_sigbus(*args):
|
||||||
|
raise InputFileError("A worker process lost access to an input file")
|
||||||
|
|
||||||
|
|
||||||
|
def process_init(q: Queue, user_init: Callable[[], None], loglevel):
|
||||||
|
"""Initialize a process pool worker"""
|
||||||
|
|
||||||
|
# Ignore SIGINT (our parent process will kill us gracefully)
|
||||||
|
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
||||||
|
|
||||||
|
# Install SIGBUS handler (so our parent process can abort somewhat gracefully)
|
||||||
|
with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS
|
||||||
|
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
|
||||||
|
signal.signal(signal.SIGBUS, process_sigbus)
|
||||||
|
|
||||||
|
# Remove any log handlers that belong to the parent process
|
||||||
|
root = logging.getLogger()
|
||||||
|
remove_all_log_handlers(root)
|
||||||
|
|
||||||
|
# Set up our single log handler to forward messages to the parent
|
||||||
|
root.setLevel(loglevel)
|
||||||
|
root.addHandler(logging.handlers.QueueHandler(q))
|
||||||
|
|
||||||
|
user_init()
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
|
def thread_init(_queue: Queue, user_init: Callable[[], None], _loglevel):
|
||||||
|
# As a thread, block SIGBUS so the main thread deals with it...
|
||||||
|
with suppress(AttributeError):
|
||||||
|
signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS})
|
||||||
|
|
||||||
|
user_init()
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
|
class StandardExecutor(Executor):
|
||||||
|
def _execute(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
use_threads: bool,
|
||||||
|
max_workers: int,
|
||||||
|
tqdm_kwargs: dict,
|
||||||
|
worker_initializer: Callable,
|
||||||
|
task: Callable,
|
||||||
|
task_arguments: Iterable,
|
||||||
|
task_finished: Callable,
|
||||||
|
):
|
||||||
|
if use_threads:
|
||||||
|
log_queue = queue.Queue(-1)
|
||||||
|
pool_class = ThreadPool
|
||||||
|
initializer = thread_init
|
||||||
|
else:
|
||||||
|
log_queue = multiprocessing.Queue(-1)
|
||||||
|
pool_class = ProcessPool
|
||||||
|
initializer = process_init
|
||||||
|
|
||||||
|
# Regardless of whether we use_threads for worker processes, the log_listener
|
||||||
|
# must be a thread. Make sure we create the listener after the worker pool,
|
||||||
|
# so that it does not get forked into the workers.
|
||||||
|
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
||||||
|
listener.start()
|
||||||
|
|
||||||
|
with self.pbar_class(**tqdm_kwargs) as pbar:
|
||||||
|
pool = pool_class(
|
||||||
|
processes=max_workers,
|
||||||
|
initializer=initializer,
|
||||||
|
initargs=(log_queue, worker_initializer, logging.getLogger("").level),
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
results = pool.imap_unordered(task, task_arguments)
|
||||||
|
for result in results:
|
||||||
|
if task_finished:
|
||||||
|
task_finished(result, pbar)
|
||||||
|
else:
|
||||||
|
pbar.update()
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
# Terminate pool so we exit instantly
|
||||||
|
pool.terminate()
|
||||||
|
# Don't try listener.join() here, will deadlock
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
||||||
|
# Unless inside pytest, exit immediately because no one wants
|
||||||
|
# to wait for child processes to finalize results that will be
|
||||||
|
# thrown away. Inside pytest, we want child processes to exit
|
||||||
|
# cleanly so that they output an error messages or coverage data
|
||||||
|
# we need from them.
|
||||||
|
pool.terminate()
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
# Terminate log listener
|
||||||
|
log_queue.put_nowait(None)
|
||||||
|
pool.close()
|
||||||
|
pool.join()
|
||||||
|
|
||||||
|
listener.join()
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def get_executor(progressbar_class):
|
||||||
|
return StandardExecutor(pbar_class=progressbar_class)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def get_progressbar_class():
|
||||||
|
return tqdm
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def get_logging_console():
|
||||||
|
return logging.StreamHandler(stream=TqdmConsole(sys.stderr))
|
||||||
@@ -0,0 +1,14 @@
|
|||||||
|
# © 2021 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
from ocrmypdf import hookimpl
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def filter_pdf_page(
|
||||||
|
page, image_filename, output_pdf
|
||||||
|
): # pylint: disable=unused-argument
|
||||||
|
return output_pdf
|
||||||
@@ -1,19 +1,9 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
@@ -49,13 +39,11 @@ def check_options(options):
|
|||||||
if gs_version < '9.20' and options.output_type != 'pdf' and not is_latin:
|
if gs_version < '9.20' and options.output_type != 'pdf' and not is_latin:
|
||||||
# https://bugs.ghostscript.com/show_bug.cgi?id=696874
|
# https://bugs.ghostscript.com/show_bug.cgi?id=696874
|
||||||
# Ghostscript < 9.20 fails to encode multibyte characters properly
|
# Ghostscript < 9.20 fails to encode multibyte characters properly
|
||||||
msg = (
|
log.warning(
|
||||||
"The installed version of Ghostscript does not work correctly "
|
f"The installed version of Ghostscript ({gs_version}) does not work "
|
||||||
"with the OCR languages you specified. Use --output-type pdf or "
|
"correctly with the OCR languages you specified. Use --output-type pdf or "
|
||||||
"upgrade to Ghostscript 9.20 or later to avoid this issue."
|
"upgrade to Ghostscript 9.20 or later to avoid this issue."
|
||||||
)
|
)
|
||||||
msg += f"Found Ghostscript {gs_version}"
|
|
||||||
log.warning(msg)
|
|
||||||
|
|
||||||
if options.output_type == 'pdfa':
|
if options.output_type == 'pdfa':
|
||||||
options.output_type = 'pdfa-2'
|
options.output_type = 'pdfa-2'
|
||||||
@@ -73,9 +61,9 @@ def rasterize_pdf_page(
|
|||||||
raster_device,
|
raster_device,
|
||||||
raster_dpi,
|
raster_dpi,
|
||||||
pageno,
|
pageno,
|
||||||
page_dpi=None,
|
page_dpi,
|
||||||
rotation=None,
|
rotation,
|
||||||
filter_vector=False,
|
filter_vector,
|
||||||
):
|
):
|
||||||
ghostscript.rasterize_pdf(
|
ghostscript.rasterize_pdf(
|
||||||
input_file,
|
input_file,
|
||||||
@@ -91,12 +79,21 @@ def rasterize_pdf_page(
|
|||||||
|
|
||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
def generate_pdfa(
|
||||||
|
pdf_pages,
|
||||||
|
pdfmark,
|
||||||
|
output_file,
|
||||||
|
compression,
|
||||||
|
pdf_version,
|
||||||
|
pdfa_part,
|
||||||
|
progressbar_class,
|
||||||
|
):
|
||||||
ghostscript.generate_pdfa(
|
ghostscript.generate_pdfa(
|
||||||
pdf_pages=[*pdf_pages, pdfmark],
|
pdf_pages=[*pdf_pages, pdfmark],
|
||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
compression=compression,
|
compression=compression,
|
||||||
pdf_version=pdf_version,
|
pdf_version=pdf_version,
|
||||||
pdfa_part=pdfa_part,
|
pdfa_part=pdfa_part,
|
||||||
|
progressbar_class=progressbar_class,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|||||||
@@ -1,19 +1,9 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -90,20 +80,14 @@ def check_options(options):
|
|||||||
program='tesseract',
|
program='tesseract',
|
||||||
package={'linux': 'tesseract-ocr'},
|
package={'linux': 'tesseract-ocr'},
|
||||||
version_checker=tesseract.version,
|
version_checker=tesseract.version,
|
||||||
need_version='4.0.0', # using backport for Travis CI
|
need_version='4.0.0-beta.1', # using backport for Travis CI
|
||||||
|
version_parser=tesseract.TesseractVersion,
|
||||||
)
|
)
|
||||||
|
|
||||||
# Decide on what renderer to use
|
# Decide on what renderer to use
|
||||||
if options.pdf_renderer == 'auto':
|
if options.pdf_renderer == 'auto':
|
||||||
options.pdf_renderer = 'sandwich'
|
options.pdf_renderer = 'sandwich'
|
||||||
|
|
||||||
if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf(
|
|
||||||
set(options.languages)
|
|
||||||
):
|
|
||||||
raise MissingDependencyError(
|
|
||||||
"You are using an alpha version of Tesseract 4.0 that does not support "
|
|
||||||
"the textonly_pdf parameter. We don't support versions this old."
|
|
||||||
)
|
|
||||||
if not tesseract.has_user_words() and (options.user_words or options.user_patterns):
|
if not tesseract.has_user_words() and (options.user_words or options.user_patterns):
|
||||||
log.warning(
|
log.warning(
|
||||||
"Tesseract 4.0 ignores --user-words and --user-patterns, so these "
|
"Tesseract 4.0 ignores --user-words and --user-patterns, so these "
|
||||||
@@ -131,9 +115,7 @@ def validate(pdfinfo, options):
|
|||||||
os.environ['OMP_THREAD_LIMIT'] = str(tess_threads)
|
os.environ['OMP_THREAD_LIMIT'] = str(tess_threads)
|
||||||
else:
|
else:
|
||||||
tess_threads = int(os.environ['OMP_THREAD_LIMIT'])
|
tess_threads = int(os.environ['OMP_THREAD_LIMIT'])
|
||||||
|
log.debug("Using Tesseract OpenMP thread limit %d", tess_threads)
|
||||||
if tess_threads > 1:
|
|
||||||
log.info("Using Tesseract OpenMP thread limit %d", tess_threads)
|
|
||||||
|
|
||||||
|
|
||||||
class TesseractOcrEngine(OcrEngine):
|
class TesseractOcrEngine(OcrEngine):
|
||||||
|
|||||||
+11
-17
@@ -1,27 +1,20 @@
|
|||||||
# © 2015-19 James R. Barlow: github.com/jbarlow83
|
# © 2015-19 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
|
from typing import Optional, Type, TypeVar
|
||||||
|
|
||||||
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
||||||
from ocrmypdf._version import __version__ as _VERSION
|
from ocrmypdf._version import __version__ as _VERSION
|
||||||
|
|
||||||
|
T = TypeVar('T')
|
||||||
|
|
||||||
def numeric(basetype, min_=None, max_=None):
|
|
||||||
|
def numeric(basetype: Type[T], min_: Optional[T] = None, max_: Optional[T] = None):
|
||||||
"""Validator for numeric params"""
|
"""Validator for numeric params"""
|
||||||
min_ = basetype(min_) if min_ is not None else None
|
min_ = basetype(min_) if min_ is not None else None
|
||||||
max_ = basetype(max_) if max_ is not None else None
|
max_ = basetype(max_) if max_ is not None else None
|
||||||
@@ -174,7 +167,8 @@ Online documentation is located at:
|
|||||||
metavar='FILE',
|
metavar='FILE',
|
||||||
help="Generate sidecar text files that contain the same text recognized "
|
help="Generate sidecar text files that contain the same text recognized "
|
||||||
"by Tesseract. This may be useful for building a OCR text database. "
|
"by Tesseract. This may be useful for building a OCR text database. "
|
||||||
"If FILE is omitted, the sidecar file be named {output_file}.txt "
|
"If FILE is omitted, the sidecar file be named {output_file}.txt; the next "
|
||||||
|
"argument must NOT be the name of the input PDF. "
|
||||||
"If FILE is set to '-', the sidecar is written to stdout (a "
|
"If FILE is set to '-', the sidecar is written to stdout (a "
|
||||||
"convenient way to preview OCR quality). The output file and sidecar "
|
"convenient way to preview OCR quality). The output file and sidecar "
|
||||||
"may not both use stdout at the same time.",
|
"may not both use stdout at the same time.",
|
||||||
@@ -417,7 +411,7 @@ Online documentation is located at:
|
|||||||
)
|
)
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--pdf-renderer',
|
'--pdf-renderer',
|
||||||
choices=['auto', 'hocr', 'sandwich'],
|
choices=['auto', 'hocr', 'sandwich', 'hocrdebug'],
|
||||||
default='auto',
|
default='auto',
|
||||||
help="Choose OCR PDF renderer - the default option is to let OCRmyPDF "
|
help="Choose OCR PDF renderer - the default option is to let OCRmyPDF "
|
||||||
"choose. See documentation for discussion.",
|
"choose. See documentation for discussion.",
|
||||||
|
|||||||
@@ -1,19 +1,8 @@
|
|||||||
# © 2016 James R. Barlow: github.com/jbarlow83
|
# © 2016 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
|
|
||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
|
|||||||
@@ -0,0 +1,191 @@
|
|||||||
|
# © 2021 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
|
"""Semaphore-free alternate executor.
|
||||||
|
|
||||||
|
There are two popular environments that do not fully support the standard Python
|
||||||
|
multiprocessing module: AWS Lambda, and Termux (a terminal emulator for Android).
|
||||||
|
|
||||||
|
This alternate executor divvies up work among worker processes before processing,
|
||||||
|
rather than having each worker consume work from a shared queue when they finish
|
||||||
|
their task. This means workers have no need to coordinate with each other. Each
|
||||||
|
worker communicates only with the main process.
|
||||||
|
|
||||||
|
This is not without drawbacks. If the tasks are not "even" in size, which cannot
|
||||||
|
be guaranteed, some workers may end up with too much work while others are idle.
|
||||||
|
It is less efficient than the standard implementation, so not th edefault.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import logging.handlers
|
||||||
|
import signal
|
||||||
|
from contextlib import suppress
|
||||||
|
from enum import Enum, auto
|
||||||
|
from itertools import islice, repeat, takewhile, zip_longest
|
||||||
|
from multiprocessing import Pipe, Process
|
||||||
|
from multiprocessing.connection import Connection, wait
|
||||||
|
from typing import Callable, Iterable, Iterator
|
||||||
|
|
||||||
|
from ocrmypdf import Executor, hookimpl
|
||||||
|
from ocrmypdf._concurrent import NullProgressBar
|
||||||
|
from ocrmypdf.exceptions import InputFileError
|
||||||
|
from ocrmypdf.helpers import remove_all_log_handlers
|
||||||
|
|
||||||
|
|
||||||
|
class MessageType(Enum):
|
||||||
|
exception = auto()
|
||||||
|
result = auto()
|
||||||
|
complete = auto()
|
||||||
|
|
||||||
|
|
||||||
|
def split_every(n: int, iterable: Iterable) -> Iterator:
|
||||||
|
"""Split iterable into groups of n.
|
||||||
|
|
||||||
|
>>> list(split_every(4, range(10)))
|
||||||
|
[[0, 1, 2], [3, 4, 5], [6, 7, 8], [9]]
|
||||||
|
|
||||||
|
https://stackoverflow.com/a/22919323
|
||||||
|
"""
|
||||||
|
iterator = iter(iterable)
|
||||||
|
return takewhile(bool, (list(islice(iterator, n)) for _ in repeat(None)))
|
||||||
|
|
||||||
|
|
||||||
|
def process_sigbus(*args):
|
||||||
|
raise InputFileError("A worker process lost access to an input file")
|
||||||
|
|
||||||
|
|
||||||
|
class ConnectionLogHandler(logging.handlers.QueueHandler):
|
||||||
|
def __init__(self, conn: Connection) -> None:
|
||||||
|
super().__init__(None)
|
||||||
|
self.conn = conn
|
||||||
|
|
||||||
|
def enqueue(self, record):
|
||||||
|
self.conn.send(('log', record))
|
||||||
|
|
||||||
|
|
||||||
|
def process_loop(
|
||||||
|
conn: Connection, user_init: Callable[[], None], loglevel, task, task_args
|
||||||
|
):
|
||||||
|
"""Initialize a process pool worker"""
|
||||||
|
|
||||||
|
# Install SIGBUS handler (so our parent process can abort somewhat gracefully)
|
||||||
|
with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS
|
||||||
|
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
|
||||||
|
signal.signal(signal.SIGBUS, process_sigbus)
|
||||||
|
|
||||||
|
# Reconfigure the root logger for this process to send all messages to a queue
|
||||||
|
h = ConnectionLogHandler(conn)
|
||||||
|
root = logging.getLogger()
|
||||||
|
remove_all_log_handlers(root)
|
||||||
|
root.setLevel(loglevel)
|
||||||
|
root.addHandler(h)
|
||||||
|
|
||||||
|
user_init()
|
||||||
|
|
||||||
|
for args in task_args:
|
||||||
|
try:
|
||||||
|
result = task(args)
|
||||||
|
except Exception as e:
|
||||||
|
conn.send((MessageType.exception, e))
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
conn.send((MessageType.result, result))
|
||||||
|
|
||||||
|
conn.send((MessageType.complete, None))
|
||||||
|
conn.close()
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
|
class LambdaExecutor(Executor):
|
||||||
|
def _execute(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
use_threads: bool,
|
||||||
|
max_workers: int,
|
||||||
|
tqdm_kwargs: dict,
|
||||||
|
worker_initializer: Callable,
|
||||||
|
task: Callable,
|
||||||
|
task_arguments: Iterable,
|
||||||
|
task_finished: Callable,
|
||||||
|
):
|
||||||
|
if use_threads and max_workers == 1:
|
||||||
|
with self.pbar_class(**tqdm_kwargs) as pbar:
|
||||||
|
for args in task_arguments:
|
||||||
|
result = task(args)
|
||||||
|
task_finished(result, pbar)
|
||||||
|
return
|
||||||
|
|
||||||
|
task_arguments = list(task_arguments)
|
||||||
|
grouped_args = list(
|
||||||
|
zip_longest(*list(split_every(max_workers, task_arguments)))
|
||||||
|
)
|
||||||
|
if not grouped_args:
|
||||||
|
return
|
||||||
|
|
||||||
|
processes = []
|
||||||
|
connections = []
|
||||||
|
for chunk in grouped_args:
|
||||||
|
parent_conn, child_conn = Pipe()
|
||||||
|
|
||||||
|
worker_args = [args for args in chunk if args is not None]
|
||||||
|
process = Process(
|
||||||
|
target=process_loop,
|
||||||
|
args=(
|
||||||
|
child_conn,
|
||||||
|
worker_initializer,
|
||||||
|
logging.getLogger("").level,
|
||||||
|
task,
|
||||||
|
worker_args,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
process.daemon = True
|
||||||
|
processes.append(process)
|
||||||
|
connections.append(parent_conn)
|
||||||
|
|
||||||
|
for process in processes:
|
||||||
|
process.start()
|
||||||
|
|
||||||
|
with self.pbar_class(**tqdm_kwargs) as pbar:
|
||||||
|
while connections:
|
||||||
|
for r in wait(connections):
|
||||||
|
try:
|
||||||
|
msg_type, msg = r.recv()
|
||||||
|
except EOFError:
|
||||||
|
connections.remove(r)
|
||||||
|
continue
|
||||||
|
|
||||||
|
if msg_type == MessageType.result:
|
||||||
|
if task_finished:
|
||||||
|
task_finished(msg, pbar)
|
||||||
|
elif msg_type == 'log':
|
||||||
|
record = msg
|
||||||
|
logger = logging.getLogger(record.name)
|
||||||
|
logger.handle(record)
|
||||||
|
elif msg_type == MessageType.complete:
|
||||||
|
connections.remove(r)
|
||||||
|
elif msg_type == MessageType.exception:
|
||||||
|
for process in processes:
|
||||||
|
process.terminate()
|
||||||
|
raise msg
|
||||||
|
|
||||||
|
for process in processes:
|
||||||
|
process.join()
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def get_executor(progressbar_class):
|
||||||
|
return LambdaExecutor(pbar_class=progressbar_class)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def get_logging_console():
|
||||||
|
return logging.StreamHandler()
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def get_progressbar_class():
|
||||||
|
return NullProgressBar
|
||||||
+78
-61
@@ -1,19 +1,9 @@
|
|||||||
# © 2016 James R. Barlow: github.com/jbarlow83
|
# © 2016 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import multiprocessing
|
import multiprocessing
|
||||||
@@ -25,9 +15,9 @@ from collections.abc import Iterable
|
|||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from functools import wraps
|
from functools import wraps
|
||||||
from io import StringIO
|
from io import StringIO
|
||||||
from math import isclose
|
from math import isclose, isfinite
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Sequence, TypeVar
|
from typing import Any, Sequence
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
|
|
||||||
@@ -49,6 +39,10 @@ class Resolution(namedtuple('Resolution', ('x', 'y'))):
|
|||||||
def is_square(self) -> bool:
|
def is_square(self) -> bool:
|
||||||
return isclose(self.x, self.y, rel_tol=1e-3)
|
return isclose(self.x, self.y, rel_tol=1e-3)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def is_finite(self) -> bool:
|
||||||
|
return isfinite(self.x) and isfinite(self.y)
|
||||||
|
|
||||||
def take_max(self, vals, yvals=None):
|
def take_max(self, vals, yvals=None):
|
||||||
if yvals is not None:
|
if yvals is not None:
|
||||||
return Resolution(max(self.x, *vals), max(self.y, *yvals))
|
return Resolution(max(self.x, *vals), max(self.y, *yvals))
|
||||||
@@ -64,13 +58,22 @@ class Resolution(namedtuple('Resolution', ('x', 'y'))):
|
|||||||
def __str__(self):
|
def __str__(self):
|
||||||
return f"{self.x:f}x{self.y:f}"
|
return f"{self.x:f}x{self.y:f}"
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self): # pragma: no cover
|
||||||
return f"Resolution({self.x}x{self.y} dpi)"
|
return f"Resolution({self.x}x{self.y} dpi)"
|
||||||
|
|
||||||
|
|
||||||
|
class NeverRaise(Exception):
|
||||||
|
"""An exception that is never raised"""
|
||||||
|
|
||||||
|
|
||||||
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
||||||
"""
|
"""Create a symbolic link at ``soft_link_name``, which references ``input_file``.
|
||||||
Helper function: relinks soft symbolic link if necessary
|
|
||||||
|
Think of this as copying ``input_file`` to ``soft_link_name`` with less overhead.
|
||||||
|
|
||||||
|
Use symlinks safely. Self-linking loops are prevented. On Windows, file copy is
|
||||||
|
used since symlinks may require administrator privileges. An existing link at the
|
||||||
|
destination is removed.
|
||||||
"""
|
"""
|
||||||
input_file = os.fspath(input_file)
|
input_file = os.fspath(input_file)
|
||||||
soft_link_name = os.fspath(soft_link_name)
|
soft_link_name = os.fspath(soft_link_name)
|
||||||
@@ -78,8 +81,8 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
|||||||
# Guard against soft linking to oneself
|
# Guard against soft linking to oneself
|
||||||
if input_file == soft_link_name:
|
if input_file == soft_link_name:
|
||||||
log.warning(
|
log.warning(
|
||||||
"No symbolic link made. You are using "
|
"No symbolic link created. You are using the original data directory "
|
||||||
"the original data directory as the working directory."
|
"as the working directory."
|
||||||
)
|
)
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -88,10 +91,7 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
|||||||
# do not delete or overwrite real (non-soft link) file
|
# do not delete or overwrite real (non-soft link) file
|
||||||
if not os.path.islink(soft_link_name):
|
if not os.path.islink(soft_link_name):
|
||||||
raise FileExistsError(f"{soft_link_name} exists and is not a link")
|
raise FileExistsError(f"{soft_link_name} exists and is not a link")
|
||||||
try:
|
os.unlink(soft_link_name)
|
||||||
os.unlink(soft_link_name)
|
|
||||||
except OSError:
|
|
||||||
log.debug("Can't unlink %s", soft_link_name)
|
|
||||||
|
|
||||||
if not os.path.exists(input_file):
|
if not os.path.exists(input_file):
|
||||||
raise FileNotFoundError(f"trying to create a broken symlink to {input_file}")
|
raise FileNotFoundError(f"trying to create a broken symlink to {input_file}")
|
||||||
@@ -120,7 +120,7 @@ def is_iterable_notstr(thing: Any) -> bool:
|
|||||||
|
|
||||||
|
|
||||||
def monotonic(L: Sequence) -> bool:
|
def monotonic(L: Sequence) -> bool:
|
||||||
"""Does list increase monotonically?"""
|
"""Does this sequence increase monotonically?"""
|
||||||
return all(b > a for a, b in zip(L, L[1:]))
|
return all(b > a for a, b in zip(L, L[1:]))
|
||||||
|
|
||||||
|
|
||||||
@@ -179,59 +179,76 @@ def is_file_writable(test_file: os.PathLike) -> bool:
|
|||||||
def check_pdf(input_file: Path) -> bool:
|
def check_pdf(input_file: Path) -> bool:
|
||||||
"""Check if a PDF complies with the PDF specification.
|
"""Check if a PDF complies with the PDF specification.
|
||||||
|
|
||||||
Checks for proper formatting and proper linearization.
|
Checks for proper formatting and proper linearization. Uses pikepdf (which in
|
||||||
|
turn, uses QPDF) to perform the checks.
|
||||||
"""
|
"""
|
||||||
pdf = None
|
|
||||||
try:
|
try:
|
||||||
pdf = pikepdf.open(input_file)
|
pdf = pikepdf.open(input_file)
|
||||||
except pikepdf.PdfError as e:
|
except pikepdf.PdfError as e:
|
||||||
log.error(e)
|
log.error(e)
|
||||||
return False
|
return False
|
||||||
else:
|
else:
|
||||||
messages = pdf.check()
|
with pdf:
|
||||||
for msg in messages:
|
messages = pdf.check()
|
||||||
if 'error' in msg.lower():
|
for msg in messages:
|
||||||
log.error(msg)
|
if 'error' in msg.lower():
|
||||||
|
log.error(msg)
|
||||||
|
else:
|
||||||
|
log.warning(msg)
|
||||||
|
|
||||||
|
sio = StringIO()
|
||||||
|
linearize_msgs = ''
|
||||||
|
try:
|
||||||
|
# If linearization is missing entirely, we do not complain. We do
|
||||||
|
# complain if linearization is present but incorrect.
|
||||||
|
pdf.check_linearization(sio)
|
||||||
|
except RuntimeError:
|
||||||
|
pass
|
||||||
|
except (
|
||||||
|
# Workaround for a problematic pikepdf version
|
||||||
|
# pragma: no cover
|
||||||
|
getattr(pikepdf, 'ForeignObjectError')
|
||||||
|
if pikepdf.__version__ == '2.1.0'
|
||||||
|
else NeverRaise
|
||||||
|
):
|
||||||
|
pass
|
||||||
else:
|
else:
|
||||||
log.warning(msg)
|
linearize_msgs = sio.getvalue()
|
||||||
|
if linearize_msgs:
|
||||||
|
log.warning(linearize_msgs)
|
||||||
|
|
||||||
sio = StringIO()
|
if not messages and not linearize_msgs:
|
||||||
linearize = None
|
return True
|
||||||
try:
|
return False
|
||||||
pdf.check_linearization(sio)
|
|
||||||
except RuntimeError:
|
|
||||||
pass
|
|
||||||
else:
|
|
||||||
linearize = sio.getvalue()
|
|
||||||
if linearize:
|
|
||||||
log.warning(linearize)
|
|
||||||
|
|
||||||
if not messages and not linearize:
|
|
||||||
return True
|
|
||||||
return False
|
|
||||||
finally:
|
|
||||||
if pdf:
|
|
||||||
pdf.close()
|
|
||||||
|
|
||||||
|
|
||||||
T = TypeVar('T')
|
def clamp(n, smallest, largest): # mypy doesn't understand types for this
|
||||||
|
"""Clamps the value of ``n`` to between ``smallest`` and ``largest``."""
|
||||||
|
|
||||||
def clamp(n: T, smallest: T, largest: T) -> T:
|
|
||||||
"""Clamps the value of n to between smallest and largest."""
|
|
||||||
return max(smallest, min(n, largest))
|
return max(smallest, min(n, largest))
|
||||||
|
|
||||||
|
|
||||||
|
def remove_all_log_handlers(logger):
|
||||||
|
"Remove all log handlers, usually used in a child process."
|
||||||
|
for handler in logger.handlers[:]:
|
||||||
|
logger.removeHandler(handler)
|
||||||
|
handler.close() # To ensure handlers with opened resources are released
|
||||||
|
|
||||||
|
|
||||||
def pikepdf_enable_mmap():
|
def pikepdf_enable_mmap():
|
||||||
try:
|
# try:
|
||||||
if pikepdf._qpdf.set_access_default_mmap(True):
|
# if pikepdf._qpdf.set_access_default_mmap(True):
|
||||||
log.debug("pikepdf mmap enabled")
|
# log.debug("pikepdf mmap enabled")
|
||||||
except AttributeError:
|
# except AttributeError:
|
||||||
log.debug("pikepdf mmap not available")
|
# log.debug("pikepdf mmap not available")
|
||||||
|
# We found a race condition probably related to pybind issue #2252 that can
|
||||||
|
# cause a crash. For now, disable pikepdf mmap to be on the safe side.
|
||||||
|
# Fix is not in pybind11 2.6.0
|
||||||
|
# log.debug("pikepdf mmap disabled")
|
||||||
|
return
|
||||||
|
|
||||||
|
|
||||||
def deprecated(func):
|
def deprecated(func):
|
||||||
"""Warn that function is deprecated"""
|
"""Warn that function is deprecated."""
|
||||||
|
|
||||||
@wraps(func)
|
@wraps(func)
|
||||||
def new_func(*args, **kwargs):
|
def new_func(*args, **kwargs):
|
||||||
|
|||||||
@@ -31,18 +31,82 @@
|
|||||||
import argparse
|
import argparse
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
from collections import namedtuple
|
|
||||||
from itertools import chain
|
from itertools import chain
|
||||||
from math import atan, cos, sin
|
from math import atan, cos, sin
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional, Tuple, Union
|
from typing import Any, NamedTuple, Optional, Tuple, Union
|
||||||
from xml.etree import ElementTree
|
from xml.etree import ElementTree
|
||||||
|
|
||||||
from reportlab.lib.colors import black, cyan, magenta, red
|
from reportlab.lib.colors import black, cyan, magenta, red
|
||||||
from reportlab.lib.units import inch
|
from reportlab.lib.units import inch
|
||||||
from reportlab.pdfgen.canvas import Canvas
|
from reportlab.pdfgen.canvas import Canvas
|
||||||
|
|
||||||
Rect = namedtuple('Rect', ['x1', 'y1', 'x2', 'y2'])
|
# According to Wikipedia these languages are supported in the ISO-8859-1 character
|
||||||
|
# set, meaning reportlab can generate them and they are compatible with hocr,
|
||||||
|
# assuming Tesseract has the necessary languages installed. Note that there may
|
||||||
|
# not be language packs for them.
|
||||||
|
HOCR_OK_LANGS = frozenset(
|
||||||
|
[
|
||||||
|
# Languages fully covered by Latin-1:
|
||||||
|
'afr', # Afrikaans
|
||||||
|
'alb', # Albanian
|
||||||
|
'ast', # Leonese
|
||||||
|
'baq', # Basque
|
||||||
|
'bre', # Breton
|
||||||
|
'cos', # Corsican
|
||||||
|
'eng', # English
|
||||||
|
'eus', # Basque
|
||||||
|
'fao', # Faoese
|
||||||
|
'gla', # Scottish Gaelic
|
||||||
|
'glg', # Galician
|
||||||
|
'glv', # Manx
|
||||||
|
'ice', # Icelandic
|
||||||
|
'ind', # Indonesian
|
||||||
|
'isl', # Icelandic
|
||||||
|
'ita', # Italian
|
||||||
|
'ltz', # Luxembourgish
|
||||||
|
'mal', # Malay Rumi
|
||||||
|
'mga', # Irish
|
||||||
|
'nor', # Norwegian
|
||||||
|
'oci', # Occitan
|
||||||
|
'por', # Portugeuse
|
||||||
|
'roh', # Romansh
|
||||||
|
'sco', # Scots
|
||||||
|
'sma', # Sami
|
||||||
|
'spa', # Spanish
|
||||||
|
'sqi', # Albanian
|
||||||
|
'swa', # Swahili
|
||||||
|
'swe', # Swedish
|
||||||
|
'tgl', # Tagalog
|
||||||
|
'wln', # Walloon
|
||||||
|
# Languages supported by Latin-1 except for a few rare characters that OCR
|
||||||
|
# is probably not trained to recognize anyway:
|
||||||
|
'cat', # Catalan
|
||||||
|
'cym', # Welsh
|
||||||
|
'dan', # Danish
|
||||||
|
'deu', # German
|
||||||
|
'dut', # Dutch
|
||||||
|
'est', # Estonian
|
||||||
|
'fin', # Finnish
|
||||||
|
'fra', # French
|
||||||
|
'hun', # Hungarian
|
||||||
|
'kur', # Kurdish
|
||||||
|
'nld', # Dutch
|
||||||
|
'wel', # Welsh
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
Element = ElementTree.Element
|
||||||
|
|
||||||
|
|
||||||
|
class Rect(NamedTuple): # pylint: disable=inherit-non-class
|
||||||
|
"""A rectangle for managing PDF coordinates."""
|
||||||
|
|
||||||
|
x1: Any
|
||||||
|
y1: Any
|
||||||
|
x2: Any
|
||||||
|
y2: Any
|
||||||
|
|
||||||
|
|
||||||
class HocrTransformError(Exception):
|
class HocrTransformError(Exception):
|
||||||
@@ -69,7 +133,7 @@ class HocrTransform:
|
|||||||
{'ff': 'ff', 'ffi': 'ffi', 'ffl': 'ffl', 'fi': 'fi', 'fl': 'fl'}
|
{'ff': 'ff', 'ffi': 'ffi', 'ffl': 'ffl', 'fi': 'fi', 'fl': 'fl'}
|
||||||
)
|
)
|
||||||
|
|
||||||
def __init__(self, hocr_filename: Union[str, Path], dpi: float):
|
def __init__(self, *, hocr_filename: Union[str, Path], dpi: float):
|
||||||
self.dpi = dpi
|
self.dpi = dpi
|
||||||
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
||||||
|
|
||||||
@@ -105,7 +169,7 @@ class HocrTransform:
|
|||||||
else:
|
else:
|
||||||
return ''
|
return ''
|
||||||
|
|
||||||
def _get_element_text(self, element):
|
def _get_element_text(self, element: Element):
|
||||||
"""
|
"""
|
||||||
Return the textual content of the element and its children
|
Return the textual content of the element and its children
|
||||||
"""
|
"""
|
||||||
@@ -119,7 +183,7 @@ class HocrTransform:
|
|||||||
return text
|
return text
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def element_coordinates(cls, element) -> Rect:
|
def element_coordinates(cls, element: Element) -> Rect:
|
||||||
"""
|
"""
|
||||||
Returns a tuple containing the coordinates of the bounding box around
|
Returns a tuple containing the coordinates of the bounding box around
|
||||||
an element
|
an element
|
||||||
@@ -133,7 +197,7 @@ class HocrTransform:
|
|||||||
return out
|
return out
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def baseline(cls, element) -> Tuple[float, float]:
|
def baseline(cls, element: Element) -> Tuple[float, float]:
|
||||||
"""
|
"""
|
||||||
Returns a tuple containing the baseline slope and intercept.
|
Returns a tuple containing the baseline slope and intercept.
|
||||||
"""
|
"""
|
||||||
@@ -149,7 +213,7 @@ class HocrTransform:
|
|||||||
"""
|
"""
|
||||||
return Rect._make((c / self.dpi * inch) for c in pxl)
|
return Rect._make((c / self.dpi * inch) for c in pxl)
|
||||||
|
|
||||||
def _child_xpath(self, html_tag, html_class=None):
|
def _child_xpath(self, html_tag: str, html_class: Optional[str] = None) -> str:
|
||||||
xpath = f".//{self.xmlns}{html_tag}"
|
xpath = f".//{self.xmlns}{html_tag}"
|
||||||
if html_class:
|
if html_class:
|
||||||
xpath += f"[@class='{html_class}']"
|
xpath += f"[@class='{html_class}']"
|
||||||
@@ -167,10 +231,14 @@ class HocrTransform:
|
|||||||
def topdown_position(self, element):
|
def topdown_position(self, element):
|
||||||
pxl_line_coords = self.element_coordinates(element)
|
pxl_line_coords = self.element_coordinates(element)
|
||||||
line_box = self.pt_from_pixel(pxl_line_coords)
|
line_box = self.pt_from_pixel(pxl_line_coords)
|
||||||
return -line_box.y2
|
# Coordinates here are still in the hocr coordinate system, so 0 on the y axis
|
||||||
|
# is the top of the page and increasing values of y will move towards the
|
||||||
|
# bottom of the page.
|
||||||
|
return line_box.y2
|
||||||
|
|
||||||
def to_pdf(
|
def to_pdf(
|
||||||
self,
|
self,
|
||||||
|
*,
|
||||||
out_filename: Path,
|
out_filename: Path,
|
||||||
image_filename: Optional[Path] = None,
|
image_filename: Optional[Path] = None,
|
||||||
show_bounding_boxes: bool = False,
|
show_bounding_boxes: bool = False,
|
||||||
@@ -277,13 +345,15 @@ class HocrTransform:
|
|||||||
def _do_line(
|
def _do_line(
|
||||||
self,
|
self,
|
||||||
pdf: Canvas,
|
pdf: Canvas,
|
||||||
line,
|
line: Optional[Element],
|
||||||
elemclass: str,
|
elemclass: str,
|
||||||
fontname: str,
|
fontname: str,
|
||||||
invisible_text: bool,
|
invisible_text: bool,
|
||||||
interword_spaces: bool,
|
interword_spaces: bool,
|
||||||
show_bounding_boxes: bool,
|
show_bounding_boxes: bool,
|
||||||
):
|
):
|
||||||
|
if not line:
|
||||||
|
return
|
||||||
pxl_line_coords = self.element_coordinates(line)
|
pxl_line_coords = self.element_coordinates(line)
|
||||||
line_box = self.pt_from_pixel(pxl_line_coords)
|
line_box = self.pt_from_pixel(pxl_line_coords)
|
||||||
line_height = line_box.y2 - line_box.y1
|
line_height = line_box.y2 - line_box.y1
|
||||||
@@ -420,10 +490,10 @@ if __name__ == "__main__":
|
|||||||
parser.add_argument('outputfile', help='Path to the PDF file to be generated')
|
parser.add_argument('outputfile', help='Path to the PDF file to be generated')
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
hocr = HocrTransform(args.hocrfile, args.resolution)
|
hocr = HocrTransform(hocr_filename=args.hocrfile, dpi=args.resolution)
|
||||||
hocr.to_pdf(
|
hocr.to_pdf(
|
||||||
args.outputfile,
|
out_filename=args.outputfile,
|
||||||
args.image,
|
image_filename=args.image,
|
||||||
args.boundingboxes,
|
show_bounding_boxes=args.boundingboxes,
|
||||||
interword_spaces=args.interword_spaces,
|
interword_spaces=args.interword_spaces,
|
||||||
)
|
)
|
||||||
|
|||||||
+62
-40
@@ -3,20 +3,10 @@
|
|||||||
#
|
#
|
||||||
# © 2013-16: jbarlow83 from Github (https://github.com/jbarlow83)
|
# © 2013-16: jbarlow83 from Github (https://github.com/jbarlow83)
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
#
|
#
|
||||||
# Python FFI wrapper for Leptonica library
|
# Python FFI wrapper for Leptonica library
|
||||||
|
|
||||||
@@ -25,7 +15,6 @@ import logging
|
|||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
import warnings
|
|
||||||
from collections import deque
|
from collections import deque
|
||||||
from collections.abc import Sequence
|
from collections.abc import Sequence
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
@@ -34,18 +23,20 @@ from functools import lru_cache
|
|||||||
from io import BytesIO, UnsupportedOperation
|
from io import BytesIO, UnsupportedOperation
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from tempfile import TemporaryFile
|
from tempfile import TemporaryFile
|
||||||
|
from warnings import warn
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
from ocrmypdf.lib._leptonica import ffi
|
from ocrmypdf.lib._leptonica import ffi
|
||||||
from ocrmypdf.subprocess import shim_paths_with_program_files
|
|
||||||
|
|
||||||
# pylint: disable=protected-access
|
# pylint: disable=protected-access
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
if os.name == 'nt':
|
if os.name == 'nt':
|
||||||
|
from ocrmypdf.subprocess._windows import shim_env_path
|
||||||
|
|
||||||
libname = 'liblept-5'
|
libname = 'liblept-5'
|
||||||
os.environ['PATH'] = shim_paths_with_program_files()
|
os.environ['PATH'] = shim_env_path()
|
||||||
else:
|
else:
|
||||||
libname = 'lept'
|
libname = 'lept'
|
||||||
_libpath = find_library(libname)
|
_libpath = find_library(libname)
|
||||||
@@ -68,6 +59,24 @@ if not _libpath:
|
|||||||
---------------------------------------------------------------------
|
---------------------------------------------------------------------
|
||||||
"""
|
"""
|
||||||
)
|
)
|
||||||
|
if os.name == 'nt':
|
||||||
|
# On Windows, recent versions of libpng require zlib. We have to make sure
|
||||||
|
# the zlib version being loaded is the same one that libpng was built with.
|
||||||
|
# This tries to import zlib from Tesseract's installation folder, falling back
|
||||||
|
# to find_library() if liblept is being loaded from somewhere else.
|
||||||
|
# Loading zlib from other places could cause a version mismatch
|
||||||
|
_zlib_path = os.path.join(os.path.dirname(_libpath), 'zlib1.dll')
|
||||||
|
if not os.path.exists(_zlib_path):
|
||||||
|
_zlib_path = find_library('zlib')
|
||||||
|
try:
|
||||||
|
zlib = ffi.dlopen(_zlib_path)
|
||||||
|
except ffi.error as e:
|
||||||
|
raise MissingDependencyError(
|
||||||
|
"""
|
||||||
|
Could not load the zlib library. It could be that Tesseract is not installed properly,
|
||||||
|
we can't find the installation on your system PATH environment variable.
|
||||||
|
"""
|
||||||
|
) from e
|
||||||
try:
|
try:
|
||||||
lept = ffi.dlopen(_libpath)
|
lept = ffi.dlopen(_libpath)
|
||||||
lept.setMsgSeverity(lept.L_SEVERITY_WARNING)
|
lept.setMsgSeverity(lept.L_SEVERITY_WARNING)
|
||||||
@@ -79,9 +88,9 @@ except ffi.error as e:
|
|||||||
|
|
||||||
class _LeptonicaErrorTrap_Redirect:
|
class _LeptonicaErrorTrap_Redirect:
|
||||||
"""
|
"""
|
||||||
Context manager to trap errors reported by Leptonica.
|
Context manager to trap errors reported by Leptonica < 1.79 or on Apple Silicon.
|
||||||
|
|
||||||
Leptonica's error return codes don't provide much informatino about what
|
Leptonica's error return codes don't provide much information about what
|
||||||
went wrong. Leptonica does, however, write more detailed errors to stderr
|
went wrong. Leptonica does, however, write more detailed errors to stderr
|
||||||
(provided this is not disabled at compile time). The Leptonica source
|
(provided this is not disabled at compile time). The Leptonica source
|
||||||
code is very consistent in its use of macros to generate errors.
|
code is very consistent in its use of macros to generate errors.
|
||||||
@@ -105,8 +114,10 @@ class _LeptonicaErrorTrap_Redirect:
|
|||||||
# Save the old stderr, and redirect stderr to temporary file
|
# Save the old stderr, and redirect stderr to temporary file
|
||||||
self.leptonica_lock.acquire()
|
self.leptonica_lock.acquire()
|
||||||
try:
|
try:
|
||||||
with suppress(AttributeError):
|
# It would make sense to do sys.stderr.flush() here, but that can deadlock
|
||||||
sys.stderr.flush()
|
# due to https://bugs.python.org/issue6721. So don't flush. Pretend
|
||||||
|
# there's nothing important in sys.stderr. If the user cared they would
|
||||||
|
# be using Leptonica 1.79 or later anyway to avoid this mess.
|
||||||
self.copy_of_stderr = os.dup(sys.stderr.fileno())
|
self.copy_of_stderr = os.dup(sys.stderr.fileno())
|
||||||
os.dup2(self.tmpfile.fileno(), sys.stderr.fileno(), inheritable=False)
|
os.dup2(self.tmpfile.fileno(), sys.stderr.fileno(), inheritable=False)
|
||||||
except AttributeError:
|
except AttributeError:
|
||||||
@@ -161,20 +172,6 @@ tls = threading.local()
|
|||||||
tls.trap = None
|
tls.trap = None
|
||||||
|
|
||||||
|
|
||||||
@ffi.callback("void(char *)")
|
|
||||||
def _stderr_handler(cstr):
|
|
||||||
msg = ffi.string(cstr).decode(errors='replace')
|
|
||||||
if msg.startswith("Error"):
|
|
||||||
logger.error(msg)
|
|
||||||
elif msg.startswith("Warning"):
|
|
||||||
logger.warning(msg)
|
|
||||||
else:
|
|
||||||
logger.debug(msg)
|
|
||||||
if tls.trap is not None:
|
|
||||||
tls.trap.append(msg)
|
|
||||||
return
|
|
||||||
|
|
||||||
|
|
||||||
class _LeptonicaErrorTrap_Queue:
|
class _LeptonicaErrorTrap_Queue:
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
self.queue = deque()
|
self.queue = deque()
|
||||||
@@ -195,18 +192,40 @@ class _LeptonicaErrorTrap_Queue:
|
|||||||
if 'Error' in output:
|
if 'Error' in output:
|
||||||
if 'image file not found' in output:
|
if 'image file not found' in output:
|
||||||
raise FileNotFoundError()
|
raise FileNotFoundError()
|
||||||
if 'pixWrite: stream not opened' in output:
|
elif 'pixWrite: stream not opened' in output:
|
||||||
raise LeptonicaIOError()
|
raise LeptonicaIOError()
|
||||||
if 'index not valid' in output:
|
elif 'index not valid' in output:
|
||||||
raise IndexError()
|
raise IndexError()
|
||||||
raise LeptonicaError(output)
|
elif 'pixGetInvBackgroundMap: w and h must be >= 5' in output:
|
||||||
|
logger.warning(
|
||||||
|
"Leptonica attempted to remove background from a low resolution - "
|
||||||
|
"you may want to review in a PDF viewer"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
raise LeptonicaError(output)
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
|
|
||||||
|
@ffi.callback("void(char *)")
|
||||||
|
def _stderr_handler(cstr):
|
||||||
|
msg = ffi.string(cstr).decode(errors='replace')
|
||||||
|
if msg.startswith("Error"):
|
||||||
|
logger.error(msg)
|
||||||
|
elif msg.startswith("Warning"):
|
||||||
|
logger.warning(msg)
|
||||||
|
else:
|
||||||
|
logger.debug(msg)
|
||||||
|
if tls.trap is not None:
|
||||||
|
tls.trap.append(msg)
|
||||||
|
return
|
||||||
|
|
||||||
lept.leptSetStderrHandler(_stderr_handler)
|
lept.leptSetStderrHandler(_stderr_handler)
|
||||||
except ffi.error:
|
except (ffi.error, MemoryError):
|
||||||
# Pre-1.79 Leptonica does not have leptSetStderrHandler
|
# Pre-1.79 Leptonica does not have leptSetStderrHandler
|
||||||
|
# And some platforms, notably Apple ARM 64, do not allow the write+execute
|
||||||
|
# memory needed to set up the callback function.
|
||||||
_LeptonicaErrorTrap = _LeptonicaErrorTrap_Redirect
|
_LeptonicaErrorTrap = _LeptonicaErrorTrap_Redirect
|
||||||
else:
|
else:
|
||||||
# 1.79 have this new symbol
|
# 1.79 have this new symbol
|
||||||
@@ -378,7 +397,7 @@ class Pix(LeptonicaObject):
|
|||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def read(cls, path):
|
def read(cls, path):
|
||||||
warnings.warn('Use Pix.open() instead', DeprecationWarning)
|
warn('Use Pix.open() instead', DeprecationWarning)
|
||||||
return cls.open(path)
|
return cls.open(path)
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
@@ -643,6 +662,9 @@ class Pix(LeptonicaObject):
|
|||||||
bg_val=200,
|
bg_val=200,
|
||||||
smooth_kernel=(2, 1),
|
smooth_kernel=(2, 1),
|
||||||
):
|
):
|
||||||
|
if self.width < tile_size[0] or self.height < tile_size[1]:
|
||||||
|
logger.info("Skipped pixMaskedThreshOnBackgroundNorm on small image")
|
||||||
|
return self
|
||||||
# Background norm doesn't work on color mapped Pix, so remove colormap
|
# Background norm doesn't work on color mapped Pix, so remove colormap
|
||||||
target_pix = self.remove_colormap(lept.REMOVE_CMAP_BASED_ON_SRC)
|
target_pix = self.remove_colormap(lept.REMOVE_CMAP_BASED_ON_SRC)
|
||||||
with _LeptonicaErrorTrap():
|
with _LeptonicaErrorTrap():
|
||||||
|
|||||||
@@ -1,18 +1,8 @@
|
|||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
# © 2017 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Bindings to external libraries"""
|
"""Bindings to external libraries"""
|
||||||
|
|||||||
@@ -1,20 +1,10 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
# © 2017 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
|||||||
+173
-189
@@ -1,29 +1,17 @@
|
|||||||
# © 2018 James R. Barlow: github.com/jbarlow83
|
# © 2018 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from functools import partial
|
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import (
|
from typing import (
|
||||||
Any,
|
|
||||||
Callable,
|
Callable,
|
||||||
Dict,
|
Dict,
|
||||||
Iterator,
|
Iterator,
|
||||||
@@ -34,16 +22,15 @@ from typing import (
|
|||||||
Optional,
|
Optional,
|
||||||
Sequence,
|
Sequence,
|
||||||
Tuple,
|
Tuple,
|
||||||
Union,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
import img2pdf
|
||||||
import pikepdf
|
import pikepdf
|
||||||
from pikepdf import Dictionary, Name, Object, Pdf, PdfImage
|
from pikepdf import Dictionary, Name, Object, Pdf, PdfImage
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
from tqdm import tqdm
|
|
||||||
|
|
||||||
from ocrmypdf import leptonica
|
from ocrmypdf import leptonica
|
||||||
from ocrmypdf._concurrent import exec_progress_pool
|
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||||
from ocrmypdf._exec import jbig2enc, pngquant
|
from ocrmypdf._exec import jbig2enc, pngquant
|
||||||
from ocrmypdf._jobcontext import PdfContext
|
from ocrmypdf._jobcontext import PdfContext
|
||||||
from ocrmypdf.exceptions import OutputFileAccessError
|
from ocrmypdf.exceptions import OutputFileAccessError
|
||||||
@@ -58,7 +45,7 @@ DEFAULT_PNG_QUALITY = 70
|
|||||||
Xref = NewType('Xref', int)
|
Xref = NewType('Xref', int)
|
||||||
|
|
||||||
|
|
||||||
class XrefExt(NamedTuple):
|
class XrefExt(NamedTuple): # pylint: disable=inherit-non-class
|
||||||
xref: Xref
|
xref: Xref
|
||||||
ext: str
|
ext: str
|
||||||
|
|
||||||
@@ -75,33 +62,42 @@ def jpg_name(root: Path, xref: Xref) -> Path:
|
|||||||
return img_name(root, xref, '.jpg')
|
return img_name(root, xref, '.jpg')
|
||||||
|
|
||||||
|
|
||||||
def tif_name(root: Path, xref: Xref) -> Path:
|
|
||||||
return img_name(root, xref, '.tif')
|
|
||||||
|
|
||||||
|
|
||||||
def extract_image_filter(
|
def extract_image_filter(
|
||||||
pike: Pdf, root: Path, image: Object, xref: Xref
|
pike: Pdf, root: Path, image: Object, xref: Xref
|
||||||
) -> Optional[Tuple[PdfImage, Tuple[Name, Object]]]:
|
) -> Optional[Tuple[PdfImage, Tuple[Name, Object]]]:
|
||||||
|
del pike # unused args
|
||||||
|
del root
|
||||||
|
|
||||||
if image.Subtype != Name.Image:
|
if image.Subtype != Name.Image:
|
||||||
return None
|
return None
|
||||||
if image.Length < 100:
|
if image.Length < 100:
|
||||||
log.debug("Skipping small image, xref %s", xref)
|
log.debug(f"Skipping small image, xref {xref}")
|
||||||
|
return None
|
||||||
|
if image.Width < 8 or image.Height < 8: # Issue 732
|
||||||
|
log.debug(f"Skipping oddly sized image, xref {xref}")
|
||||||
return None
|
return None
|
||||||
|
|
||||||
pim = PdfImage(image)
|
pim = PdfImage(image)
|
||||||
|
|
||||||
if len(pim.filter_decodeparms) > 1:
|
if len(pim.filter_decodeparms) > 1:
|
||||||
log.debug("Skipping multiply filtered, xref %s", xref)
|
log.debug(f"Skipping multiply filtered image, xref {xref}")
|
||||||
return None
|
return None
|
||||||
filtdp = pim.filter_decodeparms[0]
|
filtdp = pim.filter_decodeparms[0]
|
||||||
|
|
||||||
if pim.bits_per_component > 8:
|
if pim.bits_per_component > 8:
|
||||||
|
log.debug(f"Skipping wide gamut image, xref {xref}")
|
||||||
return None # Don't mess with wide gamut images
|
return None # Don't mess with wide gamut images
|
||||||
|
|
||||||
if filtdp[0] == Name.JPXDecode:
|
if filtdp[0] == Name.JPXDecode:
|
||||||
|
log.debug(f"Skipping JPEG2000 iamge, xref {xref}")
|
||||||
return None # Don't do JPEG2000
|
return None # Don't do JPEG2000
|
||||||
|
|
||||||
|
if filtdp[0] == Name.CCITTFaxDecode and filtdp[1].get('/K', 0) >= 0:
|
||||||
|
log.debug(f"Skipping CCITT Group 3 image, xref {xref}")
|
||||||
|
return None # pikepdf doesn't support Group 3 yet
|
||||||
|
|
||||||
if Name.Decode in image:
|
if Name.Decode in image:
|
||||||
|
log.debug(f"Skipping image with Decode table, xref {xref}")
|
||||||
return None # Don't mess with custom Decode tables
|
return None # Don't mess with custom Decode tables
|
||||||
|
|
||||||
return pim, filtdp
|
return pim, filtdp
|
||||||
@@ -110,6 +106,8 @@ def extract_image_filter(
|
|||||||
def extract_image_jbig2(
|
def extract_image_jbig2(
|
||||||
*, pike: pikepdf.Pdf, root: Path, image: Object, xref: Xref, options
|
*, pike: pikepdf.Pdf, root: Path, image: Object, xref: Xref, options
|
||||||
) -> Optional[XrefExt]:
|
) -> Optional[XrefExt]:
|
||||||
|
del options # unused arg
|
||||||
|
|
||||||
result = extract_image_filter(pike, root, image, xref)
|
result = extract_image_filter(pike, root, image, xref)
|
||||||
if result is None:
|
if result is None:
|
||||||
return None
|
return None
|
||||||
@@ -120,14 +118,29 @@ def extract_image_jbig2(
|
|||||||
and filtdp[0] != Name.JBIG2Decode
|
and filtdp[0] != Name.JBIG2Decode
|
||||||
and jbig2enc.available()
|
and jbig2enc.available()
|
||||||
):
|
):
|
||||||
try:
|
# Save any colorspace associated with the image, so that we
|
||||||
imgname = root / f'{xref:08d}'
|
# will export a pure 1-bit PNG with no palette or ICC profile.
|
||||||
with imgname.open('wb') as f:
|
# Showing the palette or ICC to jbig2enc will cause it to perform
|
||||||
ext = pim.extract_to(stream=f)
|
# colorspace transform to 1bpp, which will conflict the palette or
|
||||||
imgname.rename(imgname.with_suffix(ext))
|
# ICC if it exists.
|
||||||
except pikepdf.UnsupportedImageTypeError:
|
colorspace = pim.obj.get(pikepdf.Name.ColorSpace, None)
|
||||||
return None
|
if colorspace is not None or pim.image_mask:
|
||||||
return XrefExt(xref, ext)
|
try:
|
||||||
|
# Set to DeviceGray temporarily; we already in 1 bpc.
|
||||||
|
pim.obj.ColorSpace = pikepdf.Name.DeviceGray
|
||||||
|
imgname = root / f'{xref:08d}'
|
||||||
|
with imgname.open('wb') as f:
|
||||||
|
ext = pim.extract_to(stream=f)
|
||||||
|
imgname.rename(imgname.with_suffix(ext))
|
||||||
|
except pikepdf.UnsupportedImageTypeError:
|
||||||
|
return None
|
||||||
|
finally:
|
||||||
|
# Restore image colorspace after temporarily setting it to DeviceGray
|
||||||
|
if colorspace is not None:
|
||||||
|
pim.obj.ColorSpace = colorspace
|
||||||
|
else:
|
||||||
|
del pim.obj.ColorSpace
|
||||||
|
return XrefExt(xref, ext)
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
@@ -143,11 +156,6 @@ def extract_image_generic(
|
|||||||
if pim.bits_per_component == 1:
|
if pim.bits_per_component == 1:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
try:
|
|
||||||
pim.indexed # pikepdf 1.6.3 can't handle [/Indexed [/Array...]]
|
|
||||||
except NotImplementedError:
|
|
||||||
return None
|
|
||||||
|
|
||||||
if filtdp[0] == Name.DCTDecode and options.optimize >= 2:
|
if filtdp[0] == Name.DCTDecode and options.optimize >= 2:
|
||||||
# This is a simple heuristic derived from some training data, that has
|
# This is a simple heuristic derived from some training data, that has
|
||||||
# about a 70% chance of guessing whether the JPEG is high quality,
|
# about a 70% chance of guessing whether the JPEG is high quality,
|
||||||
@@ -203,7 +211,10 @@ def extract_image_generic(
|
|||||||
|
|
||||||
|
|
||||||
def extract_images(
|
def extract_images(
|
||||||
pike: Pdf, root: Path, options, extract_fn: Callable[..., Optional[XrefExt]],
|
pike: Pdf,
|
||||||
|
root: Path,
|
||||||
|
options,
|
||||||
|
extract_fn: Callable[..., Optional[XrefExt]],
|
||||||
) -> Iterator[Tuple[int, XrefExt]]:
|
) -> Iterator[Tuple[int, XrefExt]]:
|
||||||
"""Extract image using extract_fn
|
"""Extract image using extract_fn
|
||||||
|
|
||||||
@@ -237,7 +248,9 @@ def extract_images(
|
|||||||
# Ignore soft masks
|
# Ignore soft masks
|
||||||
smask_xref = Xref(image.SMask.objgen[0])
|
smask_xref = Xref(image.SMask.objgen[0])
|
||||||
exclude_xrefs.add(smask_xref)
|
exclude_xrefs.add(smask_xref)
|
||||||
|
log.debug(f"Skipping image {smask_xref} because it is an SMask")
|
||||||
include_xrefs.add(xref)
|
include_xrefs.add(xref)
|
||||||
|
log.debug(f"Treating {xref} as an optimization candidate")
|
||||||
if xref not in pageno_for_xref:
|
if xref not in pageno_for_xref:
|
||||||
pageno_for_xref[xref] = pageno
|
pageno_for_xref[xref] = pageno
|
||||||
|
|
||||||
@@ -248,8 +261,8 @@ def extract_images(
|
|||||||
result = extract_fn(
|
result = extract_fn(
|
||||||
pike=pike, root=root, image=image, xref=xref, options=options
|
pike=pike, root=root, image=image, xref=xref, options=options
|
||||||
)
|
)
|
||||||
except Exception as e: # pylint: disable=broad-except
|
except Exception: # pylint: disable=broad-except
|
||||||
log.debug("Image xref %s, error %s", xref, repr(e))
|
log.exception(f"While extracting image xref {xref}, an error occurred")
|
||||||
errors += 1
|
errors += 1
|
||||||
else:
|
else:
|
||||||
if result:
|
if result:
|
||||||
@@ -282,26 +295,22 @@ def extract_images_jbig2(pike: Pdf, root: Path, options) -> Dict[int, List[XrefE
|
|||||||
group = pageno // options.jbig2_page_group_size
|
group = pageno // options.jbig2_page_group_size
|
||||||
jbig2_groups[group].append(xref_ext)
|
jbig2_groups[group].append(xref_ext)
|
||||||
|
|
||||||
# Elide empty groups
|
|
||||||
jbig2_groups = {
|
|
||||||
group: xrefs for group, xrefs in jbig2_groups.items() if len(xrefs) > 0
|
|
||||||
}
|
|
||||||
log.debug("Optimizable images: JBIG2 groups: %s", (len(jbig2_groups),))
|
log.debug("Optimizable images: JBIG2 groups: %s", (len(jbig2_groups),))
|
||||||
return jbig2_groups
|
return jbig2_groups
|
||||||
|
|
||||||
|
|
||||||
def _produce_jbig2_images(
|
def _produce_jbig2_images(
|
||||||
jbig2_groups: Dict[int, List[XrefExt]], root: Path, options
|
jbig2_groups: Dict[int, List[XrefExt]], root: Path, options, executor: Executor
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Produce JBIG2 images from their groups"""
|
"""Produce JBIG2 images from their groups"""
|
||||||
|
|
||||||
def jbig2_group_args(root: Path, groups: Dict[int, List[XrefExt]]):
|
def jbig2_group_args(root: Path, groups: Dict[int, List[XrefExt]]):
|
||||||
for group, xref_exts in groups.items():
|
for group, xref_exts in groups.items():
|
||||||
prefix = f'group{group:08d}'
|
prefix = f'group{group:08d}'
|
||||||
yield dict(
|
yield (
|
||||||
cwd=fspath(root),
|
fspath(root), # =cwd
|
||||||
infiles=(img_name(root, xref, ext) for xref, ext in xref_exts),
|
(img_name(root, xref, ext) for xref, ext in xref_exts), # =infiles
|
||||||
out_prefix=prefix,
|
prefix, # =out_prefix
|
||||||
)
|
)
|
||||||
|
|
||||||
def jbig2_single_args(root, groups: Dict[int, List[XrefExt]]):
|
def jbig2_single_args(root, groups: Dict[int, List[XrefExt]]):
|
||||||
@@ -310,23 +319,20 @@ def _produce_jbig2_images(
|
|||||||
# Second loop is to ensure multiple images per page are unpacked
|
# Second loop is to ensure multiple images per page are unpacked
|
||||||
for n, xref_ext in enumerate(xref_exts):
|
for n, xref_ext in enumerate(xref_exts):
|
||||||
xref, ext = xref_ext
|
xref, ext = xref_ext
|
||||||
yield dict(
|
yield (
|
||||||
cwd=fspath(root),
|
fspath(root),
|
||||||
infile=img_name(root, xref, ext),
|
img_name(root, xref, ext),
|
||||||
outfile=root / f'{prefix}.{n:04d}',
|
root / f'{prefix}.{n:04d}',
|
||||||
)
|
)
|
||||||
|
|
||||||
def convert_generic(fn, kwargs_dict):
|
|
||||||
return fn(**kwargs_dict)
|
|
||||||
|
|
||||||
if options.jbig2_page_group_size > 1:
|
if options.jbig2_page_group_size > 1:
|
||||||
jbig2_args = jbig2_group_args
|
jbig2_args = jbig2_group_args
|
||||||
jbig2_convert = partial(convert_generic, jbig2enc.convert_group)
|
jbig2_convert = jbig2enc.convert_group_mp
|
||||||
else:
|
else:
|
||||||
jbig2_args = jbig2_single_args
|
jbig2_args = jbig2_single_args
|
||||||
jbig2_convert = partial(convert_generic, jbig2enc.convert_single)
|
jbig2_convert = jbig2enc.convert_single_mp
|
||||||
|
|
||||||
exec_progress_pool(
|
executor(
|
||||||
use_threads=True,
|
use_threads=True,
|
||||||
max_workers=options.jobs,
|
max_workers=options.jobs,
|
||||||
tqdm_kwargs=dict(
|
tqdm_kwargs=dict(
|
||||||
@@ -341,7 +347,11 @@ def _produce_jbig2_images(
|
|||||||
|
|
||||||
|
|
||||||
def convert_to_jbig2(
|
def convert_to_jbig2(
|
||||||
pike: Pdf, jbig2_groups: Dict[int, List[XrefExt]], root: Path, options
|
pike: Pdf,
|
||||||
|
jbig2_groups: Dict[int, List[XrefExt]],
|
||||||
|
root: Path,
|
||||||
|
options,
|
||||||
|
executor: Executor,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Convert images to JBIG2 and insert into PDF.
|
"""Convert images to JBIG2 and insert into PDF.
|
||||||
|
|
||||||
@@ -356,7 +366,7 @@ def convert_to_jbig2(
|
|||||||
and needs no dictionary. Currently this must be lossless JBIG2.
|
and needs no dictionary. Currently this must be lossless JBIG2.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
_produce_jbig2_images(jbig2_groups, root, options)
|
_produce_jbig2_images(jbig2_groups, root, options, executor)
|
||||||
|
|
||||||
for group, xref_exts in jbig2_groups.items():
|
for group, xref_exts in jbig2_groups.items():
|
||||||
prefix = f'group{group:08d}'
|
prefix = f'group{group:08d}'
|
||||||
@@ -380,27 +390,94 @@ def convert_to_jbig2(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def transcode_jpegs(pike: Pdf, jpegs: Sequence[Xref], root: Path, options) -> None:
|
def _optimize_jpeg(args):
|
||||||
for xref in tqdm(
|
xref, in_jpg, opt_jpg, jpeg_quality = args
|
||||||
jpegs, desc="JPEGs", unit='image', disable=not options.progress_bar
|
|
||||||
):
|
|
||||||
in_jpg = jpg_name(root, xref)
|
|
||||||
opt_jpg = in_jpg.with_suffix('.opt.jpg')
|
|
||||||
|
|
||||||
# This produces a debug warning from PIL
|
# This may produce a debug warning from PIL
|
||||||
# DEBUG:PIL.Image:Error closing: 'NoneType' object has no attribute
|
# DEBUG:PIL.Image:Error closing: 'NoneType' object has no attribute
|
||||||
# 'close'. Seems to be mostly harmless
|
# 'close'. Seems to be mostly harmless
|
||||||
# https://github.com/python-pillow/Pillow/issues/1144
|
# https://github.com/python-pillow/Pillow/issues/1144
|
||||||
with Image.open(in_jpg) as im:
|
with Image.open(in_jpg) as im:
|
||||||
im.save(opt_jpg, optimize=True, quality=options.jpeg_quality)
|
im.save(opt_jpg, optimize=True, quality=jpeg_quality)
|
||||||
|
|
||||||
if opt_jpg.stat().st_size > in_jpg.stat().st_size:
|
if opt_jpg.stat().st_size > in_jpg.stat().st_size:
|
||||||
log.debug("xref %s, jpeg, made larger - skip", xref)
|
log.debug("xref %s, jpeg, made larger - skip", xref)
|
||||||
continue
|
opt_jpg.unlink()
|
||||||
|
opt_jpg = None
|
||||||
|
return xref, opt_jpg
|
||||||
|
|
||||||
|
|
||||||
|
def transcode_jpegs(
|
||||||
|
pike: Pdf, jpegs: Sequence[Xref], root: Path, options, executor
|
||||||
|
) -> None:
|
||||||
|
def jpeg_args():
|
||||||
|
for xref in jpegs:
|
||||||
|
in_jpg = jpg_name(root, xref)
|
||||||
|
opt_jpg = in_jpg.with_suffix('.opt.jpg')
|
||||||
|
yield xref, in_jpg, opt_jpg, options.jpeg_quality
|
||||||
|
|
||||||
|
def finish_jpeg(result, pbar):
|
||||||
|
xref, opt_jpg = result
|
||||||
|
if opt_jpg:
|
||||||
|
compdata = leptonica.CompressedData.open(opt_jpg)
|
||||||
|
im_obj = pike.get_object(xref, 0)
|
||||||
|
im_obj.write(compdata.read(), filter=Name.DCTDecode)
|
||||||
|
pbar.update()
|
||||||
|
|
||||||
|
executor(
|
||||||
|
use_threads=True, # Processes are significantly slower at this task
|
||||||
|
max_workers=options.jobs,
|
||||||
|
tqdm_kwargs=dict(
|
||||||
|
desc="JPEGs",
|
||||||
|
total=len(jpegs),
|
||||||
|
unit='image',
|
||||||
|
disable=not options.progress_bar,
|
||||||
|
),
|
||||||
|
task=_optimize_jpeg,
|
||||||
|
task_arguments=jpeg_args(),
|
||||||
|
task_finished=finish_jpeg,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool:
|
||||||
|
output = filename.with_suffix('.png.pdf')
|
||||||
|
with output.open('wb') as f:
|
||||||
|
img2pdf.convert(fspath(filename), outputstream=f)
|
||||||
|
|
||||||
|
with pikepdf.open(output) as pdf_image:
|
||||||
|
foreign_image = next(pdf_image.pages[0].images.values())
|
||||||
|
local_image = pike.copy_foreign(foreign_image)
|
||||||
|
|
||||||
compdata = leptonica.CompressedData.open(opt_jpg)
|
|
||||||
im_obj = pike.get_object(xref, 0)
|
im_obj = pike.get_object(xref, 0)
|
||||||
im_obj.write(compdata.read(), filter=Name.DCTDecode)
|
im_obj.write(
|
||||||
|
local_image.read_raw_bytes(),
|
||||||
|
filter=local_image.Filter,
|
||||||
|
decode_parms=local_image.DecodeParms,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Don't copy keys from the new image...
|
||||||
|
del_keys = set(im_obj.keys()) - set(local_image.keys())
|
||||||
|
# ...except for the keep_fields, which are essential to displaying
|
||||||
|
# the image correctly and preserving its metadata. (/Decode arrays
|
||||||
|
# and /SMaskInData are implicitly discarded prior to this point.)
|
||||||
|
keep_fields = {
|
||||||
|
'/ID',
|
||||||
|
'/Intent',
|
||||||
|
'/Interpolate',
|
||||||
|
'/Mask',
|
||||||
|
'/Metadata',
|
||||||
|
'/OC',
|
||||||
|
'/OPI',
|
||||||
|
'/SMask',
|
||||||
|
'/StructParent',
|
||||||
|
}
|
||||||
|
del_keys -= keep_fields
|
||||||
|
for key in local_image.keys():
|
||||||
|
if key != Name.Length and str(key) not in keep_fields:
|
||||||
|
im_obj[key] = local_image[key]
|
||||||
|
for key in del_keys:
|
||||||
|
del im_obj[key]
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
def transcode_pngs(
|
def transcode_pngs(
|
||||||
@@ -409,6 +486,7 @@ def transcode_pngs(
|
|||||||
image_name_fn: Callable[[Path, Xref], Path],
|
image_name_fn: Callable[[Path, Xref], Path],
|
||||||
root: Path,
|
root: Path,
|
||||||
options,
|
options,
|
||||||
|
executor,
|
||||||
) -> None:
|
) -> None:
|
||||||
modified: MutableSet[Xref] = set()
|
modified: MutableSet[Xref] = set()
|
||||||
if options.optimize >= 2:
|
if options.optimize >= 2:
|
||||||
@@ -428,10 +506,7 @@ def transcode_pngs(
|
|||||||
)
|
)
|
||||||
modified.add(xref)
|
modified.add(xref)
|
||||||
|
|
||||||
def pngquant_fn(args):
|
executor(
|
||||||
pngquant.quantize(*args)
|
|
||||||
|
|
||||||
exec_progress_pool(
|
|
||||||
use_threads=True,
|
use_threads=True,
|
||||||
max_workers=options.jobs,
|
max_workers=options.jobs,
|
||||||
tqdm_kwargs=dict(
|
tqdm_kwargs=dict(
|
||||||
@@ -440,113 +515,22 @@ def transcode_pngs(
|
|||||||
unit='image',
|
unit='image',
|
||||||
disable=not options.progress_bar,
|
disable=not options.progress_bar,
|
||||||
),
|
),
|
||||||
task=pngquant_fn,
|
task=pngquant.quantize_mp,
|
||||||
task_arguments=pngquant_args(),
|
task_arguments=pngquant_args(),
|
||||||
)
|
)
|
||||||
|
|
||||||
for xref in modified:
|
for xref in modified:
|
||||||
im_obj = pike.get_object(xref, 0)
|
filename = png_name(root, xref)
|
||||||
try:
|
_transcode_png(pike, filename, xref)
|
||||||
pix = leptonica.Pix.open(png_name(root, xref))
|
|
||||||
if pix.mode == '1':
|
|
||||||
compdata = pix.generate_pdf_ci_data(leptonica.lept.L_G4_ENCODE, 0)
|
|
||||||
else:
|
|
||||||
compdata = leptonica.CompressedData.open(png_name(root, xref))
|
|
||||||
except leptonica.LeptonicaError as e:
|
|
||||||
# Most likely this means file not found, i.e. quantize did not
|
|
||||||
# produce an improved version
|
|
||||||
log.error(e)
|
|
||||||
continue
|
|
||||||
|
|
||||||
# If re-coded image is larger don't use it - we test here because
|
|
||||||
# pngquant knows the size of the temporary output file but not the actual
|
|
||||||
# object in the PDF
|
|
||||||
if len(compdata) > int(im_obj.stream_dict.Length):
|
|
||||||
log.debug(
|
|
||||||
f"pngquant: pngquant did not improve over original image "
|
|
||||||
f"{len(compdata)} > {int(im_obj.stream_dict.Length)}"
|
|
||||||
)
|
|
||||||
continue
|
|
||||||
if compdata.type == leptonica.lept.L_FLATE_ENCODE:
|
|
||||||
rewrite_png(pike, im_obj, compdata)
|
|
||||||
elif compdata.type == leptonica.lept.L_G4_ENCODE:
|
|
||||||
rewrite_png_as_g4(pike, im_obj, compdata)
|
|
||||||
|
|
||||||
|
|
||||||
def rewrite_png_as_g4(pike: Pdf, im_obj: Object, compdata) -> None:
|
def optimize(
|
||||||
im_obj.BitsPerComponent = 1
|
input_file: Path,
|
||||||
im_obj.Width = compdata.w
|
output_file: Path,
|
||||||
im_obj.Height = compdata.h
|
context,
|
||||||
|
save_settings,
|
||||||
im_obj.write(compdata.read())
|
executor: Executor = SerialExecutor(),
|
||||||
|
) -> None:
|
||||||
log.debug(f"PNG to G4 {im_obj.objgen}")
|
|
||||||
if Name.Predictor in im_obj:
|
|
||||||
del im_obj.Predictor
|
|
||||||
if Name.DecodeParms in im_obj:
|
|
||||||
del im_obj.DecodeParms
|
|
||||||
im_obj.DecodeParms = Dictionary(
|
|
||||||
K=-1, BlackIs1=bool(compdata.minisblack), Columns=compdata.w
|
|
||||||
)
|
|
||||||
|
|
||||||
im_obj.Filter = Name.CCITTFaxDecode
|
|
||||||
return
|
|
||||||
|
|
||||||
|
|
||||||
def rewrite_png(pike: Pdf, im_obj: Object, compdata) -> None:
|
|
||||||
# When a PNG is inserted into a PDF, we more or less copy the IDAT section from
|
|
||||||
# the PDF and transfer the rest of the PNG headers to PDF image metadata.
|
|
||||||
# One thing we have to do is tell the PDF reader whether a predictor was used
|
|
||||||
# on the image before Flate encoding. (Typically one is.)
|
|
||||||
# According to Leptonica source, PDF readers don't actually need us
|
|
||||||
# to specify the correct predictor, they just need a value of either:
|
|
||||||
# 1 - no predictor
|
|
||||||
# 10-14 - there is a predictor
|
|
||||||
# Leptonica's compdata->predictor only tells TRUE or FALSE
|
|
||||||
# 10-14 means the actual predictor is specified in the data, so for any
|
|
||||||
# number >= 10 the PDF reader will use whatever the PNG data specifies.
|
|
||||||
# In practice Leptonica should use Paeth, 14, but 15 seems to be the
|
|
||||||
# designated value for "optimal". So we will use 15.
|
|
||||||
# See:
|
|
||||||
# - PDF RM 7.4.4.4 Table 10
|
|
||||||
# - https://github.com/DanBloomberg/leptonica/blob/master/src/pdfio2.c#L757
|
|
||||||
predictor = 15 if compdata.predictor > 0 else 1
|
|
||||||
dparms = Dictionary(Predictor=predictor)
|
|
||||||
if predictor > 1:
|
|
||||||
dparms.BitsPerComponent = compdata.bps # Yes, this is redundant
|
|
||||||
dparms.Colors = compdata.spp
|
|
||||||
dparms.Columns = compdata.w
|
|
||||||
|
|
||||||
im_obj.BitsPerComponent = compdata.bps
|
|
||||||
im_obj.Width = compdata.w
|
|
||||||
im_obj.Height = compdata.h
|
|
||||||
|
|
||||||
log.debug(
|
|
||||||
f"PNG {im_obj.objgen}: palette={compdata.ncolors} spp={compdata.spp} bps={compdata.bps}"
|
|
||||||
)
|
|
||||||
if compdata.ncolors > 0:
|
|
||||||
# .ncolors is the number of colors in the palette, not the number of
|
|
||||||
# colors used in a true color image. The palette string is always
|
|
||||||
# given as RGB tuples even when the image is grayscale; see
|
|
||||||
# https://github.com/DanBloomberg/leptonica/blob/master/src/colormap.c#L2067
|
|
||||||
palette_pdf_string = compdata.get_palette_pdf_string()
|
|
||||||
palette_data = pikepdf.Object.parse(palette_pdf_string)
|
|
||||||
palette_stream = pikepdf.Stream(pike, bytes(palette_data))
|
|
||||||
palette = [Name.Indexed, Name.DeviceRGB, compdata.ncolors - 1, palette_stream]
|
|
||||||
cs = palette
|
|
||||||
else:
|
|
||||||
# ncolors == 0 means we are using a colorspace without a palette
|
|
||||||
if compdata.spp == 1:
|
|
||||||
cs = Name.DeviceGray
|
|
||||||
elif compdata.spp == 3:
|
|
||||||
cs = Name.DeviceRGB
|
|
||||||
elif compdata.spp == 4:
|
|
||||||
cs = Name.DeviceCMYK
|
|
||||||
im_obj.ColorSpace = cs
|
|
||||||
im_obj.write(compdata.read(), filter=Name.FlateDecode, decode_parms=dparms)
|
|
||||||
|
|
||||||
|
|
||||||
def optimize(input_file: Path, output_file: Path, context, save_settings) -> None:
|
|
||||||
options = context.options
|
options = context.options
|
||||||
if options.optimize == 0:
|
if options.optimize == 0:
|
||||||
safe_symlink(input_file, output_file)
|
safe_symlink(input_file, output_file)
|
||||||
@@ -564,14 +548,14 @@ def optimize(input_file: Path, output_file: Path, context, save_settings) -> Non
|
|||||||
root.mkdir(exist_ok=True)
|
root.mkdir(exist_ok=True)
|
||||||
|
|
||||||
jpegs, pngs = extract_images_generic(pike, root, options)
|
jpegs, pngs = extract_images_generic(pike, root, options)
|
||||||
transcode_jpegs(pike, jpegs, root, options)
|
transcode_jpegs(pike, jpegs, root, options, executor)
|
||||||
# if options.optimize >= 2:
|
# if options.optimize >= 2:
|
||||||
# Try pngifying the jpegs
|
# Try pngifying the jpegs
|
||||||
# transcode_pngs(pike, jpegs, jpg_name, root, options)
|
# transcode_pngs(pike, jpegs, jpg_name, root, options)
|
||||||
transcode_pngs(pike, pngs, png_name, root, options)
|
transcode_pngs(pike, pngs, png_name, root, options, executor)
|
||||||
|
|
||||||
jbig2_groups = extract_images_jbig2(pike, root, options)
|
jbig2_groups = extract_images_jbig2(pike, root, options)
|
||||||
convert_to_jbig2(pike, jbig2_groups, root, options)
|
convert_to_jbig2(pike, jbig2_groups, root, options, executor)
|
||||||
|
|
||||||
target_file = output_file.with_suffix('.opt.pdf')
|
target_file = output_file.with_suffix('.opt.pdf')
|
||||||
pike.remove_unreferenced_resources()
|
pike.remove_unreferenced_resources()
|
||||||
@@ -586,7 +570,7 @@ def optimize(input_file: Path, output_file: Path, context, save_settings) -> Non
|
|||||||
)
|
)
|
||||||
ratio = input_size / output_size
|
ratio = input_size / output_size
|
||||||
savings = 1 - output_size / input_size
|
savings = 1 - output_size / input_size
|
||||||
log.info(f"Optimize ratio: {ratio:.2f} savings: {(100 * savings):.1f}%")
|
log.info(f"Optimize ratio: {ratio:.2f} savings: {(savings):.1%}")
|
||||||
|
|
||||||
if savings < 0:
|
if savings < 0:
|
||||||
log.info("Image optimization did not improve the file - discarded")
|
log.info("Image optimization did not improve the file - discarded")
|
||||||
|
|||||||
+56
-44
@@ -1,19 +1,9 @@
|
|||||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""
|
"""
|
||||||
Utilities for PDF/A production and confirmation with Ghostspcript.
|
Utilities for PDF/A production and confirmation with Ghostspcript.
|
||||||
@@ -21,8 +11,7 @@ Utilities for PDF/A production and confirmation with Ghostspcript.
|
|||||||
|
|
||||||
import base64
|
import base64
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from string import Template
|
from typing import Dict, Iterator, Union
|
||||||
from typing import Dict, Union
|
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
import pkg_resources
|
import pkg_resources
|
||||||
@@ -32,29 +21,57 @@ ICC_PROFILE_RELPATH = 'data/sRGB.icc'
|
|||||||
SRGB_ICC_PROFILE = pkg_resources.resource_filename('ocrmypdf', ICC_PROFILE_RELPATH)
|
SRGB_ICC_PROFILE = pkg_resources.resource_filename('ocrmypdf', ICC_PROFILE_RELPATH)
|
||||||
|
|
||||||
|
|
||||||
# This is a template written in PostScript which is needed to create PDF/A
|
def _postscript_objdef(
|
||||||
# files, from the Ghostscript documentation. Lines beginning with % are
|
alias: str,
|
||||||
# comments. Python substitution variables have a '$' prefix.
|
dictionary: Dict[str, str],
|
||||||
pdfa_def_template = u"""%!
|
*,
|
||||||
% Define an ICC profile :
|
stream_name: str = None,
|
||||||
/ICCProfile $icc_profile
|
stream_data: bytes = None,
|
||||||
def
|
) -> Iterator[str]:
|
||||||
|
assert (stream_name is None) == (stream_data is None)
|
||||||
|
|
||||||
[/_objdef {icc_PDFA} /type /stream /OBJ pdfmark
|
objtype = '/stream' if stream_name else '/dict'
|
||||||
[{icc_PDFA} << /N 3 >> /PUT pdfmark
|
|
||||||
[{icc_PDFA} ICCProfile /PUT pdfmark
|
|
||||||
|
|
||||||
% Define the output intent dictionary :
|
if stream_name:
|
||||||
|
assert stream_data is not None
|
||||||
|
a85_data = base64.a85encode(stream_data, adobe=True).decode('ascii')
|
||||||
|
yield f'{stream_name} ' + a85_data
|
||||||
|
yield 'def'
|
||||||
|
|
||||||
[/_objdef {OutputIntent_PDFA} /type /dict /OBJ pdfmark
|
if alias != '{Catalog}': # Catalog needs no definition
|
||||||
[{OutputIntent_PDFA} <<
|
yield f'[/_objdef {alias} /type {objtype} /OBJ pdfmark'
|
||||||
/Type /OutputIntent % Must be so (the standard requires).
|
|
||||||
/S /GTS_PDFA1 % Must be so (the standard requires).
|
yield f'[{alias} <<'
|
||||||
/DestOutputProfile {icc_PDFA} % Must be so (see above).
|
for key, val in dictionary.items():
|
||||||
/OutputConditionIdentifier ($icc_identifier)
|
yield f' {key} {val}'
|
||||||
>> /PUT pdfmark
|
yield '>> /PUT pdfmark'
|
||||||
[{Catalog} <</OutputIntents [ {OutputIntent_PDFA} ]>> /PUT pdfmark
|
|
||||||
"""
|
if stream_name:
|
||||||
|
yield f'[{alias} {stream_name[1:]} /PUT pdfmark'
|
||||||
|
|
||||||
|
|
||||||
|
def _make_postscript(icc_name: str, icc_data: bytes, colors: int) -> Iterator[str]:
|
||||||
|
yield '%!'
|
||||||
|
yield from _postscript_objdef(
|
||||||
|
'{icc_PDFA}', # Not an f-string
|
||||||
|
{'/N': str(colors)},
|
||||||
|
stream_name='/ICCProfile',
|
||||||
|
stream_data=icc_data,
|
||||||
|
)
|
||||||
|
yield ''
|
||||||
|
yield from _postscript_objdef(
|
||||||
|
'{OutputIntent_PDFA}',
|
||||||
|
{
|
||||||
|
'/Type': '/OutputIntent',
|
||||||
|
'/S': '/GTS_PDFA1',
|
||||||
|
'/DestOutputProfile': '{icc_PDFA}',
|
||||||
|
'/OutputConditionIdentifier': f'({icc_name})', # Only f-string
|
||||||
|
},
|
||||||
|
)
|
||||||
|
yield ''
|
||||||
|
yield from _postscript_objdef(
|
||||||
|
'{Catalog}', {'/OutputIntents': '[ {OutputIntent_PDFA} ]'}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
||||||
@@ -85,13 +102,8 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
|||||||
else:
|
else:
|
||||||
raise NotImplementedError("Only supporting sRGB")
|
raise NotImplementedError("Only supporting sRGB")
|
||||||
|
|
||||||
# Read the ICC profile, encode as ASCII85 and convert to a string which we
|
|
||||||
# will insert in the .ps file
|
|
||||||
bytes_icc_profile = Path(icc_profile).read_bytes()
|
bytes_icc_profile = Path(icc_profile).read_bytes()
|
||||||
icc_profile = base64.a85encode(bytes_icc_profile, adobe=True).decode('ascii')
|
ps = '\n'.join(_make_postscript(icc, bytes_icc_profile, 3))
|
||||||
|
|
||||||
t = Template(pdfa_def_template)
|
|
||||||
ps = t.substitute(icc_profile=icc_profile, icc_identifier=icc)
|
|
||||||
|
|
||||||
# We should have encoded everything to pure ASCII by this point, and
|
# We should have encoded everything to pure ASCII by this point, and
|
||||||
# to be safe, only allow ASCII in PostScript
|
# to be safe, only allow ASCII in PostScript
|
||||||
@@ -100,7 +112,7 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
|||||||
|
|
||||||
|
|
||||||
def file_claims_pdfa(filename: Path):
|
def file_claims_pdfa(filename: Path):
|
||||||
"""Determines if the file claims to be PDF/A compliant
|
"""Determines if the file claims to be PDF/A compliant.
|
||||||
|
|
||||||
This only checks if the XMP metadata contains a PDF/A marker. It does not
|
This only checks if the XMP metadata contains a PDF/A marker. It does not
|
||||||
do full PDF/A validation.
|
do full PDF/A validation.
|
||||||
|
|||||||
@@ -1,19 +1,9 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PdfInfo
|
from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PdfInfo
|
||||||
|
|||||||
+191
-143
@@ -1,38 +1,30 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
|
|
||||||
|
import atexit
|
||||||
import logging
|
import logging
|
||||||
import re
|
import re
|
||||||
from collections import defaultdict, namedtuple
|
from collections import defaultdict, namedtuple
|
||||||
|
from contextlib import ExitStack
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from enum import Enum
|
from enum import Enum
|
||||||
from functools import partial
|
from functools import partial
|
||||||
from math import hypot, isclose
|
from math import hypot, inf, isclose
|
||||||
from os import PathLike
|
from os import PathLike
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Dict, List, Optional, Union
|
from typing import Container, Iterator, Optional, Tuple, Union
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
from pikepdf import PdfMatrix
|
from pikepdf import Object, Pdf, PdfMatrix
|
||||||
|
|
||||||
from ocrmypdf._concurrent import exec_progress_pool
|
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||||
from ocrmypdf.exceptions import EncryptedPdfError
|
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
||||||
from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap
|
from ocrmypdf.helpers import Resolution, available_cpu_count, pikepdf_enable_mmap
|
||||||
from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes
|
from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes
|
||||||
|
|
||||||
@@ -125,7 +117,7 @@ def _normalize_stack(graphobjs):
|
|||||||
yield (operands, operator)
|
yield (operands, operator)
|
||||||
|
|
||||||
|
|
||||||
def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||||
"""Interpret the PDF content stream.
|
"""Interpret the PDF content stream.
|
||||||
|
|
||||||
The stack represents the state of the PDF graphics stack. We are only
|
The stack represents the state of the PDF graphics stack. We are only
|
||||||
@@ -214,7 +206,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _get_dpi(ctm_shorthand, image_size):
|
def _get_dpi(ctm_shorthand, image_size) -> Resolution:
|
||||||
"""Given the transformation matrix and image size, find the image DPI.
|
"""Given the transformation matrix and image size, find the image DPI.
|
||||||
|
|
||||||
PDFs do not include image resolution information within image data.
|
PDFs do not include image resolution information within image data.
|
||||||
@@ -264,25 +256,29 @@ def _get_dpi(ctm_shorthand, image_size):
|
|||||||
a, b, c, d, _, _ = ctm_shorthand
|
a, b, c, d, _, _ = ctm_shorthand
|
||||||
|
|
||||||
# Calculate the width and height of the image in PDF units
|
# Calculate the width and height of the image in PDF units
|
||||||
image_drawn_width = hypot(a, b)
|
image_drawn = hypot(a, b), hypot(c, d)
|
||||||
image_drawn_height = hypot(c, d)
|
|
||||||
|
|
||||||
# The scale of the image is pixels per unit of default user space (1/72")
|
def calc(drawn, pixels, inches_per_pt=72.0):
|
||||||
scale_w = image_size[0] / image_drawn_width
|
# The scale of the image is pixels per unit of default user space (1/72")
|
||||||
scale_h = image_size[1] / image_drawn_height
|
scale = pixels / drawn if drawn != 0 else inf
|
||||||
|
dpi = scale * inches_per_pt
|
||||||
# DPI = scale * 72
|
return dpi
|
||||||
dpi_w = scale_w * 72.0
|
|
||||||
dpi_h = scale_h * 72.0
|
|
||||||
|
|
||||||
|
dpi_w, dpi_h = (calc(image_drawn[n], image_size[n]) for n in range(2))
|
||||||
return Resolution(dpi_w, dpi_h)
|
return Resolution(dpi_w, dpi_h)
|
||||||
|
|
||||||
|
|
||||||
class ImageInfo:
|
class ImageInfo:
|
||||||
DPI_PREC = Decimal('1.000')
|
DPI_PREC = Decimal('1.000')
|
||||||
|
|
||||||
def __init__(self, *, name='', pdfimage=None, inline=None, shorthand=None):
|
def __init__(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
name='',
|
||||||
|
pdfimage: Optional[Object] = None,
|
||||||
|
inline: Optional[Object] = None,
|
||||||
|
shorthand=None,
|
||||||
|
):
|
||||||
self._name = str(name)
|
self._name = str(name)
|
||||||
self._shorthand = shorthand
|
self._shorthand = shorthand
|
||||||
|
|
||||||
@@ -292,6 +288,8 @@ class ImageInfo:
|
|||||||
elif pdfimage is not None:
|
elif pdfimage is not None:
|
||||||
self._origin = 'xobject'
|
self._origin = 'xobject'
|
||||||
pim = pikepdf.PdfImage(pdfimage)
|
pim = pikepdf.PdfImage(pdfimage)
|
||||||
|
else:
|
||||||
|
raise ValueError("Either pdfimage or inline must be set")
|
||||||
self._width = pim.width
|
self._width = pim.width
|
||||||
self._height = pim.height
|
self._height = pim.height
|
||||||
|
|
||||||
@@ -365,6 +363,10 @@ class ImageInfo:
|
|||||||
def enc(self):
|
def enc(self):
|
||||||
return self._enc
|
return self._enc
|
||||||
|
|
||||||
|
@property
|
||||||
|
def renderable(self):
|
||||||
|
return self.dpi.is_finite and self.width >= 0 and self.height >= 0
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def dpi(self):
|
def dpi(self):
|
||||||
return _get_dpi(self._shorthand, (self._width, self._height))
|
return _get_dpi(self._shorthand, (self._width, self._height))
|
||||||
@@ -381,7 +383,7 @@ class ImageInfo:
|
|||||||
).format(**class_locals)
|
).format(**class_locals)
|
||||||
|
|
||||||
|
|
||||||
def _find_inline_images(contentsinfo):
|
def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
||||||
"Find inline images in the contentstream"
|
"Find inline images in the contentstream"
|
||||||
|
|
||||||
for n, inline in enumerate(contentsinfo.inline_images):
|
for n, inline in enumerate(contentsinfo.inline_images):
|
||||||
@@ -390,7 +392,7 @@ def _find_inline_images(contentsinfo):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _image_xobjects(container):
|
def _image_xobjects(container) -> Iterator[Tuple[Object, str]]:
|
||||||
"""Search for all XObject-based images in the container
|
"""Search for all XObject-based images in the container
|
||||||
|
|
||||||
Usually the container is a page, but it could also be a Form XObject
|
Usually the container is a page, but it could also be a Form XObject
|
||||||
@@ -410,7 +412,7 @@ def _image_xobjects(container):
|
|||||||
return
|
return
|
||||||
xobjs = resources['/XObject'].as_dict()
|
xobjs = resources['/XObject'].as_dict()
|
||||||
for xobj in xobjs:
|
for xobj in xobjs:
|
||||||
candidate = xobjs[xobj]
|
candidate: Object = xobjs[xobj]
|
||||||
if not '/Subtype' in candidate:
|
if not '/Subtype' in candidate:
|
||||||
continue
|
continue
|
||||||
if candidate['/Subtype'] == '/Image':
|
if candidate['/Subtype'] == '/Image':
|
||||||
@@ -418,7 +420,9 @@ def _image_xobjects(container):
|
|||||||
yield (pdfimage, xobj)
|
yield (pdfimage, xobj)
|
||||||
|
|
||||||
|
|
||||||
def _find_regular_images(container, contentsinfo):
|
def _find_regular_images(
|
||||||
|
container: Object, contentsinfo: ContentsInfo
|
||||||
|
) -> Iterator[ImageInfo]:
|
||||||
"""Find images stored in the container's /Resources /XObject
|
"""Find images stored in the container's /Resources /XObject
|
||||||
|
|
||||||
Usually the container is a page, but it could also be a Form XObject
|
Usually the container is a page, but it could also be a Form XObject
|
||||||
@@ -442,7 +446,7 @@ def _find_regular_images(container, contentsinfo):
|
|||||||
yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand)
|
yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand)
|
||||||
|
|
||||||
|
|
||||||
def _find_form_xobject_images(pdf, container, contentsinfo):
|
def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo):
|
||||||
"""Find any images that are in Form XObjects in the container
|
"""Find any images that are in Form XObjects in the container
|
||||||
|
|
||||||
The container may be a page, or a parent Form XObject.
|
The container may be a page, or a parent Form XObject.
|
||||||
@@ -474,7 +478,9 @@ def _find_form_xobject_images(pdf, container, contentsinfo):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _process_content_streams(*, pdf, container, shorthand=None):
|
def _process_content_streams(
|
||||||
|
*, pdf: Pdf, container: Object, shorthand=None
|
||||||
|
) -> Iterator[Union[VectorMarker, TextMarker, ImageInfo]]:
|
||||||
"""Find all individual instances of images drawn in the container
|
"""Find all individual instances of images drawn in the container
|
||||||
|
|
||||||
Usually the container is a page, but it may also be a Form XObject.
|
Usually the container is a page, but it may also be a Form XObject.
|
||||||
@@ -536,7 +542,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
|
|||||||
margin_ratio * ph, # bottom (first quadrant: bottom < top)
|
margin_ratio * ph, # bottom (first quadrant: bottom < top)
|
||||||
)
|
)
|
||||||
|
|
||||||
def rects_intersect(a, b):
|
def rects_intersect(a, b) -> bool:
|
||||||
"""
|
"""
|
||||||
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
|
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
|
||||||
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
|
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
|
||||||
@@ -552,7 +558,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
|
|||||||
return has_text
|
return has_text
|
||||||
|
|
||||||
|
|
||||||
def simplify_textboxes(miner, textbox_getter):
|
def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]:
|
||||||
"""Extract only limited content from text boxes
|
"""Extract only limited content from text boxes
|
||||||
|
|
||||||
We do this to save memory and ensure that our objects are pickleable.
|
We do this to save memory and ensure that our objects are pickleable.
|
||||||
@@ -566,97 +572,51 @@ def simplify_textboxes(miner, textbox_getter):
|
|||||||
yield TextboxInfo(box.bbox, visible, corrupt)
|
yield TextboxInfo(box.bbox, visible, corrupt)
|
||||||
|
|
||||||
|
|
||||||
def _pdf_get_pageinfo(
|
|
||||||
pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool
|
|
||||||
):
|
|
||||||
pageinfo: Dict[str, Any] = {}
|
|
||||||
pageinfo['pageno'] = pageno
|
|
||||||
pageinfo['images'] = []
|
|
||||||
|
|
||||||
page = pdf.pages[pageno]
|
|
||||||
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
|
|
||||||
width_pt = mediabox[2] - mediabox[0]
|
|
||||||
height_pt = mediabox[3] - mediabox[1]
|
|
||||||
|
|
||||||
check_this_page = pageno in check_pages
|
|
||||||
|
|
||||||
if check_this_page and detailed_analysis:
|
|
||||||
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
|
|
||||||
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
|
||||||
pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes))
|
|
||||||
bboxes = (box.bbox for box in pageinfo['textboxes'])
|
|
||||||
|
|
||||||
pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt)
|
|
||||||
else:
|
|
||||||
pageinfo['textboxes'] = []
|
|
||||||
pageinfo['has_text'] = None # i.e. "no information"
|
|
||||||
|
|
||||||
userunit = page.get('/UserUnit', Decimal(1.0))
|
|
||||||
if not isinstance(userunit, Decimal):
|
|
||||||
userunit = Decimal(userunit)
|
|
||||||
pageinfo['userunit'] = userunit
|
|
||||||
pageinfo['width_inches'] = width_pt * userunit / Decimal(72.0)
|
|
||||||
pageinfo['height_inches'] = height_pt * userunit / Decimal(72.0)
|
|
||||||
|
|
||||||
try:
|
|
||||||
pageinfo['rotate'] = int(page['/Rotate'])
|
|
||||||
except KeyError:
|
|
||||||
pageinfo['rotate'] = 0
|
|
||||||
|
|
||||||
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
|
|
||||||
|
|
||||||
if check_this_page:
|
|
||||||
pageinfo['has_vector'] = False
|
|
||||||
pageinfo['has_text'] = False
|
|
||||||
pageinfo['images'] = []
|
|
||||||
for ci in _process_content_streams(
|
|
||||||
pdf=pdf, container=page, shorthand=userunit_shorthand
|
|
||||||
):
|
|
||||||
if isinstance(ci, VectorMarker):
|
|
||||||
pageinfo['has_vector'] = True
|
|
||||||
elif isinstance(ci, TextMarker):
|
|
||||||
pageinfo['has_text'] = True
|
|
||||||
elif isinstance(ci, ImageInfo):
|
|
||||||
pageinfo['images'].append(ci)
|
|
||||||
else:
|
|
||||||
raise NotImplementedError()
|
|
||||||
else:
|
|
||||||
pageinfo['has_vector'] = None # i.e. "no information"
|
|
||||||
pageinfo['has_text'] = None
|
|
||||||
pageinfo['images'] = None
|
|
||||||
|
|
||||||
if pageinfo['images']:
|
|
||||||
dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo['images'])
|
|
||||||
pageinfo['dpi'] = dpi
|
|
||||||
pageinfo['width_pixels'] = int(round(dpi.x * float(pageinfo['width_inches'])))
|
|
||||||
pageinfo['height_pixels'] = int(round(dpi.y * float(pageinfo['height_inches'])))
|
|
||||||
|
|
||||||
return pageinfo
|
|
||||||
|
|
||||||
|
|
||||||
worker_pdf = None
|
worker_pdf = None
|
||||||
|
|
||||||
|
|
||||||
def _pdf_pageinfo_sync_init(infile):
|
def _pdf_pageinfo_sync_init(pdf: Pdf, infile: Path, pdfminer_loglevel):
|
||||||
global worker_pdf # pylint: disable=global-statement
|
global worker_pdf # pylint: disable=global-statement
|
||||||
pikepdf_enable_mmap()
|
pikepdf_enable_mmap()
|
||||||
worker_pdf = pikepdf.open(infile)
|
|
||||||
|
logging.getLogger('pdfminer').setLevel(pdfminer_loglevel)
|
||||||
|
|
||||||
|
# If the pdf is not opened, open a copy for our worker process to use
|
||||||
|
if pdf is None:
|
||||||
|
worker_pdf = pikepdf.open(infile)
|
||||||
|
|
||||||
|
def on_process_close():
|
||||||
|
worker_pdf.close()
|
||||||
|
|
||||||
|
# Close when this process exits
|
||||||
|
atexit.register(on_process_close)
|
||||||
|
|
||||||
|
|
||||||
def _pdf_pageinfo_sync(args):
|
def _pdf_pageinfo_sync(args):
|
||||||
global worker_pdf # pylint: disable=global-statement
|
pageno, thread_pdf, infile, check_pages, detailed_analysis = args
|
||||||
pageno, infile, check_pages, detailed_analysis = args
|
pdf = thread_pdf if thread_pdf is not None else worker_pdf
|
||||||
page = PageInfo(worker_pdf, pageno, infile, check_pages, detailed_analysis)
|
with ExitStack() as stack:
|
||||||
return page
|
if not pdf: # When called with SerialExecutor
|
||||||
|
pdf = stack.enter_context(pikepdf.open(infile))
|
||||||
|
page = PageInfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
||||||
|
return page
|
||||||
|
|
||||||
|
|
||||||
def _pdf_pageinfo_concurrent(
|
def _pdf_pageinfo_concurrent(
|
||||||
pdf, infile, progbar, max_workers, check_pages, detailed_analysis=False
|
pdf,
|
||||||
|
executor: Executor,
|
||||||
|
infile,
|
||||||
|
progbar,
|
||||||
|
max_workers,
|
||||||
|
check_pages,
|
||||||
|
detailed_analysis=False,
|
||||||
):
|
):
|
||||||
pages = [None] * len(pdf.pages)
|
pages = [None] * len(pdf.pages)
|
||||||
|
|
||||||
def update_pageinfo(result, pbar):
|
def update_pageinfo(result, pbar):
|
||||||
page = result
|
page = result
|
||||||
|
if not page:
|
||||||
|
raise InputFileError("Could read a page in the PDF")
|
||||||
pages[page.pageno] = page
|
pages[page.pageno] = page
|
||||||
pbar.update()
|
pbar.update()
|
||||||
|
|
||||||
@@ -664,7 +624,6 @@ def _pdf_pageinfo_concurrent(
|
|||||||
max_workers = available_cpu_count()
|
max_workers = available_cpu_count()
|
||||||
|
|
||||||
total = len(pdf.pages)
|
total = len(pdf.pages)
|
||||||
contexts = ((n, infile, check_pages, detailed_analysis) for n in range(total))
|
|
||||||
|
|
||||||
use_threads = False # No performance gain if threaded due to GIL
|
use_threads = False # No performance gain if threaded due to GIL
|
||||||
n_workers = min(1 + len(pages) // 4, max_workers)
|
n_workers = min(1 + len(pages) // 4, max_workers)
|
||||||
@@ -673,13 +632,27 @@ def _pdf_pageinfo_concurrent(
|
|||||||
# a separate process.
|
# a separate process.
|
||||||
use_threads = True
|
use_threads = True
|
||||||
|
|
||||||
exec_progress_pool(
|
# If we use a thread, we can pass the already-open Pdf for them to use
|
||||||
|
# If we use processes, we pass a None which tells the init function to open its
|
||||||
|
# own
|
||||||
|
initial_pdf = pdf if use_threads else None
|
||||||
|
|
||||||
|
contexts = (
|
||||||
|
(n, initial_pdf, infile, check_pages, detailed_analysis) for n in range(total)
|
||||||
|
)
|
||||||
|
assert n_workers == 1 if use_threads else n_workers >= 1, "Not multithreadable"
|
||||||
|
executor(
|
||||||
use_threads=use_threads,
|
use_threads=use_threads,
|
||||||
max_workers=n_workers,
|
max_workers=n_workers,
|
||||||
tqdm_kwargs=dict(
|
tqdm_kwargs=dict(
|
||||||
total=total, desc="Scanning contents", unit='page', disable=not progbar
|
total=total, desc="Scanning contents", unit='page', disable=not progbar
|
||||||
),
|
),
|
||||||
task_initializer=partial(_pdf_pageinfo_sync_init, infile),
|
worker_initializer=partial(
|
||||||
|
_pdf_pageinfo_sync_init,
|
||||||
|
initial_pdf,
|
||||||
|
infile,
|
||||||
|
logging.getLogger('pdfminer').level,
|
||||||
|
),
|
||||||
task=_pdf_pageinfo_sync,
|
task=_pdf_pageinfo_sync,
|
||||||
task_arguments=contexts,
|
task_arguments=contexts,
|
||||||
task_finished=update_pageinfo,
|
task_finished=update_pageinfo,
|
||||||
@@ -688,13 +661,87 @@ def _pdf_pageinfo_concurrent(
|
|||||||
|
|
||||||
|
|
||||||
class PageInfo:
|
class PageInfo:
|
||||||
def __init__(self, pdf, pageno, infile, check_pages, detailed_analysis=False):
|
def __init__(
|
||||||
|
self,
|
||||||
|
pdf: Pdf,
|
||||||
|
pageno: int,
|
||||||
|
infile: PathLike,
|
||||||
|
check_pages: Container[int],
|
||||||
|
detailed_analysis: bool = False,
|
||||||
|
):
|
||||||
self._pageno = pageno
|
self._pageno = pageno
|
||||||
self._infile = infile
|
self._infile = infile
|
||||||
self._detailed_analysis = detailed_analysis
|
self._detailed_analysis = detailed_analysis
|
||||||
self._pageinfo = _pdf_get_pageinfo(
|
self._gather_pageinfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
||||||
pdf, pageno, infile, check_pages, detailed_analysis
|
|
||||||
)
|
def _gather_pageinfo(
|
||||||
|
self,
|
||||||
|
pdf: Pdf,
|
||||||
|
pageno: int,
|
||||||
|
infile: PathLike,
|
||||||
|
check_pages: Container[int],
|
||||||
|
detailed_analysis: bool,
|
||||||
|
):
|
||||||
|
page = pdf.pages[pageno]
|
||||||
|
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
|
||||||
|
width_pt = mediabox[2] - mediabox[0]
|
||||||
|
height_pt = mediabox[3] - mediabox[1]
|
||||||
|
|
||||||
|
check_this_page = pageno in check_pages
|
||||||
|
|
||||||
|
if check_this_page and detailed_analysis:
|
||||||
|
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
|
||||||
|
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
||||||
|
self._textboxes = list(simplify_textboxes(miner, get_text_boxes))
|
||||||
|
bboxes = (box.bbox for box in self._textboxes)
|
||||||
|
|
||||||
|
self._has_text = _page_has_text(bboxes, width_pt, height_pt)
|
||||||
|
else:
|
||||||
|
self._textboxes = []
|
||||||
|
self._has_text = None # i.e. "no information"
|
||||||
|
|
||||||
|
userunit = page.get('/UserUnit', Decimal(1.0))
|
||||||
|
if not isinstance(userunit, Decimal):
|
||||||
|
userunit = Decimal(userunit)
|
||||||
|
self._userunit = userunit
|
||||||
|
self._width_inches = width_pt * userunit / Decimal(72.0)
|
||||||
|
self._height_inches = height_pt * userunit / Decimal(72.0)
|
||||||
|
|
||||||
|
try:
|
||||||
|
self._rotate = int(page['/Rotate'])
|
||||||
|
except KeyError:
|
||||||
|
self._rotate = 0
|
||||||
|
|
||||||
|
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
|
||||||
|
|
||||||
|
if check_this_page:
|
||||||
|
self._has_vector = False
|
||||||
|
self._has_text = False
|
||||||
|
self._images = []
|
||||||
|
for ci in _process_content_streams(
|
||||||
|
pdf=pdf, container=page, shorthand=userunit_shorthand
|
||||||
|
):
|
||||||
|
if isinstance(ci, VectorMarker):
|
||||||
|
self._has_vector = True
|
||||||
|
elif isinstance(ci, TextMarker):
|
||||||
|
self._has_text = True
|
||||||
|
elif isinstance(ci, ImageInfo):
|
||||||
|
self._images.append(ci)
|
||||||
|
else:
|
||||||
|
raise NotImplementedError()
|
||||||
|
else:
|
||||||
|
self._has_vector = None # i.e. "no information"
|
||||||
|
self._has_text = None
|
||||||
|
self._images = None
|
||||||
|
|
||||||
|
self._dpi = None
|
||||||
|
if self._images:
|
||||||
|
dpi = Resolution(0.0, 0.0).take_max(
|
||||||
|
image.dpi for image in self._images if image.renderable
|
||||||
|
)
|
||||||
|
self._dpi = dpi
|
||||||
|
self._width_pixels = int(round(dpi.x * float(self._width_inches)))
|
||||||
|
self._height_pixels = int(round(dpi.y * float(self._height_inches)))
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def pageno(self) -> int:
|
def pageno(self) -> int:
|
||||||
@@ -702,25 +749,25 @@ class PageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def has_text(self) -> bool:
|
def has_text(self) -> bool:
|
||||||
return self._pageinfo['has_text']
|
return self._has_text
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_corrupt_text(self) -> bool:
|
def has_corrupt_text(self) -> bool:
|
||||||
if not self._detailed_analysis:
|
if not self._detailed_analysis:
|
||||||
raise NotImplementedError('Did not do detailed analysis')
|
raise NotImplementedError('Did not do detailed analysis')
|
||||||
return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes'])
|
return any(tbox.is_corrupt for tbox in self._textboxes)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_vector(self) -> bool:
|
def has_vector(self) -> bool:
|
||||||
return self._pageinfo['has_vector']
|
return self._has_vector
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def width_inches(self) -> Decimal:
|
def width_inches(self) -> Decimal:
|
||||||
return self._pageinfo['width_inches']
|
return self._width_inches
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def height_inches(self) -> Decimal:
|
def height_inches(self) -> Decimal:
|
||||||
return self._pageinfo['height_inches']
|
return self._height_inches
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def width_pixels(self) -> int:
|
def width_pixels(self) -> int:
|
||||||
@@ -732,18 +779,18 @@ class PageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def rotation(self) -> int:
|
def rotation(self) -> int:
|
||||||
return self._pageinfo.get('rotate', None)
|
return self._rotate
|
||||||
|
|
||||||
@rotation.setter
|
@rotation.setter
|
||||||
def rotation(self, value):
|
def rotation(self, value):
|
||||||
if value in (0, 90, 180, 270, 360, -90, -180, -270):
|
if value in (0, 90, 180, 270, 360, -90, -180, -270):
|
||||||
self._pageinfo['rotate'] = value
|
self._rotate = value
|
||||||
else:
|
else:
|
||||||
raise ValueError("rotation must be a cardinal angle")
|
raise ValueError("rotation must be a cardinal angle")
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def images(self):
|
def images(self):
|
||||||
return self._pageinfo['images']
|
return self._images
|
||||||
|
|
||||||
def get_textareas(
|
def get_textareas(
|
||||||
self, visible: Optional[bool] = None, corrupt: Optional[bool] = None
|
self, visible: Optional[bool] = None, corrupt: Optional[bool] = None
|
||||||
@@ -758,24 +805,22 @@ class PageInfo:
|
|||||||
result = False
|
result = False
|
||||||
return result
|
return result
|
||||||
|
|
||||||
if 'textboxes' not in self._pageinfo:
|
if not self._textboxes:
|
||||||
if visible is not None and corrupt is not None:
|
if visible is not None and corrupt is not None:
|
||||||
raise NotImplementedError('Incomplete information on textboxes')
|
raise NotImplementedError('Incomplete information on textboxes')
|
||||||
return self._pageinfo['bboxes']
|
return self._textboxes
|
||||||
|
|
||||||
return (
|
return (obj.bbox for obj in self._textboxes if predicate(obj, visible, corrupt))
|
||||||
obj.bbox
|
|
||||||
for obj in self._pageinfo['textboxes']
|
|
||||||
if predicate(obj, visible, corrupt)
|
|
||||||
)
|
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def dpi(self) -> Resolution:
|
def dpi(self) -> Resolution:
|
||||||
return self._pageinfo.get('dpi', Resolution(0.0, 0.0))
|
if self._dpi is None:
|
||||||
|
return Resolution(0.0, 0.0)
|
||||||
|
return self._dpi
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def userunit(self) -> Decimal:
|
def userunit(self) -> Decimal:
|
||||||
return self._pageinfo.get('userunit', None)
|
return self._userunit
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def min_version(self) -> str:
|
def min_version(self) -> str:
|
||||||
@@ -798,10 +843,12 @@ class PdfInfo:
|
|||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
infile,
|
infile,
|
||||||
|
*,
|
||||||
detailed_analysis: bool = False,
|
detailed_analysis: bool = False,
|
||||||
progbar: bool = False,
|
progbar: bool = False,
|
||||||
max_workers: int = None,
|
max_workers: int = None,
|
||||||
check_pages=None,
|
check_pages=None,
|
||||||
|
executor: Executor = SerialExecutor(),
|
||||||
):
|
):
|
||||||
self._infile = infile
|
self._infile = infile
|
||||||
if check_pages is None:
|
if check_pages is None:
|
||||||
@@ -812,18 +859,19 @@ class PdfInfo:
|
|||||||
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
||||||
self._pages = _pdf_pageinfo_concurrent(
|
self._pages = _pdf_pageinfo_concurrent(
|
||||||
pdf,
|
pdf,
|
||||||
|
executor,
|
||||||
infile,
|
infile,
|
||||||
progbar,
|
progbar,
|
||||||
max_workers,
|
max_workers,
|
||||||
check_pages=check_pages,
|
check_pages=check_pages,
|
||||||
detailed_analysis=detailed_analysis,
|
detailed_analysis=detailed_analysis,
|
||||||
)
|
)
|
||||||
self._needs_rendering = pdf.root.get('/NeedsRendering', False)
|
self._needs_rendering = pdf.Root.get('/NeedsRendering', False)
|
||||||
self._has_acroform = False
|
self._has_acroform = False
|
||||||
if '/AcroForm' in pdf.root:
|
if '/AcroForm' in pdf.Root:
|
||||||
if len(pdf.root.AcroForm.get('/Fields', [])) > 0:
|
if len(pdf.Root.AcroForm.get('/Fields', [])) > 0:
|
||||||
self._has_acroform = True
|
self._has_acroform = True
|
||||||
elif '/XFA' in pdf.root.AcroForm:
|
elif '/XFA' in pdf.Root.AcroForm:
|
||||||
self._has_acroform = True
|
self._has_acroform = True
|
||||||
|
|
||||||
@property
|
@property
|
||||||
|
|||||||
@@ -1,19 +1,9 @@
|
|||||||
# © 2018 James R. Barlow: github.com/jbarlow83
|
# © 2018 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import re
|
import re
|
||||||
from math import copysign
|
from math import copysign
|
||||||
@@ -31,7 +21,7 @@ from pdfminer.pdffont import PDFSimpleFont, PDFUnicodeNotDefined
|
|||||||
from pdfminer.pdfpage import PDFPage
|
from pdfminer.pdfpage import PDFPage
|
||||||
from pdfminer.utils import bbox2str, matrix2str
|
from pdfminer.utils import bbox2str, matrix2str
|
||||||
|
|
||||||
from ocrmypdf.exceptions import EncryptedPdfError
|
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
||||||
|
|
||||||
STRIP_NAME = re.compile(r'[0-9]+')
|
STRIP_NAME = re.compile(r'[0-9]+')
|
||||||
|
|
||||||
@@ -233,6 +223,7 @@ def get_page_analysis(infile, pageno, pscript5_mode):
|
|||||||
)
|
)
|
||||||
interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev)
|
interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev)
|
||||||
|
|
||||||
|
patcher = None
|
||||||
if pscript5_mode:
|
if pscript5_mode:
|
||||||
patcher = patch.multiple(
|
patcher = patch.multiple(
|
||||||
'pdfminer.pdffont.PDFType3Font',
|
'pdfminer.pdffont.PDFType3Font',
|
||||||
@@ -245,12 +236,17 @@ def get_page_analysis(infile, pageno, pscript5_mode):
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
with Path(infile).open('rb') as f:
|
with Path(infile).open('rb') as f:
|
||||||
page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0)
|
page_iter = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0)
|
||||||
interp.process_page(next(page))
|
page = next(page_iter, None)
|
||||||
except PDFTextExtractionNotAllowed:
|
if page is None:
|
||||||
raise EncryptedPdfError()
|
raise InputFileError(
|
||||||
|
f"pdfminer could not process page {pageno} (counting from 0)."
|
||||||
|
)
|
||||||
|
interp.process_page(page)
|
||||||
|
except PDFTextExtractionNotAllowed as e:
|
||||||
|
raise EncryptedPdfError() from e
|
||||||
finally:
|
finally:
|
||||||
if pscript5_mode:
|
if patcher is not None:
|
||||||
patcher.stop()
|
patcher.stop()
|
||||||
|
|
||||||
return dev.get_result()
|
return dev.get_result()
|
||||||
|
|||||||
+164
-22
@@ -1,41 +1,48 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
from abc import ABC, abstractmethod, abstractstaticmethod
|
from abc import ABC, abstractmethod, abstractstaticmethod
|
||||||
from argparse import ArgumentParser, Namespace
|
from argparse import ArgumentParser, Namespace
|
||||||
from collections import namedtuple
|
from collections import namedtuple
|
||||||
|
from logging import Handler
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import TYPE_CHECKING, AbstractSet, List, Optional
|
from typing import TYPE_CHECKING, AbstractSet, List, Optional
|
||||||
|
|
||||||
import pluggy
|
import pluggy
|
||||||
|
|
||||||
|
from ocrmypdf._concurrent import Executor
|
||||||
from ocrmypdf.helpers import Resolution
|
from ocrmypdf.helpers import Resolution
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
|
# pylint: disable=ungrouped-imports
|
||||||
from ocrmypdf._jobcontext import PageContext
|
from ocrmypdf._jobcontext import PageContext
|
||||||
from ocrmypdf.pdfinfo import PdfInfo
|
from ocrmypdf.pdfinfo import PdfInfo
|
||||||
|
|
||||||
|
# pylint: enable=ungrouped-imports
|
||||||
|
|
||||||
hookspec = pluggy.HookspecMarker('ocrmypdf')
|
hookspec = pluggy.HookspecMarker('ocrmypdf')
|
||||||
|
|
||||||
# pylint: disable=unused-argument
|
# pylint: disable=unused-argument
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def get_logging_console() -> Handler:
|
||||||
|
"""Returns a custom logging handler.
|
||||||
|
|
||||||
|
Generally this is necessary when both logging output and a progress bar are both
|
||||||
|
outputting to ``sys.stderr``.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
@hookspec
|
@hookspec
|
||||||
def add_options(parser: ArgumentParser) -> None:
|
def add_options(parser: ArgumentParser) -> None:
|
||||||
"""Allows the plugin to add its own command line and API arguments.
|
"""Allows the plugin to add its own command line and API arguments.
|
||||||
@@ -69,7 +76,68 @@ def check_options(options: Namespace) -> None:
|
|||||||
Note:
|
Note:
|
||||||
This hook will be called from the main process, and may modify global state
|
This hook will be called from the main process, and may modify global state
|
||||||
before child worker processes are forked.
|
before child worker processes are forked.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def get_executor(progressbar_class) -> Executor:
|
||||||
|
"""Called to obtain an object that manages parallel execution.
|
||||||
|
|
||||||
|
This may be used to replace OCRmyPDF's default parallel execution system
|
||||||
|
with a third party alternative. For example, you could make OCRmyPDF run in a
|
||||||
|
distributed environment.
|
||||||
|
|
||||||
|
OCRmyPDF's executors are analogous to the standard Python executors in
|
||||||
|
``conconcurrent.futures``, but they do not work the same way. Executors may
|
||||||
|
be reused for different, unrelated batch operations, since all of the context
|
||||||
|
for a given job are passed to :meth:`Executor.__call__`.
|
||||||
|
|
||||||
|
Should be of type :class:`Executor` or otherwise conforming to the protocol
|
||||||
|
of that call.
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
progressbar_class: A progress bar class, which will be created when
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This hook will be called from the main process, and may modify global state
|
||||||
|
before child worker processes are forked.
|
||||||
|
Note:
|
||||||
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def get_progressbar_class():
|
||||||
|
"""Called to obtain a class that can be used to monitor progress.
|
||||||
|
|
||||||
|
A progress bar is assumed, but this could be used for any type of monitoring.
|
||||||
|
|
||||||
|
The class should follow a tqdm-like protocol. Calling the class should return
|
||||||
|
a new progress bar object, which is activated with ``__enter__`` and terminated
|
||||||
|
``__exit__``. An update method is called whenever the progress bar is updated.
|
||||||
|
Progress bar objects will not be reused; a new one will be created for each
|
||||||
|
group of tasks.
|
||||||
|
|
||||||
|
The progress bar is held in the main process/thread and not updated by child
|
||||||
|
process/threads. When a child notifies the parent of completed work, the
|
||||||
|
parent updates the progress bar.
|
||||||
|
|
||||||
|
The arguments are the same as `tqdm <https://github.com/tqdm/tqdm>`_ accepts.
|
||||||
|
|
||||||
|
Progress bars should never write to ``sys.stdout``, or they will corrupt the
|
||||||
|
output if OCRmyPDF writes a PDF to standard output.
|
||||||
|
|
||||||
|
The type of events that OCRmyPDF reports to a progress bar may change in
|
||||||
|
minor releases.
|
||||||
|
|
||||||
|
Here is how OCRmyPDF will use the progress bar:
|
||||||
|
|
||||||
|
Example:
|
||||||
|
pbar_class = pm.hook.get_progressbar_class()
|
||||||
|
with pbar_class(**tqdm_kwargs) as pbar:
|
||||||
|
...
|
||||||
|
pbar.update(1)
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
@hookspec
|
@hookspec
|
||||||
@@ -99,9 +167,9 @@ def rasterize_pdf_page(
|
|||||||
raster_device: str,
|
raster_device: str,
|
||||||
raster_dpi: Resolution,
|
raster_dpi: Resolution,
|
||||||
pageno: int,
|
pageno: int,
|
||||||
page_dpi: Optional[Resolution] = None,
|
page_dpi: Optional[Resolution],
|
||||||
rotation: Optional[int] = None,
|
rotation: Optional[int],
|
||||||
filter_vector: bool = False,
|
filter_vector: bool,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units.
|
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units.
|
||||||
|
|
||||||
@@ -164,10 +232,72 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path:
|
|||||||
produced for a given page, this function will not be called. This is not
|
produced for a given page, this function will not be called. This is not
|
||||||
the image that will be shown to OCR.
|
the image that will be shown to OCR.
|
||||||
|
|
||||||
ocrmypdf will create the PDF page based on the image format used. If you
|
If the function does not want to modify the image, it should return
|
||||||
convert the image to a JPEG, the output page will be created as a JPEG, etc.
|
``image_filename``. The hook may overwrite ``image_filename`` with a new file.
|
||||||
Note that the ocrmypdf image optimization stage may ultimately chose a
|
|
||||||
different format.
|
The output image should preserve the same physical unit dimensions, that is
|
||||||
|
(width * dpi_x, height * dpi_y). That is, if the image is resized, the DPI
|
||||||
|
must be adjusted by the reciprocal. If this is not preserved, the PDF page
|
||||||
|
will be resized and the OCR layer misaligned. OCRmyPDF does not nothing
|
||||||
|
to enforce these constraints; it is up to the plugin to do sensible things.
|
||||||
|
|
||||||
|
OCRmyPDF will create the PDF page based on the image format used (unless the
|
||||||
|
hook is overriden). If you convert the image to a JPEG, the output page will
|
||||||
|
be created as a JPEG, etc. If you change the colorspace, that change will be
|
||||||
|
kept. Note that the OCRmyPDF image optimization stage, if enabled, may
|
||||||
|
ultimately chose a different format.
|
||||||
|
|
||||||
|
If the return value is a file that does not exist, ``FileNotFoundError``
|
||||||
|
will occur. The return value should be a path to a file in the same folder
|
||||||
|
as ``image_filename``.
|
||||||
|
|
||||||
|
Implementation detail: If the value returned is falsy, OCRmyPDF will ignore
|
||||||
|
the return value and assume the input file was unmodified. This is deprecated.
|
||||||
|
To leave the image unmodified, ``image_filename`` should be returned.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This hook will be called from child processes. Modifying global state
|
||||||
|
will not affect the main process or other child processes.
|
||||||
|
Note:
|
||||||
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def filter_pdf_page(
|
||||||
|
page: 'PageContext', image_filename: Path, output_pdf: Path
|
||||||
|
) -> Path:
|
||||||
|
"""Called to convert a filtered whole page image into a PDF.
|
||||||
|
|
||||||
|
A whole page image is only produced when preprocessing command line arguments
|
||||||
|
are issued or when ``--force-ocr`` is issued. If no whole page is image is
|
||||||
|
produced for a given page, this function will not be called. This is not
|
||||||
|
the image that will be shown to OCR. The whole page image is filtered in
|
||||||
|
the hook above, ``filter_page_image``, then this function is called for
|
||||||
|
PDF conversion.
|
||||||
|
|
||||||
|
This function will only be called when OCRmyPDF runs in a mode such as
|
||||||
|
"force OCR" mode where rasterizing of all content is performed.
|
||||||
|
|
||||||
|
Clever things could be done at this stage such as segmenting the page image into
|
||||||
|
color regions or vector equivalents.
|
||||||
|
|
||||||
|
The provider of the hook implementation is responsible for ensuring that the
|
||||||
|
OCR text layer is aligned with the PDF produced here, or text misalignment
|
||||||
|
will result.
|
||||||
|
|
||||||
|
Currently this function must produce a single page PDF or the pipeline will
|
||||||
|
fail. If the intent is to remove the PDF, then create a single page empty
|
||||||
|
PDF.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
page: Context for this page.
|
||||||
|
image_filename: Filename of the input image used to create output_pdf,
|
||||||
|
for "reference" if recreating the output_pdf entirely.
|
||||||
|
output_pdf: The previous created output_pdf.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
output_pdf
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This hook will be called from child processes. Modifying global state
|
This hook will be called from child processes. Modifying global state
|
||||||
@@ -272,6 +402,7 @@ def generate_pdfa(
|
|||||||
compression: str,
|
compression: str,
|
||||||
pdf_version: str,
|
pdf_version: str,
|
||||||
pdfa_part: str,
|
pdfa_part: str,
|
||||||
|
progressbar_class,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
"""Generate a PDF/A.
|
"""Generate a PDF/A.
|
||||||
|
|
||||||
@@ -294,10 +425,21 @@ def generate_pdfa(
|
|||||||
At its own discretion, the PDF/A generator may raise the version,
|
At its own discretion, the PDF/A generator may raise the version,
|
||||||
but should not lower it.
|
but should not lower it.
|
||||||
pdfa_part: The desired PDF/A compliance level, such as ``'2B'``.
|
pdfa_part: The desired PDF/A compliance level, such as ``'2B'``.
|
||||||
|
progressbar_class: The class of a progress bar with a tqdm-like API. An
|
||||||
|
instance of this class will be initialized when PDF/A conversion
|
||||||
|
begins, using
|
||||||
|
``instance = progressbar_class(total: int, desc: str, unit:str)``,
|
||||||
|
defining the number of work units, a user-visible description,
|
||||||
|
and the name of the work units ("page"). Then ``instance.update()``
|
||||||
|
will be called when a work unit is completed. If ``None``, no
|
||||||
|
progress information is reported.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Path: If successful, the hook should return ``output_file``.
|
Path: If successful, the hook should return ``output_file``.
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This is a :ref:`firstresult hook<firstresult>`.
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
|
|
||||||
|
See also:
|
||||||
|
https://github.com/tqdm/tqdm
|
||||||
"""
|
"""
|
||||||
|
|||||||
@@ -0,0 +1 @@
|
|||||||
|
# ocrmypdf is typed
|
||||||
+4
-14
@@ -1,19 +1,9 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Utilities to measure OCR quality"""
|
"""Utilities to measure OCR quality"""
|
||||||
|
|
||||||
|
|||||||
@@ -1,48 +1,103 @@
|
|||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
#
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Wrappers to manage subprocess calls"""
|
"""Wrappers to manage subprocess calls"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import shutil
|
|
||||||
import sys
|
import sys
|
||||||
from collections.abc import Mapping
|
from collections.abc import Mapping
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from distutils.version import LooseVersion
|
from distutils.version import LooseVersion, Version
|
||||||
from functools import lru_cache
|
from functools import lru_cache
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||||
from subprocess import run as subprocess_run
|
from subprocess import run as subprocess_run
|
||||||
|
from typing import Callable, Optional, Type, Union
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
|
|
||||||
|
# pylint: disable=logging-format-interpolation
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def run(args, *, env=None, **kwargs):
|
def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs):
|
||||||
"""Wrapper around :py:func:`subprocess.run`
|
"""Wrapper around :py:func:`subprocess.run`
|
||||||
|
|
||||||
The main purpose of this wrapper is to log subprocess output in an orderly
|
The main purpose of this wrapper is to log subprocess output in an orderly
|
||||||
fashion that indentifies the responsible subprocess. An additional
|
fashion that indentifies the responsible subprocess. An additional
|
||||||
task is that this function goes to greater lengths to find possible Windows
|
task is that this function goes to greater lengths to find possible Windows
|
||||||
locations of our dependencies when they are not on the system PATH.
|
locations of our dependencies when they are not on the system PATH.
|
||||||
|
|
||||||
|
Arguments should be identical to ``subprocess.run``, except for following:
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
logs_errors_to_stdout: If True, indicates that the process writes its error
|
||||||
|
messages to stdout rather than stderr, so stdout should be logged
|
||||||
|
if there is an error. If False, stderr is logged. Could be used with
|
||||||
|
stderr=STDOUT, stdout=PIPE for example.
|
||||||
"""
|
"""
|
||||||
|
args, env, process_log, _text = _fix_process_args(args, env, kwargs)
|
||||||
|
|
||||||
|
stderr = None
|
||||||
|
stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout'
|
||||||
|
try:
|
||||||
|
proc = subprocess_run(args, env=env, **kwargs)
|
||||||
|
except CalledProcessError as e:
|
||||||
|
stderr = getattr(e, stderr_name, None)
|
||||||
|
raise
|
||||||
|
else:
|
||||||
|
stderr = getattr(proc, stderr_name, None)
|
||||||
|
finally:
|
||||||
|
if process_log.isEnabledFor(logging.DEBUG) and stderr:
|
||||||
|
with suppress(AttributeError, UnicodeDecodeError):
|
||||||
|
stderr = stderr.decode('utf-8', 'replace')
|
||||||
|
if logs_errors_to_stdout:
|
||||||
|
process_log.debug("stdout/stderr = %s", stderr)
|
||||||
|
else:
|
||||||
|
process_log.debug("stderr = %s", stderr)
|
||||||
|
return proc
|
||||||
|
|
||||||
|
|
||||||
|
def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs):
|
||||||
|
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
||||||
|
|
||||||
|
Every line of produced by stderr will be forwarded to the callback function.
|
||||||
|
The intended use is monitoring progress of subprocesses that output their
|
||||||
|
own progress indicators. In addition, each line will be logged if debug
|
||||||
|
logging is enabled.
|
||||||
|
|
||||||
|
Requires stderr to be opened in text mode for ease of handling errors. In
|
||||||
|
addition the expected encoding= and errors= arguments should be set. Note
|
||||||
|
that if stdout is already set up, it need not be binary.
|
||||||
|
"""
|
||||||
|
args, env, process_log, text = _fix_process_args(args, env, kwargs)
|
||||||
|
assert text, "Must use text=True"
|
||||||
|
|
||||||
|
with Popen(args, env=env, **kwargs) as proc:
|
||||||
|
lines = []
|
||||||
|
while proc.poll() is None:
|
||||||
|
for msg in iter(proc.stderr.readline, ''):
|
||||||
|
if process_log.isEnabledFor(logging.DEBUG):
|
||||||
|
process_log.debug(msg.strip())
|
||||||
|
callback(msg)
|
||||||
|
lines.append(msg)
|
||||||
|
stderr = ''.join(lines)
|
||||||
|
|
||||||
|
if check and proc.returncode != 0:
|
||||||
|
raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr)
|
||||||
|
return CompletedProcess(args, proc.returncode, None, stderr=stderr)
|
||||||
|
|
||||||
|
|
||||||
|
def _fix_process_args(args, env, kwargs):
|
||||||
|
assert 'universal_newlines' not in kwargs, "Use text= instead of universal_newlines"
|
||||||
|
|
||||||
if not env:
|
if not env:
|
||||||
env = os.environ
|
env = os.environ
|
||||||
|
|
||||||
@@ -50,53 +105,23 @@ def run(args, *, env=None, **kwargs):
|
|||||||
program = args[0]
|
program = args[0]
|
||||||
|
|
||||||
if os.name == 'nt':
|
if os.name == 'nt':
|
||||||
args = _fix_windows_args(program, args, env)
|
from ocrmypdf.subprocess._windows import fix_windows_args
|
||||||
|
|
||||||
|
args = fix_windows_args(program, args, env)
|
||||||
|
|
||||||
log.debug("Running: %s", args)
|
log.debug("Running: %s", args)
|
||||||
process_log = log.getChild('subprocess.' + os.path.basename(program))
|
process_log = log.getChild(os.path.basename(program))
|
||||||
if sys.version_info < (3, 7) and os.name == 'nt':
|
text = kwargs.get('text', False)
|
||||||
# Can't use close_fds=True on Windows with Python 3.6 or older
|
if sys.version_info < (3, 7):
|
||||||
# https://bugs.python.org/issue19575, etc.
|
if os.name == 'nt':
|
||||||
kwargs['close_fds'] = False
|
# Can't use close_fds=True on Windows with Python 3.6 or older
|
||||||
|
# https://bugs.python.org/issue19575, etc.
|
||||||
stderr = None
|
kwargs['close_fds'] = False
|
||||||
try:
|
if 'text' in kwargs:
|
||||||
proc = subprocess_run(args, env=env, **kwargs)
|
# Convert run(...text=) to run(...universal_newlines=) for Python 3.6
|
||||||
except CalledProcessError as e:
|
kwargs['universal_newlines'] = kwargs['text']
|
||||||
stderr = getattr(e, 'stderr', None)
|
del kwargs['text']
|
||||||
raise
|
return args, env, process_log, text
|
||||||
else:
|
|
||||||
stderr = getattr(proc, 'stderr', None)
|
|
||||||
finally:
|
|
||||||
if process_log.isEnabledFor(logging.DEBUG) and stderr:
|
|
||||||
with suppress(AttributeError, UnicodeDecodeError):
|
|
||||||
stderr = stderr.decode('utf-8', 'replace')
|
|
||||||
process_log.debug("stderr = %s", stderr)
|
|
||||||
return proc
|
|
||||||
|
|
||||||
|
|
||||||
def _fix_windows_args(program, args, env):
|
|
||||||
"""Adjust our desired program and command line arguments for use on Windows"""
|
|
||||||
|
|
||||||
if sys.version_info < (3, 8):
|
|
||||||
# bpo-33617 - Windows needs manual Path -> str conversion
|
|
||||||
args = [os.fspath(arg) for arg in args]
|
|
||||||
program = os.fspath(program)
|
|
||||||
|
|
||||||
# If we are running a .py on Windows, ensure we call it with this Python
|
|
||||||
# (to support test suite shims)
|
|
||||||
if program.lower().endswith('.py'):
|
|
||||||
args = [sys.executable] + args
|
|
||||||
|
|
||||||
paths = os.pathsep.join(os.get_exec_path(env))
|
|
||||||
if not shutil.which(args[0], path=paths):
|
|
||||||
# If the program we want is not on the PATH, add some interesting
|
|
||||||
# locations in %PROGRAMFILES% to the PATH and try again
|
|
||||||
shimmed_path = shim_paths_with_program_files(env)
|
|
||||||
new_args0 = shutil.which(args[0], path=shimmed_path)
|
|
||||||
if new_args0:
|
|
||||||
args[0] = new_args0
|
|
||||||
return args
|
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=None)
|
@lru_cache(maxsize=None)
|
||||||
@@ -117,7 +142,7 @@ def get_version(
|
|||||||
proc = run(
|
proc = run(
|
||||||
args_prog,
|
args_prog,
|
||||||
close_fds=True,
|
close_fds=True,
|
||||||
universal_newlines=True,
|
text=True,
|
||||||
stdout=PIPE,
|
stdout=PIPE,
|
||||||
stderr=STDOUT,
|
stderr=STDOUT,
|
||||||
check=True,
|
check=True,
|
||||||
@@ -136,44 +161,18 @@ def get_version(
|
|||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
f"Could not find program '{program}' on the PATH"
|
f"Could not find program '{program}' on the PATH"
|
||||||
) from e
|
) from e
|
||||||
try:
|
|
||||||
version = re.match(regex, output.strip()).group(1)
|
match = re.match(regex, output.strip())
|
||||||
except AttributeError as e:
|
if not match:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
f"The program '{program}' did not report its version. "
|
f"The program '{program}' did not report its version. "
|
||||||
f"Message was:\n{output}"
|
f"Message was:\n{output}"
|
||||||
)
|
)
|
||||||
|
version = match.group(1)
|
||||||
|
|
||||||
return version
|
return version
|
||||||
|
|
||||||
|
|
||||||
def shim_paths_with_program_files(env=None):
|
|
||||||
if not env:
|
|
||||||
env = os.environ
|
|
||||||
program_files = env.get('PROGRAMFILES', '')
|
|
||||||
if not program_files:
|
|
||||||
return env.get('PATH', '')
|
|
||||||
|
|
||||||
def path_walker():
|
|
||||||
for path in Path(program_files).iterdir():
|
|
||||||
if not path.is_dir():
|
|
||||||
continue
|
|
||||||
if path.name.lower() == 'tesseract-ocr':
|
|
||||||
yield path
|
|
||||||
elif path.name.lower() == 'gs':
|
|
||||||
yield from (p for p in path.glob('**/bin') if p.is_dir())
|
|
||||||
|
|
||||||
paths = sorted(
|
|
||||||
(p for p in path_walker()), key=lambda p: (p.name, p.parent.name), reverse=True
|
|
||||||
)
|
|
||||||
paths.extend(
|
|
||||||
Path(str_path)
|
|
||||||
for str_path in os.get_exec_path(env)
|
|
||||||
if Path(str_path) not in set(paths)
|
|
||||||
)
|
|
||||||
return os.pathsep.join(str(p) for p in paths)
|
|
||||||
|
|
||||||
|
|
||||||
missing_program = '''
|
missing_program = '''
|
||||||
The program '{program}' could not be executed or was not found on your
|
The program '{program}' could not be executed or was not found on your
|
||||||
system PATH.
|
system PATH.
|
||||||
@@ -268,13 +267,30 @@ def _error_old_version(program, package, need_version, found_version, required_f
|
|||||||
|
|
||||||
def check_external_program(
|
def check_external_program(
|
||||||
*,
|
*,
|
||||||
program,
|
program: str,
|
||||||
package,
|
package: str,
|
||||||
version_checker,
|
version_checker: Union[str, Callable],
|
||||||
need_version,
|
need_version: str,
|
||||||
required_for=None,
|
required_for: Optional[str] = None,
|
||||||
recommended=False,
|
recommended=False,
|
||||||
|
version_parser: Type[Version] = LooseVersion,
|
||||||
):
|
):
|
||||||
|
"""Check for required version of external program and raise exception if not.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
program: The name of the program to test.
|
||||||
|
package: The name of a software package that typically supplies this program.
|
||||||
|
Usually the same as program.
|
||||||
|
version_check: A callable without arguments that retrieves the installed
|
||||||
|
version of program.
|
||||||
|
need_version: The minimum required version.
|
||||||
|
required_for: The name of an argument of feature that requires this program.
|
||||||
|
recommended: If this external program is recommended, instead of raising
|
||||||
|
an exception, log a warning and allow execution to continue.
|
||||||
|
version_parser: A class that should be used to parse and compare version
|
||||||
|
numbers. Used when version numbers do not follow standard conventions.
|
||||||
|
"""
|
||||||
|
|
||||||
try:
|
try:
|
||||||
if callable(version_checker):
|
if callable(version_checker):
|
||||||
found_version = version_checker()
|
found_version = version_checker()
|
||||||
@@ -283,7 +299,7 @@ def check_external_program(
|
|||||||
except (CalledProcessError, FileNotFoundError, MissingDependencyError):
|
except (CalledProcessError, FileNotFoundError, MissingDependencyError):
|
||||||
_error_missing_program(program, package, required_for, recommended)
|
_error_missing_program(program, package, required_for, recommended)
|
||||||
if not recommended:
|
if not recommended:
|
||||||
raise MissingDependencyError()
|
raise MissingDependencyError(program)
|
||||||
return
|
return
|
||||||
|
|
||||||
def remove_leading_v(s):
|
def remove_leading_v(s):
|
||||||
@@ -294,9 +310,9 @@ def check_external_program(
|
|||||||
found_version = remove_leading_v(found_version)
|
found_version = remove_leading_v(found_version)
|
||||||
need_version = remove_leading_v(need_version)
|
need_version = remove_leading_v(need_version)
|
||||||
|
|
||||||
if found_version and LooseVersion(found_version) < LooseVersion(need_version):
|
if found_version and version_parser(found_version) < version_parser(need_version):
|
||||||
_error_old_version(program, package, need_version, found_version, required_for)
|
_error_old_version(program, package, need_version, found_version, required_for)
|
||||||
if not recommended:
|
if not recommended:
|
||||||
raise MissingDependencyError()
|
raise MissingDependencyError(program)
|
||||||
|
|
||||||
log.debug('Found %s %s', program, found_version)
|
log.debug('Found %s %s', program, found_version)
|
||||||
@@ -0,0 +1,162 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
import sys
|
||||||
|
from distutils.version import LooseVersion
|
||||||
|
from itertools import chain, filterfalse
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any, Callable, Iterator, Optional, Tuple, TypeVar, cast
|
||||||
|
|
||||||
|
try:
|
||||||
|
import winreg
|
||||||
|
except ModuleNotFoundError as e:
|
||||||
|
raise ModuleNotFoundError("This module is for Windows only") from e
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
T = TypeVar('T')
|
||||||
|
|
||||||
|
|
||||||
|
def registry_enum(
|
||||||
|
key: winreg.HKEYType, enum_fn: Callable[[winreg.HKEYType, int], T]
|
||||||
|
) -> Iterator[T]:
|
||||||
|
LIMIT = 999
|
||||||
|
n = 0
|
||||||
|
while n < LIMIT:
|
||||||
|
try:
|
||||||
|
yield enum_fn(key, n)
|
||||||
|
n += 1
|
||||||
|
except OSError:
|
||||||
|
break
|
||||||
|
if n == LIMIT:
|
||||||
|
raise ValueError(f"Too many registry keys under {key}")
|
||||||
|
|
||||||
|
|
||||||
|
def registry_subkeys(key: winreg.HKEYType) -> Iterator[str]:
|
||||||
|
return registry_enum(key, winreg.EnumKey)
|
||||||
|
|
||||||
|
|
||||||
|
def registry_values(key: winreg.HKEYType) -> Iterator[Tuple[str, Any, int]]:
|
||||||
|
return registry_enum(key, winreg.EnumValue)
|
||||||
|
|
||||||
|
|
||||||
|
def registry_path_ghostscript(env=None) -> Iterator[Path]:
|
||||||
|
try:
|
||||||
|
with winreg.OpenKey(
|
||||||
|
winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Artifex\GPL Ghostscript"
|
||||||
|
) as k:
|
||||||
|
latest_gs = max(registry_subkeys(k), key=LooseVersion)
|
||||||
|
with winreg.OpenKey(
|
||||||
|
winreg.HKEY_LOCAL_MACHINE, fr"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
|
||||||
|
) as k:
|
||||||
|
_, gs_path, _ = next(registry_values(k))
|
||||||
|
yield Path(gs_path) / 'bin'
|
||||||
|
except OSError as e:
|
||||||
|
log.warning(e)
|
||||||
|
|
||||||
|
|
||||||
|
def registry_path_tesseract(env=None) -> Iterator[Path]:
|
||||||
|
try:
|
||||||
|
with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Tesseract-OCR") as k:
|
||||||
|
for subkey, val, _valtype in registry_values(k):
|
||||||
|
if subkey == 'InstallDir':
|
||||||
|
tesseract_path = Path(val)
|
||||||
|
yield tesseract_path
|
||||||
|
except OSError as e:
|
||||||
|
log.warning(e)
|
||||||
|
|
||||||
|
|
||||||
|
def program_files_paths(env=None) -> Iterator[Path]:
|
||||||
|
if not env:
|
||||||
|
env = os.environ
|
||||||
|
program_files = env.get('PROGRAMFILES', '')
|
||||||
|
|
||||||
|
def path_walker() -> Iterator[Path]:
|
||||||
|
for path in Path(program_files).iterdir():
|
||||||
|
if not path.is_dir():
|
||||||
|
continue
|
||||||
|
if path.name.lower() == 'tesseract-ocr':
|
||||||
|
yield path
|
||||||
|
elif path.name.lower() == 'gs':
|
||||||
|
yield from (p for p in path.glob('**/bin') if p.is_dir())
|
||||||
|
|
||||||
|
return iter(
|
||||||
|
sorted(
|
||||||
|
(p for p in path_walker()),
|
||||||
|
key=lambda p: (p.name, p.parent.name),
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def paths_from_env(env=None) -> Iterator[Path]:
|
||||||
|
return (Path(p) for p in os.get_exec_path(env) if p)
|
||||||
|
|
||||||
|
|
||||||
|
def shim_path(new_paths: Callable[[Any], Iterator[Path]], env=None) -> str:
|
||||||
|
if not env:
|
||||||
|
env = os.environ
|
||||||
|
return os.pathsep.join(str(p) for p in new_paths(env) if p)
|
||||||
|
|
||||||
|
|
||||||
|
SHIMS = [
|
||||||
|
paths_from_env,
|
||||||
|
registry_path_ghostscript,
|
||||||
|
registry_path_tesseract,
|
||||||
|
program_files_paths,
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def fix_windows_args(program, args, env):
|
||||||
|
"""Adjust our desired program and command line arguments for use on Windows"""
|
||||||
|
|
||||||
|
if sys.version_info < (3, 8):
|
||||||
|
# bpo-33617 - Windows needs manual Path -> str conversion
|
||||||
|
args = [os.fspath(arg) for arg in args]
|
||||||
|
program = os.fspath(program)
|
||||||
|
|
||||||
|
# If we are running a .py on Windows, ensure we call it with this Python
|
||||||
|
# (to support test suite shims)
|
||||||
|
if program.lower().endswith('.py'):
|
||||||
|
args = [sys.executable] + args
|
||||||
|
|
||||||
|
# If the program we want is not on the PATH, check elsewhere
|
||||||
|
for shim in SHIMS:
|
||||||
|
shimmed_path = shim_path(shim, env)
|
||||||
|
new_args0 = shutil.which(args[0], path=shimmed_path)
|
||||||
|
if new_args0:
|
||||||
|
args[0] = new_args0
|
||||||
|
break
|
||||||
|
|
||||||
|
return args
|
||||||
|
|
||||||
|
|
||||||
|
def unique_everseen(iterable, key=None):
|
||||||
|
"List unique elements, preserving order. Remember all elements ever seen."
|
||||||
|
# unique_everseen('AAAABBBCCDAABBB') --> A B C D
|
||||||
|
# unique_everseen('ABBCcAD', str.lower) --> A B C D
|
||||||
|
seen = set()
|
||||||
|
seen_add = seen.add
|
||||||
|
if key is None:
|
||||||
|
key = lambda x: x
|
||||||
|
for element in iterable:
|
||||||
|
k = key(element)
|
||||||
|
if k not in seen:
|
||||||
|
seen_add(k)
|
||||||
|
yield element
|
||||||
|
|
||||||
|
|
||||||
|
def shim_env_path(env=None):
|
||||||
|
if env is None:
|
||||||
|
env = os.environ
|
||||||
|
|
||||||
|
shim_paths = chain.from_iterable(shim(env) for shim in SHIMS)
|
||||||
|
return os.pathsep.join(
|
||||||
|
str(p) for p in unique_everseen(shim_paths, key=lambda p: str.casefold(str(p)))
|
||||||
|
)
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
# © 2021 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
# Empty __init__.py file
|
||||||
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
+1
-1
@@ -9,7 +9,7 @@
|
|||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.mrqsewbu/000001_ocr.png"; bbox 0 0 1000 800; ppageno 0'>
|
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/ocrmypdf.io.h7hsz1g7/000001_ocr.png"; bbox 0 0 1000 800; ppageno 0'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 296 96 704 504">
|
<div class='ocr_carea' id='block_1_1' title="bbox 296 96 704 504">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 296 96 704 504">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 296 96 704 504">
|
||||||
<span class='ocr_line' id='line_1_1' title="bbox 296 96 704 504; baseline 0 296; x_size 169.33333; x_descenders 42.333332; x_ascenders 42.333336">
|
<span class='ocr_line' id='line_1_1' title="bbox 296 96 704 504; baseline 0 296; x_size 169.33333; x_descenders 42.333332; x_ascenders 42.333336">
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -9,7 +9,7 @@
|
|||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.ces85e5u/000001_ocr.png"; bbox 0 0 2550 3300; ppageno 0'>
|
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/ocrmypdf.io.tgp04npj/000001_ocr.png"; bbox 0 0 2550 3300; ppageno 0'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
||||||
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
||||||
|
|||||||
BIN
Binary file not shown.
+888
-888
File diff suppressed because it is too large
Load Diff
+3
-3
@@ -103,7 +103,7 @@ ERASE, REPEAT, PLAY/STOP, or LOCATE.
|
|||||||
|
|
||||||
© Will sync to standard LinnDrum or Linn 9000 sync tone.
|
© Will sync to standard LinnDrum or Linn 9000 sync tone.
|
||||||
|
|
||||||
® Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation.
|
© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation.
|
||||||
* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second,
|
* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second,
|
||||||
|
|
||||||
(even drop frame!)
|
(even drop frame!)
|
||||||
@@ -115,9 +115,9 @@ on the TAP TEMPO button.
|
|||||||
¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired.
|
¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired.
|
||||||
¢ Any TIME SIGNATURE may be used, and may be changed within a song.
|
¢ Any TIME SIGNATURE may be used, and may be changed within a song.
|
||||||
|
|
||||||
nn
|
linn
|
||||||
|
|
||||||
Linn Electronics, Inc.
|
Linn Electronics, Inc.
|
||||||
|
|
||||||
18720 Oxnard Street, Tarzana, CA 91356
|
18720 Oxnard Street, Tarzana, CA 91356
|
||||||
(818) 708-8131 TELEX #298949 LINN UR
|
(818) 708-8131 TELEX #298949 LINN UR
|
||||||
|
|
||||||
BIN
Binary file not shown.
+977
-1007
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user