Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
21fb6c82ca | ||
|
|
27f7b9f255 | ||
|
|
6f31a92ffb | ||
|
|
da2276788c | ||
|
|
dc6f1a266a | ||
|
|
9c8ddd853d | ||
|
|
014d0302f2 | ||
|
|
65568b3dbc | ||
|
|
05e2b6698d | ||
|
|
2f8e0f7d95 | ||
|
|
7e7553fc6b | ||
|
|
6b425aaebe | ||
|
|
725af43bc3 | ||
|
|
5c60309609 | ||
|
|
7d5cd55909 | ||
|
|
ec4a06fad2 | ||
|
|
48c6e2318e | ||
|
|
2eafa5e070 | ||
|
|
6aa04d7569 | ||
|
|
b9bffa97ba | ||
|
|
777ba99ccc | ||
|
|
001b3324f1 | ||
|
|
b1f2d257e2 | ||
|
|
24a08e5170 | ||
|
|
adf97fd82c | ||
|
|
a60ea72517 | ||
|
|
59f967cdcd | ||
|
|
6e439ee89e | ||
|
|
da38e1b035 | ||
|
|
28c60c4f82 | ||
|
|
a5efc4af9b | ||
|
|
1141235c42 | ||
|
|
2b6b7a4975 | ||
|
|
b062c9e8c0 | ||
|
|
ed632ae366 | ||
|
|
af742229e7 | ||
|
|
e2d998245d | ||
|
|
8c58e95c3a | ||
|
|
61600111d3 | ||
|
|
e4c45e3d3b | ||
|
|
7cabbb125f | ||
|
|
d8753dc790 | ||
|
|
ef43d7e016 | ||
|
|
17a5b8b43c | ||
|
|
13d11e76e5 | ||
|
|
61069660a2 | ||
|
|
685a06c93d | ||
|
|
6cdf68363a | ||
|
|
522ff3c21a | ||
|
|
10245dc954 | ||
|
|
3d4f80639d | ||
|
|
db9a22c9dd | ||
|
|
31683530f8 | ||
|
|
0e550a1c6d | ||
|
|
b17fb61389 | ||
|
|
d640c2ded3 | ||
|
|
a0ac448d52 | ||
|
|
e3ba13e365 | ||
|
|
0cd04abc4e | ||
|
|
ee81f3968f | ||
|
|
21cacad93b | ||
|
|
3589f4e7d1 | ||
|
|
1cdc2591e5 | ||
|
|
e05f9575a8 | ||
|
|
10c703e119 | ||
|
|
0ac15dd0b2 | ||
|
|
808b24d59f | ||
|
|
c082526dea | ||
|
|
33cdabaf65 | ||
|
|
94f8e36601 | ||
|
|
865002c7be | ||
|
|
5d0cc0a092 | ||
|
|
6c427f82ea | ||
|
|
e7a44ba87a | ||
|
|
c311768452 | ||
|
|
f53fedee63 | ||
|
|
87838127b0 | ||
|
|
4db4df5c72 | ||
|
|
11125c5367 | ||
|
|
e648411067 | ||
|
|
11365575d7 | ||
|
|
845cb5c40c | ||
|
|
b699e158be | ||
|
|
603da52026 | ||
|
|
8d0765a5e0 | ||
|
|
1ca327e13b | ||
|
|
f504fd1875 | ||
|
|
cf7c20ca16 | ||
|
|
b00fe3dc5d | ||
|
|
e6aa3a4299 | ||
|
|
24f1b57288 | ||
|
|
43302d7e12 | ||
|
|
fed0226761 | ||
|
|
27e22b4f07 | ||
|
|
79382a6039 | ||
|
|
7788d94c4a | ||
|
|
33bfba8449 | ||
|
|
1d0584c644 | ||
|
|
84b9d4d021 | ||
|
|
41efd3bf0f | ||
|
|
776ada6713 | ||
|
|
f3593c915d | ||
|
|
dfe31a2f6d | ||
|
|
0c43963d69 | ||
|
|
f29fe7f23e | ||
|
|
04996caac3 | ||
|
|
13917c051c | ||
|
|
8182fe9c92 | ||
|
|
1950acfbda | ||
|
|
fca6403083 | ||
|
|
c4e2fce1ef | ||
|
|
3546479658 | ||
|
|
72442fa3d0 | ||
|
|
8f714b1375 | ||
|
|
cb05c1d122 | ||
|
|
b0ad07bc5f | ||
|
|
514038d4ec | ||
|
|
50d76e7f6c | ||
|
|
6c78a46285 | ||
|
|
863d560632 | ||
|
|
73934c854c | ||
|
|
2be8eeec2c | ||
|
|
3dfde479e2 | ||
|
|
aea1862644 | ||
|
|
3b406112d0 | ||
|
|
fcc4c2d371 | ||
|
|
3de18ed612 | ||
|
|
93cca42e20 | ||
|
|
2d0ac4707c | ||
|
|
7d208175cf | ||
|
|
ea69e868ed | ||
|
|
beea603ab3 | ||
|
|
7966192d6e | ||
|
|
5acbd7a252 | ||
|
|
aed955ca8c | ||
|
|
298bdb8690 | ||
|
|
1a58abcc6a | ||
|
|
dbfceba020 | ||
|
|
0faa618c3c | ||
|
|
7035002c03 | ||
|
|
f8fadaef41 | ||
|
|
ee21bf9ef6 | ||
|
|
190ca81951 | ||
|
|
d48254d477 | ||
|
|
1ec2ccca14 | ||
|
|
e78f0cc56f | ||
|
|
13af3252ff | ||
|
|
0528867e0b | ||
|
|
6910c48b81 | ||
|
|
69aa3981c4 | ||
|
|
9c1e5adfe6 | ||
|
|
e642dd4b35 | ||
|
|
9de06f62ee | ||
|
|
1414a8f5dc | ||
|
|
26badf2882 | ||
|
|
8f873aaa45 | ||
|
|
8fdcb15b4e | ||
|
|
0323738ada | ||
|
|
aae5591f7e | ||
|
|
4c1ff1086c | ||
|
|
f91faf9795 | ||
|
|
793cc33a90 | ||
|
|
fbd72efd45 | ||
|
|
1115923995 | ||
|
|
8478d67b28 | ||
|
|
c75ff4687a | ||
|
|
312c1e51b5 | ||
|
|
cfe2bb25ba | ||
|
|
cd49e70154 | ||
|
|
7ce1692eef | ||
|
|
7959f7628d | ||
|
|
4634b20de5 | ||
|
|
3810e576ff | ||
|
|
01c7895044 | ||
|
|
fdc6aa03fb | ||
|
|
25cc17ee03 | ||
|
|
e8098a1475 | ||
|
|
6b773883dc | ||
|
|
4ed9622335 | ||
|
|
acc9d58c39 | ||
|
|
659e738f92 | ||
|
|
7b3d7ca92a | ||
|
|
e3126d2806 | ||
|
|
45020a7fcd | ||
|
|
f51164aff8 | ||
|
|
6f58a14351 | ||
|
|
7ba04267b1 | ||
|
|
9749564313 | ||
|
|
698e8791d7 | ||
|
|
380b981763 | ||
|
|
5abfb14c2a | ||
|
|
036afc4d88 | ||
|
|
59642a98b2 | ||
|
|
f8c6be2e26 | ||
|
|
42bf5476dd | ||
|
|
30440104ba | ||
|
|
b159e02110 | ||
|
|
a55ab05d16 | ||
|
|
25d046ae95 | ||
|
|
01b0f76e36 | ||
|
|
d74d315e8b | ||
|
|
8be9a68c5e | ||
|
|
6c34d59836 | ||
|
|
386453d178 | ||
|
|
615a7561b5 | ||
|
|
c4c64c3ea0 | ||
|
|
21279f5784 | ||
|
|
a63a21a7fc | ||
|
|
1c4d5d79f7 | ||
|
|
644581ed3c | ||
|
|
77f7621bbc | ||
|
|
42713b77d7 | ||
|
|
690f88119d | ||
|
|
78f391536b | ||
|
|
7bdd1828a9 | ||
|
|
a8f513eeeb | ||
|
|
af18bc0684 | ||
|
|
b621df6947 | ||
|
|
313c9e7dc1 | ||
|
|
9d04795f7f | ||
|
|
9a08e71e7f | ||
|
|
790d3022f6 | ||
|
|
ec311af796 | ||
|
|
c725bf79da | ||
|
|
9559f76fae | ||
|
|
45736b7c2b | ||
|
|
5629e960b9 | ||
|
|
79fd8d01a5 | ||
|
|
79fe7a0a85 | ||
|
|
b4b32a35b5 | ||
|
|
4634b3db55 | ||
|
|
f5053158d4 | ||
|
|
dfa4ce1612 | ||
|
|
585595a98e | ||
|
|
f6396fbaac | ||
|
|
3859bae85e | ||
|
|
ee1a7baae7 | ||
|
|
a4da05b66b | ||
|
|
4d67812d51 | ||
|
|
3534742ef9 | ||
|
|
8bfd46c80d | ||
|
|
cc6e9cecc0 | ||
|
|
208657f840 | ||
|
|
f3de980447 | ||
|
|
eb8992e58b | ||
|
|
72ad618ae6 | ||
|
|
f07d0c39bb | ||
|
|
9c5c7d9be0 | ||
|
|
0b19b084e2 | ||
|
|
9b4516af7a | ||
|
|
1eb45de5c9 | ||
|
|
390b9924f5 | ||
|
|
c28858a099 | ||
|
|
f00b3c00cd | ||
|
|
4e4f0bfa1f | ||
|
|
0a31acf888 | ||
|
|
b91096c615 | ||
|
|
95d9e8d91a | ||
|
|
cb6c1939e9 | ||
|
|
3764ee872a | ||
|
|
e402d5cb4b | ||
|
|
53cd04799a | ||
|
|
f2545d4496 | ||
|
|
9b81e76ed4 | ||
|
|
0956fc81aa | ||
|
|
72279e7759 | ||
|
|
6f9b948064 | ||
|
|
4eca0a165b | ||
|
|
067e61e03a | ||
|
|
1b46481f7e |
+2
-2
@@ -1,7 +1,7 @@
|
|||||||
# OCRmyPDF
|
# OCRmyPDF
|
||||||
#
|
#
|
||||||
|
|
||||||
FROM ubuntu:21.04 as base
|
FROM ubuntu:22.04 as base
|
||||||
|
|
||||||
ENV LANG=C.UTF-8
|
ENV LANG=C.UTF-8
|
||||||
ENV TZ=UTC
|
ENV TZ=UTC
|
||||||
@@ -15,6 +15,7 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
|||||||
|
|
||||||
FROM base as builder
|
FROM base as builder
|
||||||
|
|
||||||
|
# Note we need leptonica here to build jbig2
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
build-essential autoconf automake libtool \
|
build-essential autoconf automake libtool \
|
||||||
libleptonica-dev \
|
libleptonica-dev \
|
||||||
@@ -74,7 +75,6 @@ COPY --from=builder /app/misc/watcher.py /app/
|
|||||||
|
|
||||||
# Copy minimal project files to get the test suite.
|
# Copy minimal project files to get the test suite.
|
||||||
COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/
|
COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/
|
||||||
COPY --from=builder /app/requirements /app/requirements
|
|
||||||
COPY --from=builder /app/tests /app/tests
|
COPY --from=builder /app/tests /app/tests
|
||||||
|
|
||||||
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
||||||
|
|||||||
+4
-1
@@ -1 +1,4 @@
|
|||||||
ref-names: $Format:%D$
|
node: $Format:%H$
|
||||||
|
node-date: $Format:%cI$
|
||||||
|
describe-name: $Format:%(describe:tags=true)$
|
||||||
|
ref-names: $Format:%D$
|
||||||
|
|||||||
@@ -22,7 +22,7 @@ Run with verbosity or higher `-v1` to see more detailed logging. This informatio
|
|||||||
**Example file**
|
**Example file**
|
||||||
If your issue is a problem that affects only certain files, and we will require an input file (PDF or image) that demonstrates your issue.
|
If your issue is a problem that affects only certain files, and we will require an input file (PDF or image) that demonstrates your issue.
|
||||||
|
|
||||||
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/jbarlow83/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/ocrmypdf/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
||||||
|
|
||||||
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
||||||
|
|
||||||
|
|||||||
@@ -19,7 +19,7 @@ A clear and concise description of any alternative solutions or features you've
|
|||||||
**Example file**
|
**Example file**
|
||||||
If your issue concerns how OCRmyPDF processes certain files, and please provide an example file that helps illustrate how OCRmyPDF's output could be improve.
|
If your issue concerns how OCRmyPDF processes certain files, and please provide an example file that helps illustrate how OCRmyPDF's output could be improve.
|
||||||
|
|
||||||
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/jbarlow83/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/ocrmypdf/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
||||||
|
|
||||||
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# To get started with Dependabot version updates, you'll need to specify which
|
||||||
|
# package ecosystems to update and where the package manifests are located.
|
||||||
|
# Please see the documentation for all configuration options:
|
||||||
|
# https://docs.github.com/github/administering-a-repository/configuration-options-for-dependency-updates
|
||||||
|
|
||||||
|
version: 2
|
||||||
|
updates:
|
||||||
|
- package-ecosystem: "github-actions" # See documentation for possible values
|
||||||
|
directory: "/" # Location of package manifests
|
||||||
|
schedule:
|
||||||
|
interval: "weekly"
|
||||||
+44
-34
@@ -6,6 +6,7 @@ on:
|
|||||||
- master
|
- master
|
||||||
- ci
|
- ci
|
||||||
- release/*
|
- release/*
|
||||||
|
- feature/*
|
||||||
tags:
|
tags:
|
||||||
- v*
|
- v*
|
||||||
paths-ignore:
|
paths-ignore:
|
||||||
@@ -20,17 +21,19 @@ jobs:
|
|||||||
matrix:
|
matrix:
|
||||||
include:
|
include:
|
||||||
- os: ubuntu-18.04
|
- os: ubuntu-18.04
|
||||||
python: 3.6
|
python: "3.7"
|
||||||
- os: ubuntu-18.04
|
|
||||||
python: 3.7
|
|
||||||
- os: ubuntu-20.04
|
- os: ubuntu-20.04
|
||||||
python: 3.8
|
python: "3.8"
|
||||||
- os: ubuntu-20.04
|
- os: ubuntu-20.04
|
||||||
python: 3.9
|
python: "3.9"
|
||||||
|
- os: ubuntu-20.04
|
||||||
|
python: "3.10"
|
||||||
- os: ubuntu-latest
|
- os: ubuntu-latest
|
||||||
python: 3.9
|
python: "3.9"
|
||||||
- os: ubuntu-latest
|
- os: ubuntu-latest
|
||||||
python: 3.9
|
python: "pypy-3.8"
|
||||||
|
- os: ubuntu-latest
|
||||||
|
python: "3.9"
|
||||||
tesseract5: true
|
tesseract5: true
|
||||||
|
|
||||||
env:
|
env:
|
||||||
@@ -38,11 +41,11 @@ jobs:
|
|||||||
PYTHON: ${{ matrix.python }}
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v2
|
- uses: actions/checkout@v3
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v2
|
- uses: actions/setup-python@v4
|
||||||
name: Install Python
|
name: Install Python
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python }}
|
python-version: ${{ matrix.python }}
|
||||||
@@ -60,13 +63,13 @@ jobs:
|
|||||||
ghostscript \
|
ghostscript \
|
||||||
img2pdf \
|
img2pdf \
|
||||||
libffi-dev \
|
libffi-dev \
|
||||||
liblept5 \
|
|
||||||
libsm6 libxext6 libxrender-dev \
|
libsm6 libxext6 libxrender-dev \
|
||||||
pngquant \
|
pngquant \
|
||||||
poppler-utils \
|
poppler-utils \
|
||||||
tesseract-ocr \
|
tesseract-ocr \
|
||||||
tesseract-ocr-deu \
|
tesseract-ocr-deu \
|
||||||
tesseract-ocr-eng \
|
tesseract-ocr-eng \
|
||||||
|
tesseract-ocr-osd \
|
||||||
unpaper \
|
unpaper \
|
||||||
zlib1g
|
zlib1g
|
||||||
|
|
||||||
@@ -77,13 +80,22 @@ jobs:
|
|||||||
libexempi3
|
libexempi3
|
||||||
|
|
||||||
- name: Install Ubuntu 20.04 packages
|
- name: Install Ubuntu 20.04 packages
|
||||||
if: matrix.os == 'ubuntu-20.04'
|
if: matrix.os == 'ubuntu-20.04' || matrix.os == 'ubuntu-latest'
|
||||||
run: |
|
run: |
|
||||||
sudo apt-get install -y --no-install-recommends \
|
sudo apt-get install -y --no-install-recommends \
|
||||||
libexempi8
|
libexempi8
|
||||||
|
|
||||||
|
- name: Install Ubuntu packages for PyPy
|
||||||
|
if: startsWith(matrix.python, 'pypy')
|
||||||
|
run: |
|
||||||
|
sudo apt-get install -y --no-install-recommends \
|
||||||
|
libxml2-dev \
|
||||||
|
libxslt1-dev \
|
||||||
|
pypy3-dev
|
||||||
|
|
||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
|
python -m pip install --upgrade pip wheel
|
||||||
python -m pip install .[test]
|
python -m pip install .[test]
|
||||||
|
|
||||||
- name: Report versions
|
- name: Report versions
|
||||||
@@ -99,7 +111,7 @@ jobs:
|
|||||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v1
|
uses: codecov/codecov-action@v3
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
@@ -110,18 +122,18 @@ jobs:
|
|||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
os: [macos-latest]
|
os: [macos-latest]
|
||||||
python: ["3.9"]
|
python: ["3.9", "3.10"]
|
||||||
|
|
||||||
env:
|
env:
|
||||||
OS: ${{ matrix.os }}
|
OS: ${{ matrix.os }}
|
||||||
PYTHON: ${{ matrix.python }}
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v2
|
- uses: actions/checkout@v3
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v2
|
- uses: actions/setup-python@v4
|
||||||
name: Install Python
|
name: Install Python
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python }}
|
python-version: ${{ matrix.python }}
|
||||||
@@ -133,14 +145,13 @@ jobs:
|
|||||||
exempi \
|
exempi \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
jbig2enc \
|
jbig2enc \
|
||||||
leptonica \
|
|
||||||
openjpeg \
|
openjpeg \
|
||||||
pngquant \
|
pngquant \
|
||||||
tesseract
|
tesseract
|
||||||
|
|
||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip
|
python -m pip install --upgrade pip wheel
|
||||||
python -m pip install .[test]
|
python -m pip install .[test]
|
||||||
|
|
||||||
- name: Report versions
|
- name: Report versions
|
||||||
@@ -155,7 +166,7 @@ jobs:
|
|||||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v1
|
uses: codecov/codecov-action@v3
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
@@ -166,18 +177,18 @@ jobs:
|
|||||||
strategy:
|
strategy:
|
||||||
matrix:
|
matrix:
|
||||||
os: [windows-latest]
|
os: [windows-latest]
|
||||||
python: ["3.9"]
|
python: ["3.9", "3.10"]
|
||||||
|
|
||||||
env:
|
env:
|
||||||
OS: ${{ matrix.os }}
|
OS: ${{ matrix.os }}
|
||||||
PYTHON: ${{ matrix.python }}
|
PYTHON: ${{ matrix.python }}
|
||||||
|
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v2
|
- uses: actions/checkout@v3
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v2
|
- uses: actions/setup-python@v4
|
||||||
name: Install Python
|
name: Install Python
|
||||||
with:
|
with:
|
||||||
python-version: ${{ matrix.python }}
|
python-version: ${{ matrix.python }}
|
||||||
@@ -185,12 +196,11 @@ jobs:
|
|||||||
- name: Install system packages
|
- name: Install system packages
|
||||||
run: |
|
run: |
|
||||||
choco install --yes --no-progress --pre tesseract
|
choco install --yes --no-progress --pre tesseract
|
||||||
choco install --yes --no-progress ghostscript
|
choco install --yes --no-progress --ignore-checksums ghostscript
|
||||||
choco install --yes --no-progress pngquant
|
|
||||||
|
|
||||||
- name: Install Python packages
|
- name: Install Python packages
|
||||||
run: |
|
run: |
|
||||||
python -m pip install --upgrade pip
|
python -m pip install --upgrade pip wheel
|
||||||
python -m pip install .[test]
|
python -m pip install .[test]
|
||||||
|
|
||||||
- name: Test
|
- name: Test
|
||||||
@@ -198,7 +208,7 @@ jobs:
|
|||||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||||
|
|
||||||
- name: Upload coverage to Codecov
|
- name: Upload coverage to Codecov
|
||||||
uses: codecov/codecov-action@v1
|
uses: codecov/codecov-action@v3
|
||||||
with:
|
with:
|
||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
@@ -207,14 +217,14 @@ jobs:
|
|||||||
name: Build sdist and wheels
|
name: Build sdist and wheels
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/checkout@v2
|
- uses: actions/checkout@v3
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- uses: actions/setup-python@v2
|
- uses: actions/setup-python@v4
|
||||||
name: Install Python
|
name: Install Python
|
||||||
with:
|
with:
|
||||||
python-version: "3.6"
|
python-version: "3.7"
|
||||||
|
|
||||||
- name: Make wheels and sdist
|
- name: Make wheels and sdist
|
||||||
run: |
|
run: |
|
||||||
@@ -222,7 +232,7 @@ jobs:
|
|||||||
python setup.py sdist
|
python setup.py sdist
|
||||||
python setup.py bdist_wheel
|
python setup.py bdist_wheel
|
||||||
|
|
||||||
- uses: actions/upload-artifact@v2
|
- uses: actions/upload-artifact@v3
|
||||||
with:
|
with:
|
||||||
path: |
|
path: |
|
||||||
./dist/*.whl
|
./dist/*.whl
|
||||||
@@ -234,7 +244,7 @@ jobs:
|
|||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||||
steps:
|
steps:
|
||||||
- uses: actions/download-artifact@v2
|
- uses: actions/download-artifact@v3
|
||||||
with:
|
with:
|
||||||
name: artifact
|
name: artifact
|
||||||
path: dist
|
path: dist
|
||||||
@@ -264,22 +274,22 @@ jobs:
|
|||||||
- name: Set image name
|
- name: Set image name
|
||||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
||||||
|
|
||||||
- uses: actions/checkout@v2
|
- uses: actions/checkout@v3
|
||||||
with:
|
with:
|
||||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||||
|
|
||||||
- name: Login to Docker Hub
|
- name: Login to Docker Hub
|
||||||
uses: docker/login-action@v1
|
uses: docker/login-action@v2
|
||||||
with:
|
with:
|
||||||
username: jbarlow83
|
username: jbarlow83
|
||||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||||
|
|
||||||
- name: Set up QEMU
|
- name: Set up QEMU
|
||||||
uses: docker/setup-qemu-action@v1
|
uses: docker/setup-qemu-action@v2
|
||||||
|
|
||||||
- name: Set up Docker Buildx
|
- name: Set up Docker Buildx
|
||||||
id: buildx
|
id: buildx
|
||||||
uses: docker/setup-buildx-action@v1
|
uses: docker/setup-buildx-action@v2
|
||||||
|
|
||||||
- name: Print image tag
|
- name: Print image tag
|
||||||
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
||||||
|
|||||||
+23
-9
@@ -1,23 +1,37 @@
|
|||||||
repos:
|
repos:
|
||||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||||
rev: v3.4.0
|
rev: v4.3.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: check-case-conflict
|
- id: check-case-conflict
|
||||||
- id: check-merge-conflict
|
- id: check-merge-conflict
|
||||||
- id: check-toml
|
- id: check-toml
|
||||||
- id: check-yaml
|
- id: check-yaml
|
||||||
- id: debug-statements
|
- id: debug-statements
|
||||||
- repo: https://github.com/asottile/seed-isort-config
|
- repo: https://github.com/pycqa/isort
|
||||||
rev: v2.2.0
|
rev: 5.10.1
|
||||||
hooks:
|
|
||||||
- id: seed-isort-config
|
|
||||||
- repo: https://github.com/pre-commit/mirrors-isort
|
|
||||||
rev: v5.7.0 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases
|
|
||||||
hooks:
|
hooks:
|
||||||
- id: isort
|
- id: isort
|
||||||
|
args: ["--profile", "black", "-a", "from __future__ import annotations"]
|
||||||
- repo: https://github.com/psf/black
|
- repo: https://github.com/psf/black
|
||||||
rev: 20.8b1
|
rev: 22.6.0
|
||||||
hooks:
|
hooks:
|
||||||
- id: black
|
- id: black
|
||||||
language_version: python
|
language_version: python
|
||||||
exclude: ^src/ocrmypdf/lib/_leptonica.py
|
- repo: https://github.com/asottile/setup-cfg-fmt
|
||||||
|
rev: v1.20.2
|
||||||
|
hooks:
|
||||||
|
- id: setup-cfg-fmt
|
||||||
|
- repo: https://github.com/asottile/pyupgrade
|
||||||
|
rev: v2.37.2
|
||||||
|
hooks:
|
||||||
|
- id: pyupgrade
|
||||||
|
args: ["--py37-plus"]
|
||||||
|
- repo: https://github.com/pre-commit/mirrors-mypy
|
||||||
|
rev: v0.971
|
||||||
|
hooks:
|
||||||
|
- id: mypy
|
||||||
|
additional_dependencies:
|
||||||
|
- types-toml
|
||||||
|
- types-setuptools
|
||||||
|
- types-requests
|
||||||
|
- types-Pillow
|
||||||
|
|||||||
+1
-1
@@ -14,7 +14,7 @@ formats:
|
|||||||
|
|
||||||
# Optionally set the version of Python and requirements required to build your docs
|
# Optionally set the version of Python and requirements required to build your docs
|
||||||
python:
|
python:
|
||||||
version: 3.7
|
version: "3.7"
|
||||||
install:
|
install:
|
||||||
- method: pip
|
- method: pip
|
||||||
path: .
|
path: .
|
||||||
|
|||||||
@@ -1,9 +1,7 @@
|
|||||||
<img src="docs/images/logo.svg" width="240" alt="OCRmyPDF">
|
<img src="docs/images/logo.svg" width="240" alt="OCRmyPDF">
|
||||||
|
|
||||||
[](https://github.com/jbarlow83/OCRmyPDF/actions/workflows/build.yml) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions]
|
[](https://github.com/ocrmypdf/OCRmyPDF/actions/workflows/build.yml) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions]
|
||||||
|
|
||||||
[azure]: https://dev.azure.com/jim0585/ocrmypdf/_apis/build/status/jbarlow83.OCRmyPDF?branchName=master
|
|
||||||
[travis]: https://travis-ci.org/jbarlow83/OCRmyPDF.svg?branch=master "Travis build status"
|
|
||||||
[pypi]: https://img.shields.io/pypi/v/ocrmypdf.svg "PyPI version"
|
[pypi]: https://img.shields.io/pypi/v/ocrmypdf.svg "PyPI version"
|
||||||
[homebrew]: https://img.shields.io/homebrew/v/ocrmypdf.svg "Homebrew version"
|
[homebrew]: https://img.shields.io/homebrew/v/ocrmypdf.svg "Homebrew version"
|
||||||
[docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD"
|
[docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD"
|
||||||
@@ -64,7 +62,8 @@ Linux, Windows, macOS and FreeBSD are supported. Docker images are also availabl
|
|||||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||||
| Fedora | ``dnf install ocrmypdf`` |
|
| Fedora | ``dnf install ocrmypdf`` |
|
||||||
| macOS | ``brew install ocrmypdf`` |
|
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||||
|
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
| FreeBSD | ``pkg install py37-ocrmypdf`` |
|
| FreeBSD | ``pkg install py37-ocrmypdf`` |
|
||||||
| Conda | ``conda install ocrmypdf`` |
|
| Conda | ``conda install ocrmypdf`` |
|
||||||
@@ -106,11 +105,11 @@ ocrmypdf --help
|
|||||||
|
|
||||||
Our [documentation is served on Read the Docs](https://ocrmypdf.readthedocs.io/en/latest/index.html).
|
Our [documentation is served on Read the Docs](https://ocrmypdf.readthedocs.io/en/latest/index.html).
|
||||||
|
|
||||||
Please report issues on our [GitHub issues](https://github.com/jbarlow83/OCRmyPDF/issues) page, and follow the issue template for quick response.
|
Please report issues on our [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) page, and follow the issue template for quick response.
|
||||||
|
|
||||||
## Requirements
|
## Requirements
|
||||||
|
|
||||||
In addition to the required Python version (3.6+), OCRmyPDF requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. OCRmyPDF is pure Python, but uses CFFI to portably generate library bindings. OCRmyPDF works on pretty much everything: Linux, macOS, Windows and FreeBSD.
|
In addition to the required Python version (3.7+), OCRmyPDF requires external program installations of Ghostscript and Tesseract OCR. OCRmyPDF is pure Python, and runs on pretty much everything: Linux, macOS, Windows and FreeBSD.
|
||||||
|
|
||||||
## Press & Media
|
## Press & Media
|
||||||
|
|
||||||
|
|||||||
Vendored
+1
-6
@@ -1,7 +1,7 @@
|
|||||||
Format: https://www.debian.org/doc/packaging-manuals/copyright-format/1.0/
|
Format: https://www.debian.org/doc/packaging-manuals/copyright-format/1.0/
|
||||||
Upstream-Name: OCRmyPDF
|
Upstream-Name: OCRmyPDF
|
||||||
Upstream-Contact: James R. Barlow <barlow.jim@gmail.com>
|
Upstream-Contact: James R. Barlow <barlow.jim@gmail.com>
|
||||||
Source: https://github.com/jbarlow83/OCRmyPDF
|
Source: https://github.com/ocrmypdf/OCRmyPDF
|
||||||
|
|
||||||
Files: *
|
Files: *
|
||||||
Copyright:
|
Copyright:
|
||||||
@@ -60,11 +60,6 @@ Copyright: (C) 2010 Jonathan Brinley <jonathanbrinley@gmail.com>
|
|||||||
(C) 2015-16 James R. Barlow
|
(C) 2015-16 James R. Barlow
|
||||||
License: Expat
|
License: Expat
|
||||||
|
|
||||||
Files: src/ocrmypdf/_unicodefun.py
|
|
||||||
Copyright: (C) 2014 Armin Ronacher
|
|
||||||
(C) 2017 James R. Barlow
|
|
||||||
License: BSD-3-clause
|
|
||||||
|
|
||||||
Files: tests/plugins/*
|
Files: tests/plugins/*
|
||||||
Copyright: (C) 2016, 2017, 2016-2018 James R. Barlow
|
Copyright: (C) 2016, 2017, 2016-2018 James R. Barlow
|
||||||
License: Expat
|
License: Expat
|
||||||
|
|||||||
+8
-12
@@ -67,11 +67,11 @@ without modifying the PDF. This is to ensure that PDFs that were
|
|||||||
previously OCRed or were "born digital" rather than scanned are not
|
previously OCRed or were "born digital" rather than scanned are not
|
||||||
processed.
|
processed.
|
||||||
|
|
||||||
If ``--skip-text`` is issued, then no OCR will be performed on pages
|
If ``--skip-text`` is issued, then no image processing or OCR will be
|
||||||
that already have text. The page will be copied to the output. This may
|
performed on pages that already have text. The page will be copied to
|
||||||
be useful for documents that contain both "born digital" and scanned
|
the output. This may be useful for documents that contain both "born
|
||||||
content, or to use OCRmyPDF to normalize and convert to PDF/A regardless
|
digital" and scanned content, or to use OCRmyPDF to normalize and
|
||||||
of their contents.
|
convert to PDF/A regardless of their contents.
|
||||||
|
|
||||||
If ``--redo-ocr`` is issued, then a detailed text analysis is performed.
|
If ``--redo-ocr`` is issued, then a detailed text analysis is performed.
|
||||||
Text is categorized as either visible or invisible. Invisible text (OCR)
|
Text is categorized as either visible or invisible. Invisible text (OCR)
|
||||||
@@ -223,7 +223,9 @@ The ``hocr`` renderer
|
|||||||
The ``hocr`` renderer works with older versions of Tesseract. The image
|
The ``hocr`` renderer works with older versions of Tesseract. The image
|
||||||
layer is copied from the original PDF page if possible, avoiding
|
layer is copied from the original PDF page if possible, avoiding
|
||||||
potentially lossy transcoding or loss of other PDF information. If
|
potentially lossy transcoding or loss of other PDF information. If
|
||||||
preprocessing is specified, then the image layer is a new PDF.
|
preprocessing is specified, then the image layer is a new PDF. (You may
|
||||||
|
need to disable PDF/A conversion nad optimization to eliminate all
|
||||||
|
lossy transformations.)
|
||||||
|
|
||||||
Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone
|
Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone
|
||||||
looking to customize how OCR is presented should look here. A major
|
looking to customize how OCR is presented should look here. A major
|
||||||
@@ -236,12 +238,6 @@ PDF.js viewer.
|
|||||||
|
|
||||||
This works in all versions of Tesseract.
|
This works in all versions of Tesseract.
|
||||||
|
|
||||||
The ``tesseract`` renderer
|
|
||||||
--------------------------
|
|
||||||
|
|
||||||
The ``tesseract`` renderer was removed. OCRmyPDF's new approach to text
|
|
||||||
layer grafting makes it functionally equivalent to ``sandwich``.
|
|
||||||
|
|
||||||
Return code policy
|
Return code policy
|
||||||
==================
|
==================
|
||||||
|
|
||||||
|
|||||||
+16
-8
@@ -12,7 +12,7 @@ subprocess call anyway, as this provides isolation of its activities.
|
|||||||
Example
|
Example
|
||||||
=======
|
=======
|
||||||
|
|
||||||
OCRmyPDF one high-level function to run its main engine from an
|
OCRmyPDF provides one high-level function to run its main engine from an
|
||||||
application. The parameters are symmetric to the command line arguments
|
application. The parameters are symmetric to the command line arguments
|
||||||
and largely have the same functions.
|
and largely have the same functions.
|
||||||
|
|
||||||
@@ -23,7 +23,7 @@ and largely have the same functions.
|
|||||||
if __name__ == '__main__': # To ensure correct behavior on Windows and macOS
|
if __name__ == '__main__': # To ensure correct behavior on Windows and macOS
|
||||||
ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True)
|
ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True)
|
||||||
|
|
||||||
With a few exceptions, all of the command line arguments are available
|
With some exceptions, all of the command line arguments are available
|
||||||
and may be passed as equivalent keywords.
|
and may be passed as equivalent keywords.
|
||||||
|
|
||||||
A few differences are that ``verbose`` and ``quiet`` are not available.
|
A few differences are that ``verbose`` and ``quiet`` are not available.
|
||||||
@@ -41,33 +41,41 @@ execution. To do this, it will:
|
|||||||
- manage the signal flags of its worker processes
|
- manage the signal flags of its worker processes
|
||||||
- execute other subprocesses (forking and executing other programs)
|
- execute other subprocesses (forking and executing other programs)
|
||||||
|
|
||||||
The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently
|
The Python process that calls :func:`ocrmypdf.ocr()` must be sufficiently
|
||||||
privileged to perform these actions.
|
privileged to perform these actions.
|
||||||
|
|
||||||
There currently is no option to manage how jobs are scheduled other
|
There currently is no option to manage how jobs are scheduled other
|
||||||
than the argument ``jobs=`` which will limit the number of worker
|
than the argument ``jobs=`` which will limit the number of worker
|
||||||
processes.
|
processes.
|
||||||
|
|
||||||
Creating a child process to call ``ocrmypdf.ocr()`` is suggested. That
|
Creating a child process to call :func:`ocrmypdf.ocr()` is suggested. That
|
||||||
way your application will survive and remain interactive even if
|
way your application will survive and remain interactive even if
|
||||||
OCRmyPDF fails for any reason.
|
OCRmyPDF fails for any reason.
|
||||||
|
|
||||||
Programs that call ``ocrmypdf.ocr()`` should also install a SIGBUS signal
|
Programs that call :func:`ocrmypdf.ocr()` should also install a SIGBUS signal
|
||||||
handler (except on Windows), to raise an exception if access to a memory
|
handler (except on Windows), to raise an exception if access to a memory
|
||||||
mapped file fails. OCRmyPDF may use memory mapping.
|
mapped file fails. OCRmyPDF may use memory mapping.
|
||||||
|
|
||||||
``ocrmypdf.ocr()`` will take a threading lock to prevent multiple runs of itself
|
:func:`ocrmypdf.ocr()` will take a threading lock to prevent multiple runs of itself
|
||||||
in the same Python interpreter process. This is not thread-safe, because of how
|
in the same Python interpreter process. This is not thread-safe, because of how
|
||||||
OCRmyPDF's plugins and Python's library import system work. If you need to parallelize
|
OCRmyPDF's plugins and Python's library import system work. If you need to parallelize
|
||||||
OCRmyPDF, use processes.
|
OCRmyPDF, use processes.
|
||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
|
|
||||||
On Windows and macOS, the script that calls ``ocrmypdf.ocr()`` must be
|
On Windows and macOS, the script that calls :func:`ocrmypdf.ocr()` must be
|
||||||
protected by an "ifmain" guard (``if __name__ == '__main__'``). If you do
|
protected by an "ifmain" guard (``if __name__ == '__main__'``). If you do
|
||||||
not take at least one of these steps, process semantics will prevent
|
not take at least one of these steps, process semantics will prevent
|
||||||
OCRmyPDF from working correctly.
|
OCRmyPDF from working correctly.
|
||||||
|
|
||||||
|
.. warning::
|
||||||
|
|
||||||
|
On macOS with Python 3.7, you must call
|
||||||
|
:func:`multiprocessing.set_start_method("spawn")`. Without this, multiprocessing
|
||||||
|
will be unstable. From the command line, OCRmyPDF does this automatically,
|
||||||
|
but as an API user you must do this. See Python bpo-33725 for details.
|
||||||
|
Python 3.8+ also resolve this automatically.
|
||||||
|
|
||||||
Logging
|
Logging
|
||||||
-------
|
-------
|
||||||
|
|
||||||
@@ -96,7 +104,7 @@ Exceptions
|
|||||||
|
|
||||||
OCRmyPDF may throw standard Python exceptions, ``ocrmypdf.exceptions.*``
|
OCRmyPDF may throw standard Python exceptions, ``ocrmypdf.exceptions.*``
|
||||||
exceptions, some exceptions related to multiprocessing, and
|
exceptions, some exceptions related to multiprocessing, and
|
||||||
``KeyboardInterrupt``. The parent process should provide an exception
|
:exc:`KeyboardInterrupt`. The parent process should provide an exception
|
||||||
handler. OCRmyPDF will clean up its temporary files and worker processes
|
handler. OCRmyPDF will clean up its temporary files and worker processes
|
||||||
automatically when an exception occurs.
|
automatically when an exception occurs.
|
||||||
|
|
||||||
|
|||||||
+20
-12
@@ -36,18 +36,21 @@ Directory trees
|
|||||||
===============
|
===============
|
||||||
|
|
||||||
This will walk through a directory tree and run OCR on all files in
|
This will walk through a directory tree and run OCR on all files in
|
||||||
place, printing the output in a way that makes
|
place, and printing each filename in between runs:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
find . -printf '%p' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
|
find . -printf '%p\n' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
|
||||||
|
|
||||||
Alternatively, with a docker container (mounts a volume to the container
|
Alternatively, with a Docker container and streaming the file through
|
||||||
where the PDFs are stored):
|
standard input and output:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
find . -printf '%p' -name '*.pdf' -exec docker run --rm -v <host dir>:<container dir> jbarlow83/ocrmypdf '<container dir>/{}' '<container dir>/{}' \;
|
find . -name '*.pdf' -print0 | xargs -0 | while read pdf; do
|
||||||
|
pdfout=$(mktemp)
|
||||||
|
docker run --rm -i jbarlow83/ocrmypdf - - <$pdf >$pdfout && cp $pdfout $pdf
|
||||||
|
done
|
||||||
|
|
||||||
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
||||||
``find`` to create a directory list and ``parallel`` to parallelize runs
|
``find`` to create a directory list and ``parallel`` to parallelize runs
|
||||||
@@ -124,7 +127,9 @@ Users may need to customize the script to meet their requirements.
|
|||||||
|
|
||||||
"OCR_INPUT_DIRECTORY", "Set input directory to monitor (recursive)"
|
"OCR_INPUT_DIRECTORY", "Set input directory to monitor (recursive)"
|
||||||
"OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)"
|
"OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)"
|
||||||
|
"OCR_ARCHIVE_DIRECTORY", "Set archive directory for processed originals (should not be under input, requires ``OCR_ON_SUCCESS_ARCHIVE`` to be set)"
|
||||||
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
|
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
|
||||||
|
"OCR_ON_SUCCESS_ARCHIVE", "This will move the processed orignal file to ``OCR_ARCHIVE_DIRECTORY`` if the exit code is 0 (OK). Note that ``OCR_ON_SUCCESS_DELETE`` takes precedence over this option, i.e. if both options are set, the input file will be deleted."
|
||||||
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||||
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
||||||
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
|
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
|
||||||
@@ -144,16 +149,18 @@ The watcher service is included in the OCRmyPDF Docker image. To run it:
|
|||||||
docker run \
|
docker run \
|
||||||
-v <path to files to convert>:/input \
|
-v <path to files to convert>:/input \
|
||||||
-v <path to store results>:/output \
|
-v <path to store results>:/output \
|
||||||
|
-v <path to store processed originals>:/archive \
|
||||||
-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
||||||
-e OCR_ON_SUCCESS_DELETE=1 \
|
-e OCR_ON_SUCCESS_ARCHIVE=1 \
|
||||||
-e OCR_DESKEW=1 \
|
-e OCR_DESKEW=1 \
|
||||||
-e PYTHONUNBUFFERED=1 \
|
-e PYTHONUNBUFFERED=1 \
|
||||||
-it --entrypoint python3 \
|
-it --entrypoint python3 \
|
||||||
jbarlow83/ocrmypdf \
|
jbarlow83/ocrmypdf \
|
||||||
watcher.py
|
watcher.py
|
||||||
|
|
||||||
This service will watch for a file that matches ``/input/\*.pdf`` and will
|
This service will watch for a file that matches ``/input/\*.pdf``,
|
||||||
convert it to a OCRed PDF in ``/output/``. The parameters to this image are:
|
convert it to a OCRed PDF in ``/output/``, and move the processed
|
||||||
|
original to ``/archive``. The parameters to this image are:
|
||||||
|
|
||||||
.. csv-table:: watcher.py parameters for Docker
|
.. csv-table:: watcher.py parameters for Docker
|
||||||
:header: "Parameter", "Description"
|
:header: "Parameter", "Description"
|
||||||
@@ -161,10 +168,11 @@ convert it to a OCRed PDF in ``/output/``. The parameters to this image are:
|
|||||||
|
|
||||||
"``-v <path to files to convert>:/input``", "Files placed in this location will be OCRed"
|
"``-v <path to files to convert>:/input``", "Files placed in this location will be OCRed"
|
||||||
"``-v <path to store results>:/output``", "This is where OCRed files will be stored"
|
"``-v <path to store results>:/output``", "This is where OCRed files will be stored"
|
||||||
"``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1"
|
"``-v <path to store processed originals>:/archive``", "Archive processed originals here"
|
||||||
"``-e OCR_ON_SUCCESS_DELETE=1``", "Define environment variable"
|
"``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||||
"``-e OCR_DESKEW=1``", "Define environment variable"
|
"``-e OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals"
|
||||||
"``-e PYTHONBUFFERED=1``", "This will force STDOUT to be unbuffered and allow you to see messages in docker logs"
|
"``-e OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs"
|
||||||
|
"``-e PYTHONBUFFERED=1``", "This will force ``STDOUT`` to be unbuffered and allow you to see messages in docker logs"
|
||||||
|
|
||||||
This service relies on polling to check for changes to the filesystem. It
|
This service relies on polling to check for changes to the filesystem. It
|
||||||
may not be suitable for some environments, such as filesystems shared on a
|
may not be suitable for some environments, such as filesystems shared on a
|
||||||
|
|||||||
+16
-8
@@ -31,11 +31,18 @@
|
|||||||
# Add any Sphinx extension module names here, as strings. They can be
|
# Add any Sphinx extension module names here, as strings. They can be
|
||||||
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
||||||
# ones.
|
# ones.
|
||||||
extensions = ['sphinx.ext.napoleon', 'sphinx_issues']
|
extensions = [
|
||||||
|
'sphinx.ext.autodoc',
|
||||||
|
'sphinx.ext.intersphinx',
|
||||||
|
'sphinx.ext.autosummary',
|
||||||
|
'sphinx.ext.napoleon',
|
||||||
|
'sphinx_issues',
|
||||||
|
]
|
||||||
|
|
||||||
# Extension settings
|
# Extension settings
|
||||||
|
intersphinx_mapping = {'https://docs.python.org/': None}
|
||||||
napoleon_use_rtype = False
|
napoleon_use_rtype = False
|
||||||
issues_github_path = "jbarlow83/OCRmyPDF"
|
issues_github_path = "ocrmypdf/OCRmyPDF"
|
||||||
|
|
||||||
# Add any paths that contain templates here, relative to this directory.
|
# Add any paths that contain templates here, relative to this directory.
|
||||||
templates_path = ['_templates']
|
templates_path = ['_templates']
|
||||||
@@ -56,7 +63,7 @@ master_doc = 'index'
|
|||||||
# General information about the project.
|
# General information about the project.
|
||||||
project = 'ocrmypdf'
|
project = 'ocrmypdf'
|
||||||
copyright = (
|
copyright = (
|
||||||
'2020, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.'
|
'2022, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.'
|
||||||
)
|
)
|
||||||
author = 'James R. Barlow'
|
author = 'James R. Barlow'
|
||||||
|
|
||||||
@@ -84,15 +91,16 @@ if on_rtd:
|
|||||||
'pikepdf',
|
'pikepdf',
|
||||||
'pikepdf.models',
|
'pikepdf.models',
|
||||||
'pikepdf.models.metadata',
|
'pikepdf.models.metadata',
|
||||||
'ocrmypdf.leptonica',
|
|
||||||
]
|
]
|
||||||
sys.modules.update((mod_name, Mock()) for mod_name in MOCK_MODULES)
|
sys.modules.update((mod_name, Mock()) for mod_name in MOCK_MODULES)
|
||||||
|
|
||||||
|
try:
|
||||||
from pkg_resources import get_distribution, DistributionNotFound
|
from importlib_metadata import version as package_version
|
||||||
|
except ModuleNotFoundError:
|
||||||
|
from importlib.metadata import version as package_version
|
||||||
|
|
||||||
# The full version, including alpha/beta/rc tags.
|
# The full version, including alpha/beta/rc tags.
|
||||||
release = get_distribution('ocrmypdf').version
|
release = package_version('ocrmypdf')
|
||||||
version = '.'.join(release.split('.')[:2])
|
version = '.'.join(release.split('.')[:2])
|
||||||
|
|
||||||
|
|
||||||
@@ -275,7 +283,7 @@ htmlhelp_basename = 'ocrmypdfdoc'
|
|||||||
|
|
||||||
# -- Options for LaTeX output ---------------------------------------------
|
# -- Options for LaTeX output ---------------------------------------------
|
||||||
|
|
||||||
latex_elements = {
|
latex_elements = { # type: ignore
|
||||||
# The paper size ('letterpaper' or 'a4paper').
|
# The paper size ('letterpaper' or 'a4paper').
|
||||||
#
|
#
|
||||||
# 'papersize': 'letterpaper',
|
# 'papersize': 'letterpaper',
|
||||||
|
|||||||
@@ -46,17 +46,6 @@ Style guide: Is it OCRmyPDF or ocrmypdf?
|
|||||||
|
|
||||||
The program/project is OCRmyPDF and the name of the executable or library is ocrmypdf.
|
The program/project is OCRmyPDF and the name of the executable or library is ocrmypdf.
|
||||||
|
|
||||||
Known ports/packagers
|
|
||||||
=====================
|
|
||||||
|
|
||||||
OCRmyPDF has been ported to many platforms already. If you are interesting in
|
|
||||||
porting to a new platform, check with
|
|
||||||
`Repology <https://repology.org/projects/?search=ocrmypdf>`__ to see the status
|
|
||||||
of that platform.
|
|
||||||
|
|
||||||
Packager maintainers, please ensure that the command line completion scripts in
|
|
||||||
``misc/`` are installed.
|
|
||||||
|
|
||||||
Copyright and license
|
Copyright and license
|
||||||
=====================
|
=====================
|
||||||
|
|
||||||
|
|||||||
+21
-12
@@ -104,6 +104,9 @@ This produces a file named "output.pdf" and a companion text file named
|
|||||||
because of options like ``--skip-big`` or ``--tesseract-timeout``, those pages
|
because of options like ``--skip-big`` or ``--tesseract-timeout``, those pages
|
||||||
will not be in the sidecar.
|
will not be in the sidecar.
|
||||||
|
|
||||||
|
If you don't want to generate the output PDF, use ``--output-type=none`` to
|
||||||
|
avoid generating one. Set the output filename to ``-`` (i.e. redirect to stdout).
|
||||||
|
|
||||||
To extract all text from a PDF, whether generated from OCR or otherwise,
|
To extract all text from a PDF, whether generated from OCR or otherwise,
|
||||||
use a program like Poppler's ``pdftotext`` or ``pdfgrep``.
|
use a program like Poppler's ``pdftotext`` or ``pdfgrep``.
|
||||||
|
|
||||||
@@ -181,10 +184,7 @@ might remove desirable content, especially from poor quality scans.
|
|||||||
ignored. This should not be used on documents that contain color
|
ignored. This should not be used on documents that contain color
|
||||||
photos as it may remove them.
|
photos as it may remove them.
|
||||||
- ``--deskew`` will correct pages were scanned at a skewed angle by
|
- ``--deskew`` will correct pages were scanned at a skewed angle by
|
||||||
rotating them back into place. Skew determination and correction is
|
rotating them back into place.
|
||||||
performed using `Postl's variance of line
|
|
||||||
sums <http://www.leptonica.org/skew-measurement.html>`__ algorithm as
|
|
||||||
implemented in `Leptonica <http://www.leptonica.org/index.html>`__.
|
|
||||||
- ``--clean`` uses
|
- ``--clean`` uses
|
||||||
`unpaper <https://www.flameeyes.eu/projects/unpaper>`__ to clean up
|
`unpaper <https://www.flameeyes.eu/projects/unpaper>`__ to clean up
|
||||||
pages before OCR, but does not alter the final output. This makes it
|
pages before OCR, but does not alter the final output. This makes it
|
||||||
@@ -200,7 +200,7 @@ might remove desirable content, especially from poor quality scans.
|
|||||||
|
|
||||||
.. warning::
|
.. warning::
|
||||||
|
|
||||||
``--clean-final`` and ``-remove-background`` may leave undesirable
|
``--clean-final`` and ``--remove-background`` may leave undesirable
|
||||||
visual artifacts in some images where their algorithms have
|
visual artifacts in some images where their algorithms have
|
||||||
shortcomings. Files should be visually reviewed after using these
|
shortcomings. Files should be visually reviewed after using these
|
||||||
options.
|
options.
|
||||||
@@ -243,10 +243,11 @@ You can also optimize all images without performing any OCR:
|
|||||||
|
|
||||||
ocrmypdf --tesseract-timeout=0 --optimize 3 --skip-text input.pdf output.pdf
|
ocrmypdf --tesseract-timeout=0 --optimize 3 --skip-text input.pdf output.pdf
|
||||||
|
|
||||||
Perform OCR only certain pages
|
Process only certain pages
|
||||||
------------------------------
|
--------------------------
|
||||||
|
|
||||||
You can ask OCRmyPDF to only apply OCR to certain pages.
|
You can ask OCRmyPDF to only apply `image processing <#image-processing>`__
|
||||||
|
and OCR to certain pages.
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -260,10 +261,10 @@ overlap pages. OCRmyPDF does not currently account for document page numbers,
|
|||||||
such as an introduction section of a book that uses Roman numerals. It simply
|
such as an introduction section of a book that uses Roman numerals. It simply
|
||||||
counts the number of virtual pieces of paper since the start.
|
counts the number of virtual pieces of paper since the start.
|
||||||
|
|
||||||
Regardless of the argument to ``--pages``, OCRmyPDF will optimize all pages in
|
Regardless of the argument to ``--pages``, OCRmyPDF will optimize all pages/images
|
||||||
the file and convert it to PDF/A, unless you disable those options. In this
|
in the file and convert it to PDF/A, unless you disable those options. Both of these
|
||||||
example, we want to OCR only the title and otherwise change the PDF as little
|
steps are "whole file" operations. In this example, we want to OCR only the title
|
||||||
as possible:
|
and otherwise change the PDF as little as possible:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -339,6 +340,9 @@ levels in the GCC compiler.
|
|||||||
- Enables lossless optimizations, such as transcoding images to more
|
- Enables lossless optimizations, such as transcoding images to more
|
||||||
efficient formats. Also compress other uncompressed objects in the
|
efficient formats. Also compress other uncompressed objects in the
|
||||||
PDF and enables the more efficient "object streams" within the PDF.
|
PDF and enables the more efficient "object streams" within the PDF.
|
||||||
|
(If ``--jbig2-lossy`` is issued, then lossy JBIG2 optimization is used.
|
||||||
|
The decision to use lossy JBIG2 is separate from standard optimization
|
||||||
|
settings.)
|
||||||
* - ``--optimize 2``
|
* - ``--optimize 2``
|
||||||
- All of the above, and enables lossy optimizations and color quantization.
|
- All of the above, and enables lossy optimizations and color quantization.
|
||||||
* - ``--optimize 3``
|
* - ``--optimize 3``
|
||||||
@@ -359,3 +363,8 @@ fo a PDF.
|
|||||||
ocrmypdf --optimize 3 in.pdf out.pdf # Make it small
|
ocrmypdf --optimize 3 in.pdf out.pdf # Make it small
|
||||||
|
|
||||||
Some users may consider enabling lossy JBIG2. See: :ref:`jbig2-lossy`.
|
Some users may consider enabling lossy JBIG2. See: :ref:`jbig2-lossy`.
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
|
||||||
|
Image processing and PDF/A conversion can also introduce lossy transformations
|
||||||
|
to your PDF images, even when ``--optimize 1`` is in use.
|
||||||
|
|||||||
+1
-1
@@ -59,7 +59,7 @@ Using the Docker image on the command line
|
|||||||
==========================================
|
==========================================
|
||||||
|
|
||||||
**Unlike typical Docker containers**, in this section the OCRmyPDF Docker
|
**Unlike typical Docker containers**, in this section the OCRmyPDF Docker
|
||||||
container is emphemeral – it runs for one OCR job and terminates, just like a
|
container is ephemeral – it runs for one OCR job and terminates, just like a
|
||||||
command line program. We are using Docker to deliver an application (as opposed
|
command line program. We are using Docker to deliver an application (as opposed
|
||||||
to the more conventional case, where a Docker container runs as a server).
|
to the more conventional case, where a Docker container runs as a server).
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,239 @@
|
|||||||
|
<?xml version="1.0" encoding="UTF-8" standalone="no"?>
|
||||||
|
<svg
|
||||||
|
width="256"
|
||||||
|
height="256"
|
||||||
|
viewBox="0 0 256 256.00001"
|
||||||
|
version="1.1"
|
||||||
|
xml:space="preserve"
|
||||||
|
style="clip-rule:evenodd;fill-rule:evenodd;stroke-linecap:round;stroke-linejoin:round;stroke-miterlimit:1.5"
|
||||||
|
id="svg270"
|
||||||
|
sodipodi:docname="logo-square-256.svg"
|
||||||
|
inkscape:export-filename="/home/jb/src/ocrmypdf/docs/images/logo-square.png"
|
||||||
|
inkscape:export-xdpi="96"
|
||||||
|
inkscape:export-ydpi="96"
|
||||||
|
inkscape:version="1.1.2 (0a00cf5339, 2022-02-04)"
|
||||||
|
xmlns:inkscape="http://www.inkscape.org/namespaces/inkscape"
|
||||||
|
xmlns:sodipodi="http://sodipodi.sourceforge.net/DTD/sodipodi-0.dtd"
|
||||||
|
xmlns="http://www.w3.org/2000/svg"
|
||||||
|
xmlns:svg="http://www.w3.org/2000/svg"
|
||||||
|
xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#"
|
||||||
|
xmlns:cc="http://creativecommons.org/ns#"
|
||||||
|
xmlns:dc="http://purl.org/dc/elements/1.1/"
|
||||||
|
xmlns:serif="http://www.serif.com/"><metadata
|
||||||
|
id="metadata276"><rdf:RDF><cc:Work
|
||||||
|
rdf:about=""><dc:format>image/svg+xml</dc:format><dc:type
|
||||||
|
rdf:resource="http://purl.org/dc/dcmitype/StillImage" /></cc:Work></rdf:RDF></metadata><defs
|
||||||
|
id="defs274" /><sodipodi:namedview
|
||||||
|
pagecolor="#ffffff"
|
||||||
|
bordercolor="#666666"
|
||||||
|
borderopacity="1"
|
||||||
|
objecttolerance="10"
|
||||||
|
gridtolerance="10"
|
||||||
|
guidetolerance="10"
|
||||||
|
inkscape:pageopacity="0"
|
||||||
|
inkscape:pageshadow="2"
|
||||||
|
inkscape:window-width="2396"
|
||||||
|
inkscape:window-height="1691"
|
||||||
|
id="namedview272"
|
||||||
|
showgrid="false"
|
||||||
|
lock-margins="false"
|
||||||
|
inkscape:zoom="2.0079523"
|
||||||
|
inkscape:cx="189.74554"
|
||||||
|
inkscape:cy="54.533168"
|
||||||
|
inkscape:window-x="26"
|
||||||
|
inkscape:window-y="23"
|
||||||
|
inkscape:window-maximized="0"
|
||||||
|
inkscape:current-layer="svg270"
|
||||||
|
inkscape:pagecheckerboard="0"
|
||||||
|
width="256px"
|
||||||
|
fit-margin-top="0"
|
||||||
|
fit-margin-left="0"
|
||||||
|
fit-margin-right="0"
|
||||||
|
fit-margin-bottom="0" />
|
||||||
|
<g
|
||||||
|
id="svg"
|
||||||
|
transform="matrix(0.48534351,0,0,0.4057699,1.8106874,71.192214)">
|
||||||
|
<rect
|
||||||
|
x="0"
|
||||||
|
y="0"
|
||||||
|
width="520"
|
||||||
|
height="280"
|
||||||
|
style="fill:#ffffff"
|
||||||
|
id="rect188" />
|
||||||
|
<g
|
||||||
|
transform="matrix(1.03522,0,0,1.23823,-69.7528,-83.422)"
|
||||||
|
id="g267">
|
||||||
|
<g
|
||||||
|
transform="translate(243.977,20.0703)"
|
||||||
|
id="g218">
|
||||||
|
<g
|
||||||
|
id="Page">
|
||||||
|
<g
|
||||||
|
transform="matrix(0.961773,0,0,1.05962,6.19811,-3.01071)"
|
||||||
|
id="g192">
|
||||||
|
<path
|
||||||
|
d="m 328.5,97.682 c 0,-1.217 -0.517,-2.386 -1.444,-3.264 -7.03,-6.66 -37.614,-35.638 -44.828,-42.474 -0.977,-0.925 -2.327,-1.448 -3.738,-1.448 -13.997,0 -90.407,0 -111.151,0 -2.871,0 -5.198,2.113 -5.198,4.718 0,27.837 0,170.351 0,198.186 0,2.605 2.327,4.717 5.197,4.717 24.904,0 131.821,0 156.2,0 2.74,0 4.962,-2.016 4.962,-4.504 0,-24.345 0,-139.717 0,-155.931 z"
|
||||||
|
style="fill:#fdfdfd;stroke:#333333;stroke-width:3.95px"
|
||||||
|
id="path190" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
id="Dog-ear"
|
||||||
|
serif:id="Dog ear"
|
||||||
|
transform="translate(-4,2)">
|
||||||
|
<path
|
||||||
|
d="m 277.072,48.496 v 45.352 c 0,1.324 0.526,2.593 1.462,3.529 0.936,0.936 2.205,1.462 3.529,1.462 12.485,0 44.078,0 44.078,0"
|
||||||
|
style="fill:#f5f5f5;stroke:#333333;stroke-width:4px"
|
||||||
|
id="path194" />
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="translate(-29.6816,-0.395178)"
|
||||||
|
id="g216">
|
||||||
|
<g
|
||||||
|
transform="matrix(1.00243,0,0,1.11818,-144.72,-8.80181)"
|
||||||
|
id="g200">
|
||||||
|
<path
|
||||||
|
d="m 465.73,119.654 c 0,-2.049 -1.856,-3.713 -4.142,-3.713 H 310.259 c -2.286,0 -4.142,1.664 -4.142,3.713 v 63.454 c 0,2.049 1.856,3.713 4.142,3.713 h 151.329 c 2.286,0 4.142,-1.664 4.142,-3.713 z"
|
||||||
|
style="fill:#f80000;stroke:#ffffff;stroke-width:3.77px"
|
||||||
|
id="path198" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(1.24571,0,0,1.35864,116.812,84.3924)"
|
||||||
|
id="g214">
|
||||||
|
<g
|
||||||
|
transform="matrix(64,0,0,64,42.1437,77.6203)"
|
||||||
|
id="g204">
|
||||||
|
<path
|
||||||
|
d="m 0.084,0 v -0.68 h 0.213 c 0.074,0 0.137,0.017 0.19,0.05 0.053,0.034 0.079,0.09 0.079,0.168 0,0.077 -0.028,0.134 -0.085,0.17 -0.057,0.037 -0.121,0.055 -0.193,0.055 H 0.213 V 0 Z m 0.209,-0.572 h -0.08 v 0.228 h 0.082 c 0.039,0 0.07,-0.009 0.094,-0.027 0.024,-0.017 0.037,-0.045 0.04,-0.083 0,-0.044 -0.012,-0.075 -0.036,-0.092 -0.024,-0.017 -0.057,-0.026 -0.1,-0.026 z"
|
||||||
|
style="fill:#ffffff;fill-rule:nonzero"
|
||||||
|
id="path202" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(64,0,0,64,79.7117,77.6203)"
|
||||||
|
id="g208">
|
||||||
|
<path
|
||||||
|
d="M 0.332,0 H 0.084 v -0.68 h 0.252 c 0.105,0 0.182,0.032 0.233,0.095 0.051,0.063 0.076,0.144 0.076,0.241 0,0.105 -0.027,0.189 -0.082,0.251 C 0.508,-0.031 0.431,0 0.332,0 Z M 0.337,-0.57 H 0.213 v 0.461 H 0.33 c 0.055,0 0.099,-0.018 0.132,-0.054 C 0.495,-0.199 0.511,-0.259 0.511,-0.344 0.511,-0.415 0.497,-0.47 0.469,-0.51 0.441,-0.55 0.397,-0.57 0.337,-0.57 Z"
|
||||||
|
style="fill:#ffffff;fill-rule:nonzero"
|
||||||
|
id="path206" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(64,0,0,64,123.424,77.6203)"
|
||||||
|
id="g212">
|
||||||
|
<path
|
||||||
|
d="M 0.405,-0.288 H 0.213 V 0 H 0.084 v -0.68 h 0.385 l 0.02,0.102 H 0.213 v 0.189 h 0.173 z"
|
||||||
|
style="fill:#ffffff;fill-rule:nonzero"
|
||||||
|
id="path210" />
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(1,0,0,1.52217,67.3796,10.7507)"
|
||||||
|
id="g222">
|
||||||
|
<rect
|
||||||
|
x="23.500999"
|
||||||
|
y="81.300003"
|
||||||
|
width="162.30499"
|
||||||
|
height="61.77"
|
||||||
|
style="fill:#b4d5ff"
|
||||||
|
id="rect220" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(0.967536,0,0,0.961535,5.90498,47.9703)"
|
||||||
|
id="g236">
|
||||||
|
<g
|
||||||
|
transform="matrix(90.4804,0,0,90.4804,82.6698,167.705)"
|
||||||
|
id="g226">
|
||||||
|
<path
|
||||||
|
d="m 0.057,-0.337 c 0,-0.105 0.027,-0.19 0.082,-0.257 0.055,-0.066 0.132,-0.1 0.231,-0.102 0.107,0 0.186,0.034 0.237,0.103 0.051,0.069 0.077,0.152 0.077,0.249 0,0.105 -0.027,0.191 -0.082,0.258 -0.055,0.067 -0.133,0.1 -0.232,0.1 C 0.264,0.014 0.185,-0.02 0.134,-0.089 0.083,-0.157 0.057,-0.24 0.057,-0.337 Z m 0.135,-0.001 c 0,0.071 0.014,0.13 0.043,0.175 0.029,0.045 0.073,0.068 0.134,0.068 0.055,0 0.098,-0.02 0.131,-0.061 0.033,-0.041 0.049,-0.103 0.049,-0.188 0,-0.071 -0.014,-0.129 -0.043,-0.174 -0.029,-0.045 -0.073,-0.068 -0.134,-0.068 -0.053,0 -0.097,0.022 -0.13,0.067 -0.033,0.045 -0.05,0.105 -0.05,0.181 z"
|
||||||
|
style="fill:#333333;fill-rule:nonzero"
|
||||||
|
id="path224" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(90.4804,0,0,90.4804,147.906,167.705)"
|
||||||
|
id="g230">
|
||||||
|
<path
|
||||||
|
d="M 0.505,-0.557 C 0.473,-0.567 0.448,-0.574 0.429,-0.579 0.41,-0.583 0.388,-0.585 0.361,-0.585 c -0.054,0 -0.096,0.022 -0.125,0.066 -0.029,0.044 -0.044,0.104 -0.044,0.181 0,0.066 0.012,0.123 0.037,0.171 0.025,0.048 0.066,0.072 0.124,0.072 0.029,0 0.056,-0.003 0.081,-0.009 0.025,-0.006 0.047,-0.013 0.068,-0.022 L 0.551,-0.03 C 0.525,-0.017 0.494,-0.006 0.457,0.002 0.42,0.01 0.388,0.014 0.36,0.014 0.254,0.014 0.177,-0.02 0.129,-0.088 0.081,-0.156 0.057,-0.239 0.057,-0.337 c 0,-0.105 0.027,-0.19 0.08,-0.257 0.053,-0.067 0.129,-0.1 0.228,-0.1 0.02,0 0.048,0.003 0.083,0.01 0.035,0.007 0.068,0.018 0.097,0.034 z"
|
||||||
|
style="fill:#333333;fill-rule:nonzero"
|
||||||
|
id="path228" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(90.4804,0,0,90.4804,199.751,167.705)"
|
||||||
|
id="g234">
|
||||||
|
<path
|
||||||
|
d="m 0.293,-0.572 h -0.08 v 0.208 h 0.082 c 0.039,0 0.071,-0.008 0.096,-0.024 0.025,-0.015 0.038,-0.041 0.038,-0.077 0,-0.038 -0.012,-0.065 -0.036,-0.082 -0.024,-0.017 -0.057,-0.025 -0.1,-0.025 z M 0.479,0 0.335,-0.26 C 0.328,-0.259 0.32,-0.259 0.312,-0.259 0.304,-0.258 0.296,-0.258 0.288,-0.258 H 0.213 V 0 H 0.084 v -0.68 h 0.213 c 0.074,0 0.137,0.017 0.19,0.051 0.053,0.034 0.079,0.087 0.079,0.158 0,0.042 -0.011,0.078 -0.032,0.108 -0.022,0.031 -0.05,0.054 -0.084,0.071 L 0.617,0 Z"
|
||||||
|
style="fill:#333333;fill-rule:nonzero"
|
||||||
|
id="path232" />
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(0.916882,0,0,1,121.475,-32.6535)"
|
||||||
|
id="g246">
|
||||||
|
<g
|
||||||
|
transform="matrix(86.953,0,0,86.953,152.996,241.878)"
|
||||||
|
id="g240">
|
||||||
|
<path
|
||||||
|
d="M 0.479,-0.428 C 0.5,-0.451 0.527,-0.47 0.562,-0.484 c 0.034,-0.013 0.065,-0.02 0.092,-0.02 0.066,0 0.113,0.019 0.141,0.058 0.027,0.039 0.041,0.086 0.041,0.142 V 0 H 0.705 v -0.298 c 0,-0.031 -0.007,-0.054 -0.022,-0.071 -0.015,-0.016 -0.036,-0.024 -0.064,-0.024 -0.019,0 -0.038,0.005 -0.059,0.015 -0.021,0.01 -0.039,0.021 -0.056,0.034 0.001,0.007 0.001,0.013 0.002,0.02 0.001,0.007 0.001,0.013 0.001,0.02 V 0 H 0.376 v -0.298 c 0,-0.031 -0.007,-0.054 -0.022,-0.071 -0.015,-0.016 -0.036,-0.024 -0.063,-0.024 -0.017,0 -0.033,0.003 -0.05,0.01 -0.017,0.007 -0.034,0.016 -0.049,0.027 V 0 H 0.062 V -0.485 H 0.13 l 0.032,0.044 c 0.022,-0.02 0.049,-0.035 0.08,-0.047 0.031,-0.011 0.058,-0.016 0.083,-0.016 0.038,0 0.07,0.007 0.095,0.02 0.025,0.014 0.045,0.033 0.059,0.056 z"
|
||||||
|
style="fill:#333333;fill-rule:nonzero"
|
||||||
|
id="path238" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(86.953,0,0,86.953,228.906,241.878)"
|
||||||
|
id="g244">
|
||||||
|
<path
|
||||||
|
d="M 0.156,0.023 0.179,-0.034 0.006,-0.467 0.14,-0.485 0.252,-0.191 0.358,-0.485 H 0.495 L 0.278,0.064 C 0.263,0.103 0.236,0.137 0.197,0.165 0.158,0.193 0.118,0.212 0.075,0.222 L 0.029,0.115 C 0.052,0.105 0.077,0.093 0.104,0.079 0.13,0.064 0.147,0.046 0.156,0.023 Z"
|
||||||
|
style="fill:#333333;fill-rule:nonzero"
|
||||||
|
id="path242" />
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
id="Selectors"
|
||||||
|
transform="matrix(0.965977,0,0,0.807602,67.3796,67.3718)">
|
||||||
|
<g
|
||||||
|
id="Right-selector"
|
||||||
|
serif:id="Right selector">
|
||||||
|
<g
|
||||||
|
transform="matrix(1.03522,0,0,1.23823,2.07044,0)"
|
||||||
|
id="g250">
|
||||||
|
<path
|
||||||
|
d="M 185.806,161.156 V 67.132"
|
||||||
|
style="fill:none;stroke:#4c9fff;stroke-width:4px;stroke-linecap:butt"
|
||||||
|
id="path248" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(1.03522,0,0,1.23823,161.788,169.469)"
|
||||||
|
id="g254">
|
||||||
|
<circle
|
||||||
|
cx="31.523001"
|
||||||
|
cy="34.313999"
|
||||||
|
r="10.021"
|
||||||
|
style="fill:#4c9fff;stroke:#4c9fff;stroke-width:4px;stroke-linecap:butt"
|
||||||
|
id="circle252" />
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
id="Left-selector"
|
||||||
|
serif:id="Left selector">
|
||||||
|
<g
|
||||||
|
transform="matrix(1.03522,0,0,1.23823,-170.092,0)"
|
||||||
|
id="g259">
|
||||||
|
<path
|
||||||
|
d="M 185.806,161.156 V 67.132"
|
||||||
|
style="fill:none;stroke:#4c9fff;stroke-width:4px;stroke-linecap:butt"
|
||||||
|
id="path257" />
|
||||||
|
</g>
|
||||||
|
<g
|
||||||
|
transform="matrix(1.03522,0,0,1.23823,-10.3742,28.2274)"
|
||||||
|
id="g263">
|
||||||
|
<circle
|
||||||
|
cx="31.523001"
|
||||||
|
cy="34.313999"
|
||||||
|
r="10.021"
|
||||||
|
style="fill:#4c9fff;stroke:#4c9fff;stroke-width:4px;stroke-linecap:butt"
|
||||||
|
id="circle261" />
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
</g>
|
||||||
|
</svg>
|
||||||
|
After Width: | Height: | Size: 11 KiB |
@@ -1,6 +1,8 @@
|
|||||||
OCRmyPDF documentation
|
OCRmyPDF documentation
|
||||||
======================
|
======================
|
||||||
|
|
||||||
|
.. figure:: images/logo.svg
|
||||||
|
|
||||||
OCRmyPDF adds an optical character recognition (OCR) text layer to scanned PDF
|
OCRmyPDF adds an optical character recognition (OCR) text layer to scanned PDF
|
||||||
files, allowing them to be searched.
|
files, allowing them to be searched.
|
||||||
|
|
||||||
@@ -38,6 +40,7 @@ image processing and OCR to existing PDFs.
|
|||||||
plugins
|
plugins
|
||||||
apiref
|
apiref
|
||||||
contributing
|
contributing
|
||||||
|
maintainers
|
||||||
|
|
||||||
Indices and tables
|
Indices and tables
|
||||||
==================
|
==================
|
||||||
|
|||||||
+76
-183
@@ -12,21 +12,23 @@ system/platform. This version may be out of date, however.
|
|||||||
|
|
||||||
These platforms have one-liner installs:
|
These platforms have one-liner installs:
|
||||||
|
|
||||||
+-------------------------------+-------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||||
+-------------------------------+-------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||||
+-------------------------------+-------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| Fedora | ``dnf install ocrmypdf`` |
|
| Fedora | ``dnf install ocrmypdf`` |
|
||||||
+-------------------------------+-------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| macOS | ``brew install ocrmypdf`` |
|
| macOS | ``brew install ocrmypdf`` |
|
||||||
+-------------------------------+-------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||||
+-------------------------------+-------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| FreeBSD | ``pkg install py38-ocrmypdf`` |
|
| FreeBSD | ``pkg install textproc/py-ocrmypdf`` |
|
||||||
+-------------------------------+-------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
| Conda (WSL, macOS, Linux) | ``conda install ocrmypdf`` |
|
| Conda (WSL, macOS, Linux) | ``conda install ocrmypdf`` |
|
||||||
+-------------------------------+-------------------------------+
|
+-------------------------------+-----------------------------------------+
|
||||||
|
| Snap (snapcraft packaging) | ``snap install ocrmypdf`` |
|
||||||
|
+-------------------------------+-----------------------------------------+
|
||||||
|
|
||||||
More detailed procedures are outlined below. If you want to do a manual
|
More detailed procedures are outlined below. If you want to do a manual
|
||||||
install, or install a more recent version than your platform provides, read on.
|
install, or install a more recent version than your platform provides, read on.
|
||||||
@@ -41,11 +43,11 @@ Installing on Linux
|
|||||||
Debian and Ubuntu 18.04 or newer
|
Debian and Ubuntu 18.04 or newer
|
||||||
--------------------------------
|
--------------------------------
|
||||||
|
|
||||||
.. |deb-stable| image:: https://repology.org/badge/version-for-repo/debian_stable/ocrmypdf.svg
|
.. |deb-11| image:: https://repology.org/badge/version-for-repo/debian_11/ocrmypdf.svg
|
||||||
:alt: Debian 9 stable ("stretch")
|
:alt: Debian 11
|
||||||
|
|
||||||
.. |deb-testing| image:: https://repology.org/badge/version-for-repo/debian_testing/ocrmypdf.svg
|
.. |deb-12| image:: https://repology.org/badge/version-for-repo/debian_12/ocrmypdf.svg
|
||||||
:alt: Debian 10 testing ("buster")
|
:alt: Debian 12
|
||||||
|
|
||||||
.. |deb-unstable| image:: https://repology.org/badge/version-for-repo/debian_unstable/ocrmypdf.svg
|
.. |deb-unstable| image:: https://repology.org/badge/version-for-repo/debian_unstable/ocrmypdf.svg
|
||||||
:alt: Debian unstable
|
:alt: Debian unstable
|
||||||
@@ -56,17 +58,17 @@ Debian and Ubuntu 18.04 or newer
|
|||||||
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
||||||
:alt: Ubuntu 20.04 LTS
|
:alt: Ubuntu 20.04 LTS
|
||||||
|
|
||||||
.. |ubu-2010| image:: https://repology.org/badge/version-for-repo/ubuntu_20_10/ocrmypdf.svg
|
.. |ubu-2204| image:: https://repology.org/badge/version-for-repo/ubuntu_22_04/ocrmypdf.svg
|
||||||
:alt: Ubuntu 20.10
|
:alt: Ubuntu 22.04 LTS
|
||||||
|
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| **OCRmyPDF versions in Debian & Ubuntu** |
|
| **OCRmyPDF versions in Debian & Ubuntu** |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |latest| |
|
| |latest| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |deb-stable| |deb-testing| |deb-unstable| |
|
| |deb-11| |deb-12| |deb-unstable| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |ubu-1804| |ubu-2004| |ubu-2010| |
|
| |ubu-1804| |ubu-2004| |ubu-2204| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
|
|
||||||
Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users
|
Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users
|
||||||
@@ -80,8 +82,7 @@ As indicated in the table above, Debian and Ubuntu releases may lag
|
|||||||
behind the latest version. If the version available for your platform is
|
behind the latest version. If the version available for your platform is
|
||||||
out of date, you could opt to install the latest version from source.
|
out of date, you could opt to install the latest version from source.
|
||||||
See `Installing HEAD revision from
|
See `Installing HEAD revision from
|
||||||
sources <#installing-head-revision-from-sources>`__. Ubuntu 16.10 to 17.10
|
sources <#installing-head-revision-from-sources>`__.
|
||||||
inclusive also had ocrmypdf, but these versions are end of life.
|
|
||||||
|
|
||||||
For full details on version availability for your platform, check the
|
For full details on version availability for your platform, check the
|
||||||
`Debian Package Tracker <https://tracker.debian.org/pkg/ocrmypdf>`__ or
|
`Debian Package Tracker <https://tracker.debian.org/pkg/ocrmypdf>`__ or
|
||||||
@@ -91,18 +92,18 @@ For full details on version availability for your platform, check the
|
|||||||
|
|
||||||
OCRmyPDF for Debian and Ubuntu currently omit the JBIG2 encoder.
|
OCRmyPDF for Debian and Ubuntu currently omit the JBIG2 encoder.
|
||||||
OCRmyPDF works fine without it but will produce larger output files.
|
OCRmyPDF works fine without it but will produce larger output files.
|
||||||
If you build jbig2enc from source, ocrmypdf 7.0.0 and later will
|
If you build jbig2enc from source, ocrmypdf will
|
||||||
automatically detect it (specifically the ``jbig2`` binary) on the
|
automatically detect it (specifically the ``jbig2`` binary) on the
|
||||||
``PATH``. To add JBIG2 encoding, see :ref:`jbig2`.
|
``PATH``. To add JBIG2 encoding, see :ref:`jbig2`.
|
||||||
|
|
||||||
Fedora
|
Fedora
|
||||||
------
|
------
|
||||||
|
|
||||||
.. |fedora-32| image:: https://repology.org/badge/version-for-repo/fedora_32/ocrmypdf.svg
|
.. |fedora-35| image:: https://repology.org/badge/version-for-repo/fedora_35/ocrmypdf.svg
|
||||||
:alt: Fedora 32
|
:alt: Fedora 35
|
||||||
|
|
||||||
.. |fedora-33| image:: https://repology.org/badge/version-for-repo/fedora_33/ocrmypdf.svg
|
.. |fedora-36| image:: https://repology.org/badge/version-for-repo/fedora_36/ocrmypdf.svg
|
||||||
:alt: Fedora 33
|
:alt: Fedora 36
|
||||||
|
|
||||||
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
||||||
:alt: Fedore Rawhide
|
:alt: Fedore Rawhide
|
||||||
@@ -112,7 +113,7 @@ Fedora
|
|||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |latest| |
|
| |latest| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |fedora-32| |fedora-33| |fedora-rawhide| |
|
| |fedora-35| |fedora-36| |fedora-rawhide| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
|
|
||||||
Users of Fedora 29 or later may simply
|
Users of Fedora 29 or later may simply
|
||||||
@@ -138,9 +139,29 @@ from sources <#installing-head-revision-from-sources>`__.
|
|||||||
|
|
||||||
.. _ubuntu-lts-latest:
|
.. _ubuntu-lts-latest:
|
||||||
|
|
||||||
Installing the latest version on Ubuntu 20.04 LTS
|
Installing the latest version on Ubuntu 22.04 LTS
|
||||||
-------------------------------------------------
|
-------------------------------------------------
|
||||||
|
|
||||||
|
Ubuntu 22.04 includes ocrmypdf 13.4.0 - you can install that with
|
||||||
|
``apt install ocrmypdf``. To install a more recent version for the current
|
||||||
|
user, follow these steps:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
sudo apt-get update
|
||||||
|
sudo apt-get -y install ocrmypdf python3-pip
|
||||||
|
|
||||||
|
pip install --user --upgrade ocrmypdf
|
||||||
|
|
||||||
|
If you get the message ``WARNING: The script ocrmypdf is installed in
|
||||||
|
'/home/$USER/.local/bin' which is not on PATH.``, you may need to re-login
|
||||||
|
or open a new shell, or manually add this to your user's PATH.
|
||||||
|
|
||||||
|
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||||
|
|
||||||
|
Ubuntu 20.04 LTS
|
||||||
|
----------------
|
||||||
|
|
||||||
Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with ``apt``. To
|
Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with ``apt``. To
|
||||||
install a more recent version, uninstall the system-provided version of
|
install a more recent version, uninstall the system-provided version of
|
||||||
ocrmypdf, and install the following dependencies:
|
ocrmypdf, and install the following dependencies:
|
||||||
@@ -152,7 +173,6 @@ ocrmypdf, and install the following dependencies:
|
|||||||
sudo apt-get -y install \
|
sudo apt-get -y install \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
icc-profiles-free \
|
icc-profiles-free \
|
||||||
liblept5 \
|
|
||||||
libxml2 \
|
libxml2 \
|
||||||
pngquant \
|
pngquant \
|
||||||
python3-pip \
|
python3-pip \
|
||||||
@@ -172,6 +192,8 @@ To install for the current user only:
|
|||||||
export PATH=$HOME/.local/bin:$PATH
|
export PATH=$HOME/.local/bin:$PATH
|
||||||
pip3 install --user ocrmypdf
|
pip3 install --user ocrmypdf
|
||||||
|
|
||||||
|
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||||
|
|
||||||
Ubuntu 18.04 LTS
|
Ubuntu 18.04 LTS
|
||||||
----------------
|
----------------
|
||||||
|
|
||||||
@@ -186,10 +208,8 @@ of ocrmypdf, and install the following dependencies:
|
|||||||
sudo apt-get -y install \
|
sudo apt-get -y install \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
icc-profiles-free \
|
icc-profiles-free \
|
||||||
liblept5 \
|
|
||||||
libxml2 \
|
libxml2 \
|
||||||
pngquant \
|
pngquant \
|
||||||
python3-cffi \
|
|
||||||
python3-distutils \
|
python3-distutils \
|
||||||
python3-pkg-resources \
|
python3-pkg-resources \
|
||||||
python3-reportlab \
|
python3-reportlab \
|
||||||
@@ -214,69 +234,6 @@ user's ``PATH`` to check for the user's Python packages.
|
|||||||
|
|
||||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||||
|
|
||||||
Ubuntu 16.04 LTS
|
|
||||||
----------------
|
|
||||||
|
|
||||||
No package is available for Ubuntu 16.04. OCRmyPDF 8.0 and newer require
|
|
||||||
Python 3.6. Ubuntu 16.04 ships Python 3.5, but you can install Python
|
|
||||||
3.6 on it. Or, you can skip Python 3.6 and install OCRmyPDF 7.x or older
|
|
||||||
- for that procedure, please see the installation documentation for the
|
|
||||||
version of OCRmyPDF you plan to use.
|
|
||||||
|
|
||||||
**Install system packages for OCRmyPDF**
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
sudo apt-get update
|
|
||||||
sudo apt-get install -y software-properties-common python-software-properties
|
|
||||||
sudo add-apt-repository -y \
|
|
||||||
ppa:jonathonf/python-3.6 \
|
|
||||||
ppa:alex-p/tesseract-ocr
|
|
||||||
sudo apt-get update
|
|
||||||
sudo apt-get install -y \
|
|
||||||
ghostscript \
|
|
||||||
libexempi3 \
|
|
||||||
libffi6 \
|
|
||||||
pngquant \
|
|
||||||
python3.6 \
|
|
||||||
qpdf \
|
|
||||||
tesseract-ocr \
|
|
||||||
unpaper
|
|
||||||
|
|
||||||
This will install a Python 3.6 binary at ``/usr/bin/python3.6``
|
|
||||||
alongside the system's Python 3.5. Do not remove the system Python. This
|
|
||||||
will also install Tesseract 4.0 from a PPA, since the version available
|
|
||||||
in Ubuntu 16.04 is too old for OCRmyPDF.
|
|
||||||
|
|
||||||
Now install pip for Python 3.6. This will install the Python 3.6 version
|
|
||||||
of ``pip`` at ``/usr/local/bin/pip``.
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
curl https://bootstrap.pypa.io/get-pip.py | sudo python3.6
|
|
||||||
|
|
||||||
**Install OCRmyPDF**
|
|
||||||
|
|
||||||
OCRmyPDF requires the locale to be set for UTF-8. **On some minimal
|
|
||||||
Ubuntu installations**, such as the Ubuntu 16.04 Docker images it may be
|
|
||||||
necessary to set the locale.
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
# Optional: Only need to set these if they are not already set
|
|
||||||
export LC_ALL=C.UTF-8
|
|
||||||
export LANG=C.UTF-8
|
|
||||||
|
|
||||||
Now install OCRmyPDF for the current user, and ensure that the ``PATH``
|
|
||||||
environment variable contains ``$HOME/.local/bin``.
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
export PATH=$HOME/.local/bin:$PATH
|
|
||||||
pip3.6 install --user ocrmypdf
|
|
||||||
|
|
||||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
|
||||||
|
|
||||||
Arch Linux (AUR)
|
Arch Linux (AUR)
|
||||||
----------------
|
----------------
|
||||||
|
|
||||||
@@ -357,48 +314,6 @@ To install OCRmyPDF for Alpine Linux:
|
|||||||
|
|
||||||
apk add ocrmypdf
|
apk add ocrmypdf
|
||||||
|
|
||||||
Mageia 7
|
|
||||||
--------
|
|
||||||
|
|
||||||
There is no OS-level packaging available for Mageia, so you must install the
|
|
||||||
dependencies:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
# As root user
|
|
||||||
urpmi.update -a
|
|
||||||
urpmi \
|
|
||||||
ghostscript \
|
|
||||||
icc-profiles-openicc \
|
|
||||||
jbig2dec \
|
|
||||||
lib64leptonica5 \
|
|
||||||
pngquant \
|
|
||||||
python3-pip \
|
|
||||||
python3-cffi \
|
|
||||||
python3-distutils-extra \
|
|
||||||
python3-pkg-resources \
|
|
||||||
python3-reportlab \
|
|
||||||
qpdf \
|
|
||||||
tesseract \
|
|
||||||
tesseract-osd \
|
|
||||||
tesseract-eng \
|
|
||||||
tesseract-fra
|
|
||||||
|
|
||||||
To install ocrmypdf for the system:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
# As root user
|
|
||||||
pip3 install ocrmypdf
|
|
||||||
ldconfig
|
|
||||||
|
|
||||||
Or, to install for the current user only:
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
export PATH=$HOME/.local/bin:$PATH
|
|
||||||
pip3 install --user ocrmypdf
|
|
||||||
|
|
||||||
Other Linux packages
|
Other Linux packages
|
||||||
--------------------
|
--------------------
|
||||||
|
|
||||||
@@ -433,19 +348,6 @@ languages you can optionally install them all:
|
|||||||
|
|
||||||
brew install tesseract-lang # Optional: Install all language packs
|
brew install tesseract-lang # Optional: Install all language packs
|
||||||
|
|
||||||
.. note::
|
|
||||||
|
|
||||||
Users who previously installed OCRmyPDF on macOS using
|
|
||||||
``pip install ocrmypdf`` should remove the pip version
|
|
||||||
(``pip3 uninstall ocrmypdf``) before switching to the Homebrew
|
|
||||||
version.
|
|
||||||
|
|
||||||
.. note::
|
|
||||||
|
|
||||||
Users who previously installed OCRmyPDF from the private tap should
|
|
||||||
switch to the mainline version (``brew untap jbarlow83/ocrmypdf``)
|
|
||||||
and install from there.
|
|
||||||
|
|
||||||
Manual installation on macOS
|
Manual installation on macOS
|
||||||
----------------------------
|
----------------------------
|
||||||
|
|
||||||
@@ -479,19 +381,19 @@ Update the homebrew pip:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip3 install --upgrade pip
|
pip install --upgrade pip
|
||||||
|
|
||||||
You can then install OCRmyPDF from PyPI, for the current user:
|
You can then install OCRmyPDF from PyPI, for the current user:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip3 install --user ocrmypdf
|
pip install --user ocrmypdf
|
||||||
|
|
||||||
or system-wide:
|
or system-wide:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip3 install ocrmypdf
|
pip install ocrmypdf
|
||||||
|
|
||||||
The command line program should now be available:
|
The command line program should now be available:
|
||||||
|
|
||||||
@@ -553,8 +455,8 @@ to change the PATH.
|
|||||||
Windows Subsystem for Linux
|
Windows Subsystem for Linux
|
||||||
---------------------------
|
---------------------------
|
||||||
|
|
||||||
#. Install Ubuntu 18.04 for Windows Subsystem for Linux, if not already installed.
|
#. Install Ubuntu 22.04 for Windows Subsystem for Linux, if not already installed.
|
||||||
#. Follow the procedure to install :ref:`OCRmyPDF on Ubuntu 18.04 <ubuntu-lts-latest>`.
|
#. Follow the procedure to install :ref:`OCRmyPDF on Ubuntu 22.04 <ubuntu-lts-latest>`.
|
||||||
#. Open the Windows command prompt and create a symlink:
|
#. Open the Windows command prompt and create a symlink:
|
||||||
|
|
||||||
.. code-block:: powershell
|
.. code-block:: powershell
|
||||||
@@ -575,7 +477,7 @@ Cygwin64
|
|||||||
|
|
||||||
First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``::
|
First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``::
|
||||||
|
|
||||||
python36 (or later)
|
python37 (or later)
|
||||||
python3?-devel
|
python3?-devel
|
||||||
python3?-pip
|
python3?-pip
|
||||||
python3?-lxml
|
python3?-lxml
|
||||||
@@ -626,16 +528,13 @@ your command prompt can run the docker "hello world" container.
|
|||||||
Installing on FreeBSD
|
Installing on FreeBSD
|
||||||
=====================
|
=====================
|
||||||
|
|
||||||
.. image:: https://repology.org/badge/version-for-repo/freebsd/python:ocrmypdf.svg
|
.. image:: https://repology.org/badge/version-for-repo/freebsd/ocrmypdf.svg
|
||||||
:alt: FreeBSD
|
:alt: FreeBSD
|
||||||
:target: https://repology.org/project/python:ocrmypdf/versions
|
:target: https://repology.org/project/ocrmypdf/versions
|
||||||
|
|
||||||
FreeBSD 11.3, 12.0, 12.1-RELEASE and 13.0-CURRENT are supported. Other
|
|
||||||
versions likely work but have not been tested.
|
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pkg install py38-ocrmypdf
|
pkg install textproc/py-ocrmypdf
|
||||||
|
|
||||||
To install a more recent version, you could attempt to first install the system
|
To install a more recent version, you could attempt to first install the system
|
||||||
version with ``pkg``, then use ``pip install --user ocrmypdf``.
|
version with ``pkg``, then use ``pip install --user ocrmypdf``.
|
||||||
@@ -684,18 +583,18 @@ try:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip3 install --user ocrmypdf
|
pip install --user ocrmypdf
|
||||||
|
|
||||||
You should then be able to run ``ocrmypdf --version`` and see that the
|
You should then be able to run ``ocrmypdf --version`` and see that the
|
||||||
latest version was located.
|
latest version was located.
|
||||||
|
|
||||||
Since ``pip3 install --user`` does not work correctly on some platforms,
|
Since ``pip install --user`` does not work correctly on some platforms,
|
||||||
notably Ubuntu 16.04 and older, and the Homebrew version of Python,
|
notably Ubuntu 16.04 and older, and the Homebrew version of Python,
|
||||||
instead use this for a system wide installation:
|
instead use this for a system wide installation:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip3 install ocrmypdf
|
pip install ocrmypdf
|
||||||
|
|
||||||
.. note::
|
.. note::
|
||||||
|
|
||||||
@@ -711,16 +610,10 @@ OCRmyPDF currently requires these external programs and libraries to be
|
|||||||
installed, and must be satisfied using the operating system package
|
installed, and must be satisfied using the operating system package
|
||||||
manager. ``pip`` cannot provide them.
|
manager. ``pip`` cannot provide them.
|
||||||
|
|
||||||
- Python 3.6 or newer
|
The following versions are required:
|
||||||
- Ghostscript 9.15 or newer
|
|
||||||
- qpdf 8.1.0 or newer
|
|
||||||
- Tesseract 4.0.0-beta or newer
|
|
||||||
|
|
||||||
As of ocrmypdf 7.2.1, the following versions are recommended:
|
- Python 3.7 or newer
|
||||||
|
|
||||||
- Python 3.7 or 3.8
|
|
||||||
- Ghostscript 9.23 or newer
|
- Ghostscript 9.23 or newer
|
||||||
- qpdf 8.2.1
|
|
||||||
- Tesseract 4.0.0 or newer
|
- Tesseract 4.0.0 or newer
|
||||||
- jbig2enc 0.29 or newer
|
- jbig2enc 0.29 or newer
|
||||||
- pngquant 2.5 or newer
|
- pngquant 2.5 or newer
|
||||||
@@ -752,7 +645,7 @@ unfortunately, the ``pip install`` command cannot satisfy all of them.
|
|||||||
Installing HEAD revision from sources
|
Installing HEAD revision from sources
|
||||||
=====================================
|
=====================================
|
||||||
|
|
||||||
If you have ``git`` and Python 3.6 or newer installed, you can install
|
If you have ``git`` and Python 3.7 or newer installed, you can install
|
||||||
from source. When the ``pip`` installer runs, it will alert you if
|
from source. When the ``pip`` installer runs, it will alert you if
|
||||||
dependencies are missing.
|
dependencies are missing.
|
||||||
|
|
||||||
@@ -766,7 +659,7 @@ environment:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip3 install git+https://github.com/jbarlow83/OCRmyPDF.git
|
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
|
|
||||||
Or, to install in `development
|
Or, to install in `development
|
||||||
mode <https://pythonhosted.org/setuptools/setuptools.html#development-mode>`__,
|
mode <https://pythonhosted.org/setuptools/setuptools.html#development-mode>`__,
|
||||||
@@ -774,18 +667,18 @@ allowing customization of OCRmyPDF, use the ``-e`` flag:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
pip3 install -e git+https://github.com/jbarlow83/OCRmyPDF.git
|
pip install -e git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
|
|
||||||
You may find it easiest to install in a virtual environment, rather than
|
You may find it easiest to install in a virtual environment, rather than
|
||||||
system-wide:
|
system-wide:
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
git clone -b master https://github.com/jbarlow83/OCRmyPDF.git
|
git clone -b master https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
python3 -m venv
|
python3 -m venv
|
||||||
source venv/bin/activate
|
source venv/bin/activate
|
||||||
cd OCRmyPDF
|
cd OCRmyPDF
|
||||||
pip3 install .
|
pip install .
|
||||||
|
|
||||||
However, ``ocrmypdf`` will only be accessible on the system PATH when
|
However, ``ocrmypdf`` will only be accessible on the system PATH when
|
||||||
you activate the virtual environment.
|
you activate the virtual environment.
|
||||||
@@ -808,8 +701,8 @@ To install all of the development and test requirements:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
git clone -b master https://github.com/jbarlow83/OCRmyPDF.git
|
git clone -b master https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
python3 -m venv
|
python -m venv
|
||||||
source venv/bin/activate
|
source venv/bin/activate
|
||||||
cd OCRmyPDF
|
cd OCRmyPDF
|
||||||
pip install -e .[test]
|
pip install -e .[test]
|
||||||
|
|||||||
+14
-9
@@ -2,7 +2,12 @@
|
|||||||
Introduction
|
Introduction
|
||||||
============
|
============
|
||||||
|
|
||||||
OCRmyPDF is a Python 3 application and library that adds OCR layers to PDFs.
|
OCRmyPDF is an application and library that adds text "layers" to images
|
||||||
|
in PDFs, making scanned image PDFs searchable. It uses OCR to guess what text
|
||||||
|
is contained in images. It is written in Python. OCRmyPDF supports plugins
|
||||||
|
that allow customization of its processing steps, and is very tolerant of
|
||||||
|
PDFs that contain scanned images and "born digital" content that needs no
|
||||||
|
text recognition.
|
||||||
|
|
||||||
About OCR
|
About OCR
|
||||||
=========
|
=========
|
||||||
@@ -26,7 +31,7 @@ exactly. They contain `vector
|
|||||||
graphics <http://vector-conversions.com/vectorizing/raster_vs_vector.html>`__
|
graphics <http://vector-conversions.com/vectorizing/raster_vs_vector.html>`__
|
||||||
that can contain raster objects such as scanned images. Because PDFs can
|
that can contain raster objects such as scanned images. Because PDFs can
|
||||||
contain multiple pages (unlike many image formats) and can contain fonts
|
contain multiple pages (unlike many image formats) and can contain fonts
|
||||||
and text, it is a good formats for exchanging scanned documents.
|
and text, it is a good format for exchanging scanned documents.
|
||||||
|
|
||||||
|image|
|
|image|
|
||||||
|
|
||||||
@@ -35,9 +40,9 @@ have one image. Some scanners or scanning software will segment pages
|
|||||||
into monochromatic text and color regions for example, to improve the
|
into monochromatic text and color regions for example, to improve the
|
||||||
compression ratio and appearance of the page.
|
compression ratio and appearance of the page.
|
||||||
|
|
||||||
Rasterizing a PDF is the process of generating an image suitable for
|
Rasterizing a PDF is the process of generating corresponding raster images.
|
||||||
display or analyzing with an OCR engine. OCR engines like Tesseract work
|
OCR engines like Tesseract work with images, not scalable vector graphics
|
||||||
with images, not vector objects.
|
or mixed raster-vector-text graphics such as PDF.
|
||||||
|
|
||||||
About PDF/A
|
About PDF/A
|
||||||
===========
|
===========
|
||||||
@@ -76,7 +81,7 @@ OCRmyPDF analyzes each page of a PDF to determine the colorspace and
|
|||||||
resolution (DPI) needed to capture all of the information on that page
|
resolution (DPI) needed to capture all of the information on that page
|
||||||
without losing content. It uses
|
without losing content. It uses
|
||||||
`Ghostscript <http://ghostscript.com/>`__ to rasterize the page, and
|
`Ghostscript <http://ghostscript.com/>`__ to rasterize the page, and
|
||||||
then performs on OCR on the rasterized image to create an OCR "layer".
|
then performs on OCR the rasterized image to create an OCR "layer".
|
||||||
The layer is then grafted back onto the original PDF.
|
The layer is then grafted back onto the original PDF.
|
||||||
|
|
||||||
While one can use a program like Ghostscript or ImageMagick to get an
|
While one can use a program like Ghostscript or ImageMagick to get an
|
||||||
@@ -84,9 +89,9 @@ image and put the image through Tesseract, that actually creates a new
|
|||||||
PDF and many details may be lost. OCRmyPDF can produce a minimally
|
PDF and many details may be lost. OCRmyPDF can produce a minimally
|
||||||
changed PDF as output.
|
changed PDF as output.
|
||||||
|
|
||||||
OCRmyPDF also some image processing options like deskew which improve
|
OCRmyPDF also provides some image processing options, like deskew, which
|
||||||
the appearance of files and quality of OCR. When these are used, the OCR
|
improves the appearance of files and quality of OCR. When these are used,
|
||||||
layer is grafted onto the processed image instead.
|
the OCR layer is grafted onto the processed image instead.
|
||||||
|
|
||||||
By default, OCRmyPDF produces archival PDFs – PDF/A, which are a
|
By default, OCRmyPDF produces archival PDFs – PDF/A, which are a
|
||||||
stricter subset of PDF features designed for long term archives. If
|
stricter subset of PDF features designed for long term archives. If
|
||||||
|
|||||||
@@ -32,6 +32,9 @@ For all other Linux, you must build a JBIG2 encoder from source:
|
|||||||
|
|
||||||
.. _jbig2-lossy:
|
.. _jbig2-lossy:
|
||||||
|
|
||||||
|
Dependencies include libtoolize and libleptonica, which on Ubuntu systems
|
||||||
|
are packaged as libtool and libleptonica-dev.
|
||||||
|
|
||||||
Lossy mode JBIG2
|
Lossy mode JBIG2
|
||||||
================
|
================
|
||||||
|
|
||||||
|
|||||||
@@ -54,6 +54,33 @@ to what languages it should search for. Multiple languages can be
|
|||||||
requested using either ``-l eng+fra`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fra``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
|
Gentoo users
|
||||||
|
============
|
||||||
|
|
||||||
|
On Gentoo the package ``app-text/tessdata_fast``, which ``app-text/tesseract`` depends on, handles Tesseract languages.
|
||||||
|
It accepts USE flags to select what languages should be installed, these can be set in ``/etc/portage/package.use``.
|
||||||
|
Alternatively one can globally set the `L10N use extension <https://wiki.gentoo.org/wiki/Localization/Guide#L10N>`__ in ``/etc/portage/make.conf``.
|
||||||
|
This enables these languages for all packages (e.g. including aspell).
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
# Display a list of all Tesseract language packs
|
||||||
|
equery uses app-text/tessdata_fast
|
||||||
|
|
||||||
|
# Add English and German language support for Tesseract only
|
||||||
|
echo 'app-text/tessdata_fast l10n_de l10n_en' >> /etc/portage/package.use
|
||||||
|
|
||||||
|
# Add global English and German language support (the `l10n_` from equery has to be omited)
|
||||||
|
echo L10N="de en" >> /etc/portage/make.conf
|
||||||
|
|
||||||
|
# update system to reflect changed USE flags
|
||||||
|
emerge --update --deep --newuse @world
|
||||||
|
|
||||||
|
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||||
|
to what languages it should search for. Multiple languages can be
|
||||||
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
|
``-l eng -l fra``.
|
||||||
|
|
||||||
macOS users
|
macOS users
|
||||||
===========
|
===========
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,60 @@
|
|||||||
|
================
|
||||||
|
Maintainer notes
|
||||||
|
================
|
||||||
|
|
||||||
|
This is for those who package OCRmyPDF for downstream use. (Thank you
|
||||||
|
for your hard work.)
|
||||||
|
|
||||||
|
Known ports/packagers
|
||||||
|
=====================
|
||||||
|
|
||||||
|
OCRmyPDF has been ported to many platforms already. If you are interesting in
|
||||||
|
porting to a new platform, check with
|
||||||
|
`Repology <https://repology.org/projects/?search=ocrmypdf>`__ to see the status
|
||||||
|
of that platform.
|
||||||
|
|
||||||
|
Make sure you can package pikepdf
|
||||||
|
---------------------------------
|
||||||
|
|
||||||
|
pikepdf, created by the same author, is a mixed Python and C++14 package with
|
||||||
|
much stiffer build requirements. If you want to use OCRmyPDF on some novel platform
|
||||||
|
or distribution, first make sure you can package pikepdf.
|
||||||
|
|
||||||
|
Non-Python dependencies
|
||||||
|
-----------------------
|
||||||
|
|
||||||
|
Note that we have non-Python dependencies. In particular, OCRmyPDF requires
|
||||||
|
Ghostscript and Tesseract OCR to be installed and needs to be able to locate their
|
||||||
|
binaries on the system PATH. On Windows, OCRmyPDF will also check the registry
|
||||||
|
for their locations.
|
||||||
|
|
||||||
|
Tesseract OCR relies on SIMD for performance and only has proper support for this
|
||||||
|
on ARM and x86_64. Performance may be poor on other processor architectures.
|
||||||
|
|
||||||
|
Versioning scheme
|
||||||
|
-----------------
|
||||||
|
|
||||||
|
OCRmyPDF uses setuptools-scm for versioning, which derives the version from
|
||||||
|
Git as a single source of truth. This may be unsuitable for some distributions, e.g.
|
||||||
|
to indicate that your distribution modifies OCRmyPDF in some way.
|
||||||
|
|
||||||
|
You can patch the ``__version__`` variable in ``src/ocrmypdf/_version.py`` if
|
||||||
|
necessary.
|
||||||
|
|
||||||
|
OCRmyPDF uses setuptools-scm-git-archive to ensure that tarballs downloaded from
|
||||||
|
GitHub contain version information. Unfortunately, these tarballs are not always
|
||||||
|
deterministic. See this
|
||||||
|
`issue <https://github.com/ocrmypdf/OCRmyPDF/issues/841#issuecomment-936562696>`_.
|
||||||
|
|
||||||
|
jbig2enc
|
||||||
|
--------
|
||||||
|
|
||||||
|
OCRmyPDF will use jbig2enc, a JBIG2 encoder, if one can be found. Some distributions
|
||||||
|
have shied away from packaging JBIG2 because it contains patented algorithms, but
|
||||||
|
all patents have expired since 2017. If possible, consider packaging it too to
|
||||||
|
improve OCRmyPDF's compression.
|
||||||
|
|
||||||
|
Command line completions
|
||||||
|
------------------------
|
||||||
|
|
||||||
|
Please ensure that command line completions are installed.
|
||||||
+23
-1
@@ -152,6 +152,21 @@ hooks. As such, you cannot "chain" a series of plugin filters together in this
|
|||||||
way. Instead, a single hook implementation should be responsible for any such
|
way. Instead, a single hook implementation should be responsible for any such
|
||||||
chaining operations.
|
chaining operations.
|
||||||
|
|
||||||
|
Examples
|
||||||
|
========
|
||||||
|
|
||||||
|
* OCRmyPDF's test suite contains several plugins that are used to simulate certain
|
||||||
|
test conditions.
|
||||||
|
* `ocrmypdf-papermerge <https://github.com/papermerge/OCRmyPDF_papermerge>`_ is
|
||||||
|
a production plugin that integrates OCRmyPDF and the Papermerge document
|
||||||
|
management system.
|
||||||
|
|
||||||
|
|
||||||
|
Suppressing or overriding other plugins
|
||||||
|
---------------------------------------
|
||||||
|
|
||||||
|
.. autofunction:: ocrmypdf.pluginspec.initialize
|
||||||
|
|
||||||
Custom command line arguments
|
Custom command line arguments
|
||||||
-----------------------------
|
-----------------------------
|
||||||
|
|
||||||
@@ -162,7 +177,7 @@ Custom command line arguments
|
|||||||
Execution and progress reporting
|
Execution and progress reporting
|
||||||
--------------------------------
|
--------------------------------
|
||||||
|
|
||||||
.. autoclass: ocrmypdf.pluginspec.Executor
|
.. autoclass:: ocrmypdf.pluginspec.Executor
|
||||||
:members:
|
:members:
|
||||||
|
|
||||||
.. autofunction:: ocrmypdf.pluginspec.get_logging_console
|
.. autofunction:: ocrmypdf.pluginspec.get_logging_console
|
||||||
@@ -206,3 +221,10 @@ PDF/A production
|
|||||||
----------------
|
----------------
|
||||||
|
|
||||||
.. autofunction:: ocrmypdf.pluginspec.generate_pdfa
|
.. autofunction:: ocrmypdf.pluginspec.generate_pdfa
|
||||||
|
|
||||||
|
PDF optimization
|
||||||
|
----------------
|
||||||
|
|
||||||
|
.. autofunction:: ocrmypdf.pluginspec.optimize_pdf
|
||||||
|
|
||||||
|
.. autofunction:: ocrmypdf.pluginspec.is_optimization_enabled
|
||||||
@@ -12,6 +12,260 @@ may be unreliable. Use the API to depend on precise behavior.
|
|||||||
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||||
wish to use some of its features for working with PDFs.
|
wish to use some of its features for working with PDFs.
|
||||||
|
|
||||||
|
The most recent release of OCRmyPDF is |OCRmyPDF PyPI|. Any newer versions
|
||||||
|
referred to in these notes may exist the main branch but have not been
|
||||||
|
tagged yet.
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
|
||||||
|
Attention maintainers: that these release notes may be updated with information
|
||||||
|
about a forthcoming release that has not been tagged yet. A release is only
|
||||||
|
official when it's tagged and posted to PyPI.
|
||||||
|
|
||||||
|
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||||
|
|
||||||
|
v13.6.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Added a shim to prevent an "error during error handling" for Python 3.7 and 3.8.
|
||||||
|
- Modernized some type annotations.
|
||||||
|
- Improved annotations on our _windows module to help IDEs and mypy figure out what
|
||||||
|
we're doing.
|
||||||
|
|
||||||
|
v13.6.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Require setuptools-scm 7.0.5 to avoid possible issues with source distributions in
|
||||||
|
earlier versions of setuptools-scm.
|
||||||
|
- Suppress a spurious warning, improve tests, improve typing and other miscellany.
|
||||||
|
|
||||||
|
v13.6.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Added a new ``initialize`` plugin hook, making it possible to suppress built-in
|
||||||
|
plugins more easily, among other possibilities.
|
||||||
|
- Fixed an issue where unpaper would exit with a "wrong stream" error, probably
|
||||||
|
related to images with an odd integer width. :issue:`887, 665`
|
||||||
|
|
||||||
|
v13.5.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Added a new ``optimize_pdf`` plugin hook, making it possible to create plugins that
|
||||||
|
replace or enhance OCRmyPDF's PDF optimizer.
|
||||||
|
- Removed all max version restrictions. Our new policy is to blacklist known-bad releases
|
||||||
|
and only block known-bad versions of dependencies.
|
||||||
|
- The naming schema for object that holds all OCR text that OCRmyPDF inserts has
|
||||||
|
changed. This has always been an implementation detail (and remains so), but possibly,
|
||||||
|
someone was relying on it and would appreciate the heads-up.
|
||||||
|
- Cleanup.
|
||||||
|
|
||||||
|
v13.4.7
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed PermissionError when cleaning up temporary files in rare cases. :issue:`974`
|
||||||
|
- Fixed PermissionError when calling ``os.nice`` on platforms that lack it. :issue:`973`
|
||||||
|
- Suppressed some warnings from libxmp during tests.
|
||||||
|
|
||||||
|
v13.4.6
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Convert error on corrupt ICC profiles into a warning. Thanks to @oscherler.
|
||||||
|
|
||||||
|
v13.4.5
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Remove upper bound on pdfminer.six version.
|
||||||
|
- Documentation.
|
||||||
|
|
||||||
|
v13.4.4
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Updated pdfminer.six version.
|
||||||
|
- Docker image changed to Ubuntu 22.04 now that it is released and provides the
|
||||||
|
dependencies we need. This seems more consistent than our recent change to
|
||||||
|
Debian.
|
||||||
|
|
||||||
|
v13.4.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fix error on pytest.skip() with older versions of pytest.
|
||||||
|
- Documentation updates.
|
||||||
|
|
||||||
|
v13.4.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Worked around a
|
||||||
|
`major regression in Ghostscript 9.56.0 <https://bugs.ghostscript.com/show_bug.cgi?id=705187>`__
|
||||||
|
where **all OCR text is stripped out of the PDF**. It simply removes all text,
|
||||||
|
even generated by software other than OCRmyPDF. Fortunately, we can ask
|
||||||
|
Ghostscript 9.56.0 to use its old behavior that worked correctly for our purposes.
|
||||||
|
Users must avoid the combination (Ghostscript 9.56.0, ocrmypdf <13.4.2) since
|
||||||
|
older versions of OCRmyPDF have no way of detecting that this particular
|
||||||
|
version of Ghostscript removes all OCR text.
|
||||||
|
- Marked pdfminer 20220319 as supported.
|
||||||
|
- Fixed some deprecation warnings from recent versions of Pillow and pytest.
|
||||||
|
- Test suite now covers Python 3.10 (Python 3.10 worked fine before, but was not
|
||||||
|
being tested).
|
||||||
|
- Docker image now uses debian:bookworm-slim as the base image to fix the Docker
|
||||||
|
image build.
|
||||||
|
|
||||||
|
v13.4.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Temporarily make threads rather than processes the default executor worker, due
|
||||||
|
to a persistent deadlock issue when processes are used. Add a new command line
|
||||||
|
argument ``--no-use-threads`` to disable this.
|
||||||
|
|
||||||
|
v13.4.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed test failures when using pikepdf 5.0.0.
|
||||||
|
- Various improvements to the optimizer. In particular, we now recognize PDF images
|
||||||
|
that are encoded with both deflate (PNG) and DCT (JPEG), and also produce PDF
|
||||||
|
with images compressed with deflate and DCT, since this often yields file size
|
||||||
|
improvements compared to plain DCT.
|
||||||
|
|
||||||
|
v13.3.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Made a harmless but "scary" exception after failing to optimize an image less scary.
|
||||||
|
- Added a warning if a page image is too large for unpaper to clean. The image is
|
||||||
|
passed through without cleaning. This is due to a hard-coded limitation in a
|
||||||
|
C library used by unpaper so it cannot be rectified easily.
|
||||||
|
- We now use better default settings when calling img2pdf.
|
||||||
|
- We no longer try to optimize images that we failed to save in certain situations.
|
||||||
|
- We now account for some differences in text output from Tesseract 5 compared to
|
||||||
|
Tesseract 4.
|
||||||
|
- Better handling of Ghostscript producing empty images when attempting to rasterize
|
||||||
|
page images.
|
||||||
|
|
||||||
|
v13.2.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Removed all runtime uses of distutils since it is deprecated in standard library. We
|
||||||
|
previous used ``distutils.version`` to examine version numbers of dependencies
|
||||||
|
at run time, and now use ``packaging.version`` for this. This is a new
|
||||||
|
dependency.
|
||||||
|
- Fixed an error message advising the user that Ghostscript was not installed being
|
||||||
|
suppressed when this condition actually happens.
|
||||||
|
- Fixed an issue with incorrect page number and totals being displayed in the progress
|
||||||
|
bar. This was purely a display/presentation issue. :issue:`876`.
|
||||||
|
|
||||||
|
v13.1.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed issue with attempting to deskew a blank page on Tesseract 5. :issue:`868`.
|
||||||
|
|
||||||
|
v13.1.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Changed to using Python concurrent.futures-based parallel execution instead of
|
||||||
|
pools, since futures have now exceed pools in features.
|
||||||
|
- If a child worker is terminated (perhaps by the operating system or the user
|
||||||
|
killing it in a task manager), the parallel task will fail an error message.
|
||||||
|
Previously, the main ocrmypdf process would "hang" indefinitely, waiting for the
|
||||||
|
child to report.
|
||||||
|
- Added new argument ``--tesseract-thresholding`` to provide control over Tesseract 5's
|
||||||
|
threshold parameter.
|
||||||
|
- Documentation updates and changes. Better documentation for ``--output-type none``,
|
||||||
|
added a few releases ago. Removed some obsolete documentation.
|
||||||
|
- Improved bash completions - thanks to @FPille.
|
||||||
|
|
||||||
|
v13.0.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
**Breaking changes**
|
||||||
|
|
||||||
|
- The deprecated module ``ocrmypdf.leptonica`` has been removed.
|
||||||
|
- We no longer depend on Leptonica (``liblept``) or CFFI (``libffi``,
|
||||||
|
``python3-cffi``). (Note that Tesseract still requires Leptonica; OCRmyPDF no longer
|
||||||
|
directly uses this library.)
|
||||||
|
- The argument ``--remove-background`` is temporarily disabled while we search for an
|
||||||
|
alternative to the Leptonica implementation of this feature.
|
||||||
|
- The ``--threshold`` argument has been removed, since this also depended on Leptonica.
|
||||||
|
Tesseract 5.x has implemented improvements to thresholding, so this feature will be
|
||||||
|
redundant anyway.
|
||||||
|
- ``--deskew`` was previous calculated by a Leptonica algorithm. We now use a feature
|
||||||
|
of Tesseract to find the appropriate the angle to deskew a page. The deskew angle
|
||||||
|
according to Tesseract may differ from Leptonica's algorithm. At least in theory,
|
||||||
|
Tesseract's deskew angle is informed by a more complex analysis than Leptonica,
|
||||||
|
so this should improve results in general. We also use Pillow to perform the
|
||||||
|
deskewing, which may affect the appearance of the image compared to Leptonica.
|
||||||
|
- Support for Python 3.6 was dropped, since this release is approaching end of life.
|
||||||
|
- We now require pikepdf 4.0 or newer. This, in turn, means that OCRmyPDF requires
|
||||||
|
a system compatible with the manylinux2014 specification. This change was "forced"
|
||||||
|
by Pillow not releasing manylinux2010 wheels anymore.
|
||||||
|
- We no longer provide requirements.txt-style files. Use ``pip install ocrmypdf[...]``
|
||||||
|
instead.
|
||||||
|
- Bumped required versions of several libraries.
|
||||||
|
|
||||||
|
**Fixes**
|
||||||
|
|
||||||
|
- Fixed an issue where OCRmyPDF failed to find Ghostscript on Windows even when
|
||||||
|
installed, and would exit with an error.
|
||||||
|
- By removing Leptonica, we fixed all issues related to Leptonica on Apple
|
||||||
|
Silicon or Leptonica failing to import on Windows.
|
||||||
|
|
||||||
|
v12.7.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed "invalid version number" error for Tesseract packaging with nonstandard
|
||||||
|
version "5.0.0-rc1.20211030".
|
||||||
|
- Fixed use of deprecated ``importlib.resources.read_binary``.
|
||||||
|
- Replace some uses of string paths with ``pathlib.Path``.
|
||||||
|
- Fixed a leaked file handle when using ``--output-type none``.
|
||||||
|
- Removed shims to support versions of pikepdf that are no longer supported.
|
||||||
|
|
||||||
|
v12.7.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Declare support for pdfminer.six v20211012.
|
||||||
|
|
||||||
|
v12.7.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed test suite failure when using pikepdf 3.2.0 that was compiled with pybind11
|
||||||
|
2.8.0. :issue:`843`
|
||||||
|
- Improve advice to user about using ``--max-image-mpixels`` if OCR fails for this
|
||||||
|
reason.
|
||||||
|
- Minor documentation fixes. (Thanks to @mara004.)
|
||||||
|
- Don't require importlib-metadata and importlib-resources backports on versions of
|
||||||
|
Python where the standard library implementation is sufficient.
|
||||||
|
(Thanks to Marco Genasci.)
|
||||||
|
|
||||||
|
v12.6.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Implemented ``--output-type=none`` to skip producing PDFs for applications that
|
||||||
|
only want sidecar files (:issue:`787`).
|
||||||
|
- Fixed ambiguities in descriptions of behavior of ``--jbig2-lossy``.
|
||||||
|
- Various improvements to documentation.
|
||||||
|
|
||||||
|
v12.5.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed build failure for the combination of PyPy 3.6 and pikepdf 3.0. This
|
||||||
|
combination can work in a source build but does not work with wheels.
|
||||||
|
- Accepted bot that wanted to upgrade our deprecated requirements.txt.
|
||||||
|
- Documentation updates.
|
||||||
|
- Replace pkg_resources and install dependency on setuptools with
|
||||||
|
importlib-metadata and importlib-resources.
|
||||||
|
- Fixed regression in hocrtransform causing text to be omitted when this
|
||||||
|
renderer was used.
|
||||||
|
- Fixed some typing errors.
|
||||||
|
|
||||||
|
v12.4.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- When grafting text layers, use pikepdf's ``unparse_content_stream`` if available.
|
||||||
|
- Confirmed support for pluggy 1.0. (Thanks @QuLogic.)
|
||||||
|
- Fixed some typing issues, improved pre-commit settings, and fixed issues
|
||||||
|
flagged by linters.
|
||||||
|
- PyPy 7.3.3 (=Python 3.6) is now supported. Note that PyPy does not necessarily
|
||||||
|
run faster, because the vast majority of OCRmyPDF's execution time is spent
|
||||||
|
running OCR or generally executing native code. However, PyPy may bring speed
|
||||||
|
improvements in some areas.
|
||||||
|
|
||||||
v12.3.3
|
v12.3.3
|
||||||
=======
|
=======
|
||||||
|
|
||||||
|
|||||||
+17
-22
@@ -19,50 +19,45 @@
|
|||||||
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
# SOFTWARE.
|
# SOFTWARE.
|
||||||
|
|
||||||
# This script must be edited to meet your needs.
|
from __future__ import annotations
|
||||||
|
|
||||||
|
# This script must be edited to meet your needs.
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
import ocrmypdf
|
import ocrmypdf
|
||||||
|
|
||||||
# pylint: disable=logging-format-interpolation
|
# pylint: disable=logging-format-interpolation
|
||||||
# pylint: disable=logging-not-lazy
|
# pylint: disable=logging-not-lazy
|
||||||
|
|
||||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
script_dir = Path(__file__).parent
|
||||||
print(script_dir + '/batch.py: Start')
|
|
||||||
|
|
||||||
if len(sys.argv) > 1:
|
if len(sys.argv) > 1:
|
||||||
start_dir = sys.argv[1]
|
start_dir = Path(sys.argv[1])
|
||||||
else:
|
else:
|
||||||
start_dir = '.'
|
start_dir = Path('.')
|
||||||
|
|
||||||
if len(sys.argv) > 2:
|
if len(sys.argv) > 2:
|
||||||
log_file = sys.argv[2]
|
log_file = Path(sys.argv[2])
|
||||||
else:
|
else:
|
||||||
log_file = script_dir + '/ocr-tree.log'
|
log_file = script_dir.with_name('ocr-tree.log')
|
||||||
|
|
||||||
logging.basicConfig(
|
logging.basicConfig(
|
||||||
level=logging.INFO,
|
level=logging.INFO,
|
||||||
format='%(asctime)s %(message)s',
|
format='%(asctime)s %(message)s',
|
||||||
filename=log_file,
|
filename=log_file,
|
||||||
filemode='w',
|
filemode='a',
|
||||||
)
|
)
|
||||||
|
|
||||||
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
||||||
|
|
||||||
for dir_name, subdirs, file_list in os.walk(start_dir):
|
for filename in start_dir.glob("**/*.py"):
|
||||||
logging.info(dir_name + '\n')
|
logging.info(f"Processing {filename}")
|
||||||
os.chdir(dir_name)
|
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||||
for filename in file_list:
|
if result == ocrmypdf.ExitCode.already_done_ocr:
|
||||||
file_ext = os.path.splitext(filename)[1]
|
logging.error("Skipped document because it already contained text")
|
||||||
if file_ext == '.pdf':
|
elif result == ocrmypdf.ExitCode.ok:
|
||||||
full_path = dir_name + '/' + filename
|
logging.info("OCR complete")
|
||||||
print(full_path)
|
logging.info(result)
|
||||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
|
||||||
if result == ocrmypdf.ExitCode.already_done_ocr:
|
|
||||||
print("Skipped document because it already contained text")
|
|
||||||
elif result == ocrmypdf.ExitCode.ok:
|
|
||||||
print("OCR complete")
|
|
||||||
logging.info(result)
|
|
||||||
|
|||||||
+256
-66
@@ -1,6 +1,6 @@
|
|||||||
# ocrmypdf completion -*- shell-script -*-
|
# ocrmypdf completion -*- shell-script -*-
|
||||||
|
|
||||||
# Copyright 2019 Frank Pille
|
# Copyright 2019, 2021 Frank Pille
|
||||||
# Copyright 2020 Alex Willner
|
# Copyright 2020 Alex Willner
|
||||||
#
|
#
|
||||||
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
# Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||||
@@ -23,95 +23,285 @@
|
|||||||
|
|
||||||
set -o errexit
|
set -o errexit
|
||||||
|
|
||||||
_ocrmypdf()
|
__ocrmypdf_arguments()
|
||||||
{
|
{
|
||||||
local cur prev cword words split
|
local arguments="--help (show help message)
|
||||||
|
--language (language(s) of the file to be OCRed)
|
||||||
|
--image-dpi (assume this DPI if input image DPI is unknown)
|
||||||
|
--output-type (select PDF output options)
|
||||||
|
--sidecar (write OCR to text file)
|
||||||
|
--version (print program version and exit)
|
||||||
|
--jobs (how many worker processes to use)
|
||||||
|
--quiet (suppress INFO messages)
|
||||||
|
--verbose (set verbosity level)
|
||||||
|
--title (set metadata)
|
||||||
|
--author (set metadata)
|
||||||
|
--subject (set metadata)
|
||||||
|
--keywords (set metadata)
|
||||||
|
--rotate-pages (rotate pages to correct orientation)
|
||||||
|
--remove-background (attempt to remove background from pages)
|
||||||
|
--deskew (fix small horizontal alignment skew)
|
||||||
|
--clean (clean document images before OCR)
|
||||||
|
--clean-final (clean document images and keep result)
|
||||||
|
--unpaper-args (a quoted string of arguments to pass to unpaper)
|
||||||
|
--oversample (oversample images to this DPI)
|
||||||
|
--remove-vectors (don\'t send vector objects to OCR)
|
||||||
|
--threshold (threshold images before OCR)
|
||||||
|
--force-ocr (OCR documents that already have printable text)
|
||||||
|
--skip-text (skip OCR on any pages that already contain text)
|
||||||
|
--redo-ocr (redo OCR on any pages that seem to have OCR already)
|
||||||
|
--skip-big (skip OCR on pages larger than this many MPixels)
|
||||||
|
--optimize (select optimization level)
|
||||||
|
--jpeg-quality (JPEG quality [0..100])
|
||||||
|
--png-quality (PNG quality [0..100])
|
||||||
|
--jbig2-lossy (enable lossy JBIG2 (see docs))
|
||||||
|
--pages (apply OCR to only the specified pages)
|
||||||
|
--max-image-mpixels (image decompression bomb threshold)
|
||||||
|
--pdf-renderer (select PDF renderer options)
|
||||||
|
--rotate-pages-threshold (page rotation confidence)
|
||||||
|
--pdfa-image-compression (set PDF/A image compression options)
|
||||||
|
--fast-web-view (if file size if above this amount in MB linearize PDF)
|
||||||
|
--plugin (name of plugin to import)
|
||||||
|
--keep-temporary-files (keep temporary files (debug)
|
||||||
|
--tesseract-config (set custom tesseract config file)
|
||||||
|
--tesseract-pagesegmode (set tesseract --psm)
|
||||||
|
--tesseract-oem (set tesseract --oem)
|
||||||
|
--tesseract-thresholding (set tesseract image thresholding)
|
||||||
|
--tesseract-timeout (maximum number of seconds to wait for OCR)
|
||||||
|
--user-words (specify location of user words file)
|
||||||
|
--user-patterns (specify location of user patterns file)
|
||||||
|
--no-progress-bar (disable the progress bar)
|
||||||
|
"
|
||||||
|
|
||||||
# Homebrew on Macs have version 1.3 of bash-completion which doesn't include - see #502
|
COMPREPLY=( $( compgen -W "$arguments" -- "$cur") )
|
||||||
if declare -F _init_completions >/dev/null 2>&1; then
|
|
||||||
_init_completion -s || return
|
# Remove description if only one completion exists
|
||||||
else
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
COMPREPLY=()
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
_get_comp_words_by_ref cur prev words cword
|
|
||||||
fi
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
if [[ $cur == -* ]]; then
|
__ocrmypdf_output-type()
|
||||||
COMPREPLY=( $( compgen -W '--language --image-dpi --output-type
|
{
|
||||||
--sidecar --version --jobs --quiet --verbose --title --author
|
local choices="pdfa (output a PDF/A (default))
|
||||||
--subject --keywords --rotate-pages --remove-background --deskew
|
pdf (output a standard PDF)
|
||||||
--clean --clean-final --unpaper-args --oversample --remove-vectors
|
pdfa-1 (output a PDF/A-1b)
|
||||||
--threshold --force-ocr --skip-text --redo-ocr
|
pdfa-2 (output a PDF/A-2b)
|
||||||
--skip-big --jpeg-quality --png-quality --jbig2-lossy
|
pdfa-3 (output a PDF/A-3b)
|
||||||
--max-image-mpixels --tesseract-config --tesseract-pagesegmode
|
none (do not produce an output PDF (for example, if you only care about --sidecar))"
|
||||||
--help --tesseract-oem --pdf-renderer --tesseract-timeout
|
|
||||||
--rotate-pages-threshold --pdfa-image-compression --user-words
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
--user-patterns --keep-temporary-files --output-type
|
|
||||||
--no-progress-bar --pages --fast-web-view' \
|
# Remove description if only one completion exists
|
||||||
-- "$cur" ) )
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
return
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
else
|
|
||||||
_filedir
|
|
||||||
return
|
|
||||||
fi
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
__ocrmypdf_verbose()
|
||||||
|
{
|
||||||
|
local choices="0 (standard output messages)
|
||||||
|
1 (troubleshooting output messages)
|
||||||
|
2 (debugging output messages)"
|
||||||
|
|
||||||
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
|
|
||||||
|
# Remove description if only one completion exists
|
||||||
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
__ocrmypdf_optimize()
|
||||||
|
{
|
||||||
|
local choices="0 (do not optimize)
|
||||||
|
1 (do safe, lossless optimizations (default))
|
||||||
|
2 (do some lossy optimizations)
|
||||||
|
3 (do aggressive lossy optimizations (including lossy JBIG2))"
|
||||||
|
|
||||||
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
|
|
||||||
|
# Remove description if only one completion exists
|
||||||
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
__ocrmypdf_pdf-renderer()
|
||||||
|
{
|
||||||
|
local choices="auto (auto select PDF renderer)
|
||||||
|
hocr (use hOCR renderer)
|
||||||
|
hocrdebug (uses hOCR renderer in debug mode, showing recognized text)
|
||||||
|
sandwich (use sandwich renderer)"
|
||||||
|
|
||||||
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
|
|
||||||
|
# Remove description if only one completion exists
|
||||||
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
__ocrmypdf_pdfa-image-compression()
|
||||||
|
{
|
||||||
|
local choices="auto (let Ghostscript decide how to compress images)
|
||||||
|
jpeg (convert color and grayscale images to JPEG)
|
||||||
|
lossless (convert color and grayscale images to lossless (PNG))"
|
||||||
|
|
||||||
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
|
|
||||||
|
# Remove description if only one completion exists
|
||||||
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
__ocrmypdf_tesseract-pagesegmode()
|
||||||
|
{
|
||||||
|
local choices="0 (orientation and script detection (OSD) only)
|
||||||
|
1 (automatic page segmentation with OSD)
|
||||||
|
2 (automatic page segmentation, but no OSD, or OCR)
|
||||||
|
3 (fully automatic page segmentation, but no OSD (default))
|
||||||
|
4 (assume a single column of text of variable sizes)
|
||||||
|
5 (assume a single uniform block of vertically aligned text)
|
||||||
|
6 (assume a single uniform block of text)
|
||||||
|
7 (treat the image as a single text line)
|
||||||
|
8 (treat the image as a single word)
|
||||||
|
9 (treat the image as a single word in a circle)
|
||||||
|
10 (treat the image as a single character)
|
||||||
|
11 (sparse text - find as much text as possible in no particular order)
|
||||||
|
12 (sparse text with OSD)
|
||||||
|
13 (raw line - treat the image as a single text line)"
|
||||||
|
|
||||||
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
|
|
||||||
|
# Remove description if only one completion exists
|
||||||
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
__ocrmypdf_tesseract-oem()
|
||||||
|
{
|
||||||
|
local choices="0 (legacy engine only)
|
||||||
|
1 (neural nets LSTM engine only)
|
||||||
|
2 (legacy + LSTM engines)
|
||||||
|
3 (default, based on what is available)"
|
||||||
|
|
||||||
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
|
|
||||||
|
# Remove description if only one completion exists
|
||||||
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
__ocrmypdf_tesseract-thresholding()
|
||||||
|
{
|
||||||
|
local choices="auto (let OCRmyPDF pick thresholding - current always uses otsu)
|
||||||
|
otsu (use hOCR renderer)
|
||||||
|
adaptive-otsu (use adaptive Otsu thresholding)
|
||||||
|
sauvola (use Sauvola thresholding)"
|
||||||
|
|
||||||
|
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||||
|
# Remove description if only one completion exists
|
||||||
|
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||||
|
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||||
|
fi
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
__ocrmypdf_check_previous()
|
||||||
|
{
|
||||||
case $prev in
|
case $prev in
|
||||||
--version|-h|--help)
|
-h|--help|--version)
|
||||||
return
|
return 0
|
||||||
;;
|
|
||||||
--user-words|--user-patterns|--tesseract-config)
|
|
||||||
_filedir
|
|
||||||
return
|
|
||||||
;;
|
|
||||||
--output-type)
|
|
||||||
COMPREPLY=( $( compgen -W 'pdfa pdf pdfa-1 pdfa-2 pdfa-3' -- \
|
|
||||||
"$cur" ) )
|
|
||||||
return
|
|
||||||
;;
|
|
||||||
--pdf-renderer)
|
|
||||||
COMPREPLY=( $( compgen -W 'auto hocr sandwich' -- "$cur" ) )
|
|
||||||
return
|
|
||||||
;;
|
|
||||||
--pdfa-image-compression)
|
|
||||||
COMPREPLY=( $( compgen -W 'auto jpeg lossless' -- "$cur" ) )
|
|
||||||
return
|
|
||||||
;;
|
|
||||||
-O|--optimize|--tesseract-oem)
|
|
||||||
COMPREPLY=( $( compgen -W '{0..3}' -- "$cur" ) )
|
|
||||||
return
|
|
||||||
;;
|
|
||||||
--jpeg-quality|--png-quality)
|
|
||||||
COMPREPLY=( $( compgen -W '{0..100}' -- "$cur" ) )
|
|
||||||
return
|
|
||||||
;;
|
;;
|
||||||
-l|--language)
|
-l|--language)
|
||||||
COMPREPLY=$( command tesseract --list-langs 2>/dev/null )
|
COMPREPLY=$( command tesseract --list-langs 2>/dev/null )
|
||||||
COMPREPLY=( $( compgen -W '${COMPREPLY[@]##*:}' -- "$cur" ) )
|
COMPREPLY=( $( compgen -W '${COMPREPLY[@]##*:}' -- "$cur" ) )
|
||||||
return
|
return 0
|
||||||
;;
|
;;
|
||||||
--image-dpi|--oversample|--skip-big|--max-image-mpixels|\
|
--output-type)
|
||||||
--tesseract-timeout|--rotate-pages-threshold)
|
__ocrmypdf_output-type
|
||||||
COMPREPLY=( $( compgen -P "$cur" -W '{0..9}' ) )
|
return 0
|
||||||
return
|
|
||||||
;;
|
;;
|
||||||
-j|--jobs)
|
-j|--jobs)
|
||||||
COMPREPLY=( $( compgen -W '{1..'$( _ncpus )'}' -- "$cur" ) )
|
COMPREPLY=( $( compgen -W '{1..'$( _ncpus )'}' -- "$cur" ) )
|
||||||
return
|
return 0
|
||||||
;;
|
;;
|
||||||
-v|--verbose)
|
-v|--verbose)
|
||||||
COMPREPLY=( $( compgen -W '{0..2}' -- "$cur" ) ) # max level ?
|
__ocrmypdf_verbose
|
||||||
return
|
return 0
|
||||||
|
;;
|
||||||
|
-O|--optimize)
|
||||||
|
__ocrmypdf_optimize
|
||||||
|
return 0
|
||||||
|
;;
|
||||||
|
--pdf-renderer)
|
||||||
|
__ocrmypdf_pdf-renderer
|
||||||
|
return 0
|
||||||
|
;;
|
||||||
|
--pdfa-image-compression)
|
||||||
|
__ocrmypdf_pdfa-image-compression
|
||||||
|
return 0
|
||||||
;;
|
;;
|
||||||
--tesseract-pagesegmode)
|
--tesseract-pagesegmode)
|
||||||
COMPREPLY=( $( compgen -W '{1..13}' -- "$cur" ) )
|
__ocrmypdf_tesseract-pagesegmode
|
||||||
return
|
return 0
|
||||||
;;
|
;;
|
||||||
--sidecar|--title|--author|--subject|--keywords|--unpaper-args|--pages|--fast-web-view)
|
--tesseract-oem)
|
||||||
|
__ocrmypdf_tesseract-oem
|
||||||
|
return 0
|
||||||
|
;;
|
||||||
|
--tesseract-thresholding)
|
||||||
|
__ocrmypdf_tesseract-thresholding
|
||||||
|
return 0
|
||||||
|
;;
|
||||||
|
|
||||||
|
--title|--author|--subject|--keywords|--unpaper-args|--pages|--plugin|\
|
||||||
|
--jpeg-quality|--png-quality|--image-dpi|--oversample|--skip-big|--max-image-mpixels|\
|
||||||
|
--tesseract-timeout|--rotate-pages-threshold|--fast-web-view)
|
||||||
# argument required but no completions available
|
# argument required but no completions available
|
||||||
return
|
return 0
|
||||||
|
;;
|
||||||
|
--tesseract-config|--user-words|--user-patterns|--sidecar)
|
||||||
|
_filedir
|
||||||
|
return 0
|
||||||
;;
|
;;
|
||||||
esac
|
esac
|
||||||
|
|
||||||
$split && return
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
_ocrmypdf()
|
||||||
|
{
|
||||||
|
local OLDIFS="$IFS"
|
||||||
|
local IFS=$'\n'
|
||||||
|
|
||||||
|
local cur prev
|
||||||
|
|
||||||
|
# Homebrew on Macs have version 1.3 of bash-completion which doesn't include - see #502
|
||||||
|
if declare -F _init_completion >/dev/null 2>&1; then
|
||||||
|
_init_completion || return
|
||||||
|
else
|
||||||
|
COMPREPLY=()
|
||||||
|
_get_comp_words_by_ref cur prev
|
||||||
|
fi
|
||||||
|
|
||||||
|
if __ocrmypdf_check_previous -ne 0; then
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "$cur" == -* ]]; then
|
||||||
|
__ocrmypdf_arguments
|
||||||
|
else
|
||||||
|
_filedir
|
||||||
|
fi
|
||||||
|
|
||||||
|
IFS="$OLDIFS"
|
||||||
|
|
||||||
|
return
|
||||||
} &&
|
} &&
|
||||||
complete -F _ocrmypdf ocrmypdf
|
complete -F _ocrmypdf ocrmypdf
|
||||||
|
|
||||||
|
|||||||
@@ -29,7 +29,6 @@ complete -c ocrmypdf -s d -l deskew -d "fix small horizontal alignment skew"
|
|||||||
complete -c ocrmypdf -s c -l clean -d "clean document images before OCR"
|
complete -c ocrmypdf -s c -l clean -d "clean document images before OCR"
|
||||||
complete -c ocrmypdf -s i -l clean-final -d "clean document images and keep result"
|
complete -c ocrmypdf -s i -l clean-final -d "clean document images and keep result"
|
||||||
complete -c ocrmypdf -l remove-vectors -d "don't send vector objects to OCR"
|
complete -c ocrmypdf -l remove-vectors -d "don't send vector objects to OCR"
|
||||||
complete -c ocrmypdf -l threshold -d "threshold images before OCR"
|
|
||||||
|
|
||||||
complete -c ocrmypdf -s f -l force-ocr -d "OCR documents that already have printable text"
|
complete -c ocrmypdf -s f -l force-ocr -d "OCR documents that already have printable text"
|
||||||
complete -c ocrmypdf -s s -l skip-ocr -d "skip OCR on pages that text, otherwise try OCR"
|
complete -c ocrmypdf -s s -l skip-ocr -d "skip OCR on pages that text, otherwise try OCR"
|
||||||
@@ -54,6 +53,7 @@ function __fish_ocrmypdf_output_type
|
|||||||
echo -e "pdfa-1\t"(_ "output a PDF/A-1b")
|
echo -e "pdfa-1\t"(_ "output a PDF/A-1b")
|
||||||
echo -e "pdfa-2\t"(_ "output a PDF/A-2b")
|
echo -e "pdfa-2\t"(_ "output a PDF/A-2b")
|
||||||
echo -e "pdfa-3\t"(_ "output a PDF/A-3b")
|
echo -e "pdfa-3\t"(_ "output a PDF/A-3b")
|
||||||
|
echo -e "none\t"(_ "do not produce an output PDF (for example, if you only care about --sidecar)")
|
||||||
end
|
end
|
||||||
complete -c ocrmypdf -x -l output-type -a '(__fish_ocrmypdf_output_type)' -d "select PDF output options"
|
complete -c ocrmypdf -x -l output-type -a '(__fish_ocrmypdf_output_type)' -d "select PDF output options"
|
||||||
|
|
||||||
@@ -105,20 +105,20 @@ complete -c ocrmypdf -x -l pages -d "apply OCR to only the specified pages"
|
|||||||
complete -c ocrmypdf -x -l tesseract-config -d "set custom tesseract config file"
|
complete -c ocrmypdf -x -l tesseract-config -d "set custom tesseract config file"
|
||||||
|
|
||||||
function __fish_ocrmypdf_tesseract_pagesegmode
|
function __fish_ocrmypdf_tesseract_pagesegmode
|
||||||
echo -e "0\t"(_ "orientation and script detection (OSD) only")
|
echo -e "0\t"(_ "orientation and script detection (OSD) only")
|
||||||
echo -e "1\t"(_ "automatic page segmentation with OSD")
|
echo -e "1\t"(_ "automatic page segmentation with OSD")
|
||||||
echo -e "2\t"(_ "automatic page segmentation, but no OSD, or OCR")
|
echo -e "2\t"(_ "automatic page segmentation, but no OSD, or OCR")
|
||||||
echo -e "3\t"(_ "fully automatic page segmentation, but no OSD (default)")
|
echo -e "3\t"(_ "fully automatic page segmentation, but no OSD (default)")
|
||||||
echo -e "4\t"(_ "assume a single column of text of variable sizes")
|
echo -e "4\t"(_ "assume a single column of text of variable sizes")
|
||||||
echo -e "5\t"(_ "assume a single uniform block of vertically aligned text")
|
echo -e "5\t"(_ "assume a single uniform block of vertically aligned text")
|
||||||
echo -e "6\t"(_ "assume a single uniform block of text")
|
echo -e "6\t"(_ "assume a single uniform block of text")
|
||||||
echo -e "7\t"(_ "treat the image as a single text line")
|
echo -e "7\t"(_ "treat the image as a single text line")
|
||||||
echo -e "8\t"(_ "treat the image as a single word")
|
echo -e "8\t"(_ "treat the image as a single word")
|
||||||
echo -e "9\t"(_ "treat the image as a single word in a circle")
|
echo -e "9\t"(_ "treat the image as a single word in a circle")
|
||||||
echo -e "10\t"(_ "treat the image as a single character")
|
echo -e "10\t"(_ "treat the image as a single character")
|
||||||
echo -e "11\t"(_ "sparse text - find as much text as possible in no particular order")
|
echo -e "11\t"(_ "sparse text - find as much text as possible in no particular order")
|
||||||
echo -e "12\t"(_ "sparse text with OSD")
|
echo -e "12\t"(_ "sparse text with OSD")
|
||||||
echo -e "13\t"(_ "raw line - treat the image as a single text line")
|
echo -e "13\t"(_ "raw line - treat the image as a single text line")
|
||||||
end
|
end
|
||||||
complete -c ocrmypdf -x -l tesseract-pagesegmode -a '(__fish_ocrmypdf_tesseract_pagesegmode)' -d "set tesseract --psm"
|
complete -c ocrmypdf -x -l tesseract-pagesegmode -a '(__fish_ocrmypdf_tesseract_pagesegmode)' -d "set tesseract --psm"
|
||||||
|
|
||||||
@@ -129,6 +129,15 @@ function __fish_ocrmypdf_tesseract_oem
|
|||||||
echo -e "3\t"(_ "default, based on what is available")
|
echo -e "3\t"(_ "default, based on what is available")
|
||||||
end
|
end
|
||||||
complete -c ocrmypdf -x -l tesseract-oem -a '(__fish_ocrmypdf_tesseract_oem)' -d "set tesseract --oem"
|
complete -c ocrmypdf -x -l tesseract-oem -a '(__fish_ocrmypdf_tesseract_oem)' -d "set tesseract --oem"
|
||||||
|
|
||||||
|
function __fish_ocrmypdf_tesseract_thresholding
|
||||||
|
echo -e "auto\t"(_ "let OCRmyPDF pick thresholding (current always uses otsu)")
|
||||||
|
echo -e "otsu\t"(_ "legacy Otsu thresholding")
|
||||||
|
echo -e "adaptive-otsu\t"(_ "use adaptive Otsu thresholding")
|
||||||
|
echo -e "sauvola\t"(_ "use Sauvola thresholding")
|
||||||
|
end
|
||||||
|
complete -c ocrmypdf -x -l tesseract-thresholding -a '(__fish_ocrmypdf_tesseract_thresholding)' -d "set tesseract thresholding method (needs Tesseract 5.x)"
|
||||||
|
|
||||||
complete -c ocrmypdf -x -l tesseract-timeout -d "maximum number of seconds to wait for OCR"
|
complete -c ocrmypdf -x -l tesseract-timeout -d "maximum number of seconds to wait for OCR"
|
||||||
complete -c ocrmypdf -x -l rotate-pages-threshold -d "page rotation confidence"
|
complete -c ocrmypdf -x -l rotate-pages-threshold -d "page rotation confidence"
|
||||||
|
|
||||||
|
|||||||
@@ -37,6 +37,8 @@ To use this as an API:
|
|||||||
)
|
)
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|||||||
+3
-2
@@ -19,8 +19,9 @@
|
|||||||
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
# SOFTWARE.
|
# SOFTWARE.
|
||||||
|
|
||||||
# This script must be edited to meet your needs.
|
from __future__ import annotations
|
||||||
|
|
||||||
|
# This script must be edited to meet your needs.
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
@@ -46,7 +47,7 @@ if len(sys.argv) > 1:
|
|||||||
else:
|
else:
|
||||||
start_dir = '.'
|
start_dir = '.'
|
||||||
|
|
||||||
for dir_name, subdirs, file_list in os.walk(start_dir):
|
for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||||
logging.info(dir_name)
|
logging.info(dir_name)
|
||||||
os.chdir(dir_name)
|
os.chdir(dir_name)
|
||||||
for filename in file_list:
|
for filename in file_list:
|
||||||
|
|||||||
+16
-4
@@ -20,9 +20,12 @@
|
|||||||
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||||
# SOFTWARE.
|
# SOFTWARE.
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
import time
|
import time
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
@@ -44,8 +47,10 @@ def getenv_bool(name: str, default: str = 'False'):
|
|||||||
|
|
||||||
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
|
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
|
||||||
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
|
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
|
||||||
|
ARCHIVE_DIRECTORY = os.getenv('OCR_ARCHIVE_DIRECTORY', '/processed')
|
||||||
OUTPUT_DIRECTORY_YEAR_MONTH = getenv_bool('OCR_OUTPUT_DIRECTORY_YEAR_MONTH')
|
OUTPUT_DIRECTORY_YEAR_MONTH = getenv_bool('OCR_OUTPUT_DIRECTORY_YEAR_MONTH')
|
||||||
ON_SUCCESS_DELETE = getenv_bool('OCR_ON_SUCCESS_DELETE')
|
ON_SUCCESS_DELETE = getenv_bool('OCR_ON_SUCCESS_DELETE')
|
||||||
|
ON_SUCCESS_ARCHIVE = getenv_bool('OCR_ON_SUCCESS_ARCHIVE')
|
||||||
DESKEW = getenv_bool('OCR_DESKEW')
|
DESKEW = getenv_bool('OCR_DESKEW')
|
||||||
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
||||||
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
||||||
@@ -108,9 +113,13 @@ def execute_ocrmypdf(file_path):
|
|||||||
deskew=DESKEW,
|
deskew=DESKEW,
|
||||||
**OCR_JSON_SETTINGS,
|
**OCR_JSON_SETTINGS,
|
||||||
)
|
)
|
||||||
if exit_code == 0 and ON_SUCCESS_DELETE:
|
if exit_code == 0:
|
||||||
log.info(f'OCR is done. Deleting: {file_path}')
|
if ON_SUCCESS_DELETE:
|
||||||
file_path.unlink()
|
log.info(f'OCR is done. Deleting: {file_path}')
|
||||||
|
file_path.unlink()
|
||||||
|
elif ON_SUCCESS_ARCHIVE:
|
||||||
|
log.info(f'OCR is done. Archiving {file_path.name} to {ARCHIVE_DIRECTORY}')
|
||||||
|
shutil.move(file_path, f'{ARCHIVE_DIRECTORY}/{file_path.name}')
|
||||||
else:
|
else:
|
||||||
log.info('OCR is done')
|
log.info('OCR is done')
|
||||||
|
|
||||||
@@ -135,13 +144,16 @@ def main():
|
|||||||
f"Starting OCRmyPDF watcher with config:\n"
|
f"Starting OCRmyPDF watcher with config:\n"
|
||||||
f"Input Directory: {INPUT_DIRECTORY}\n"
|
f"Input Directory: {INPUT_DIRECTORY}\n"
|
||||||
f"Output Directory: {OUTPUT_DIRECTORY}\n"
|
f"Output Directory: {OUTPUT_DIRECTORY}\n"
|
||||||
f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}"
|
f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}\n"
|
||||||
|
f"Archive Directory: {ARCHIVE_DIRECTORY}"
|
||||||
)
|
)
|
||||||
log.debug(
|
log.debug(
|
||||||
f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n"
|
f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n"
|
||||||
f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n"
|
f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n"
|
||||||
f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n"
|
f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n"
|
||||||
|
f"ARCHIVE_DIRECTORY: {ARCHIVE_DIRECTORY}\n"
|
||||||
f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n"
|
f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n"
|
||||||
|
f"ON_SUCCESS_ARCHIVE: {ON_SUCCESS_ARCHIVE}\n"
|
||||||
f"DESKEW: {DESKEW}\n"
|
f"DESKEW: {DESKEW}\n"
|
||||||
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
||||||
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
||||||
|
|||||||
+4
-2
@@ -24,6 +24,8 @@ to emphasize that SaaS deployments should make sure they comply with
|
|||||||
Ghostscript's license as well as OCRmyPDF's.
|
Ghostscript's license as well as OCRmyPDF's.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import os
|
import os
|
||||||
import shlex
|
import shlex
|
||||||
from subprocess import PIPE, run
|
from subprocess import PIPE, run
|
||||||
@@ -37,7 +39,7 @@ app.secret_key = "secret"
|
|||||||
app.config['MAX_CONTENT_LENGTH'] = 50_000_000
|
app.config['MAX_CONTENT_LENGTH'] = 50_000_000
|
||||||
app.config.from_envvar("OCRMYPDF_WEBSERVICE_SETTINGS", silent=True)
|
app.config.from_envvar("OCRMYPDF_WEBSERVICE_SETTINGS", silent=True)
|
||||||
|
|
||||||
ALLOWED_EXTENSIONS = set(["pdf"])
|
ALLOWED_EXTENSIONS = {"pdf"}
|
||||||
|
|
||||||
|
|
||||||
def allowed_file(filename):
|
def allowed_file(filename):
|
||||||
@@ -59,7 +61,7 @@ def do_ocrmypdf(file):
|
|||||||
return Response("--sidecar not supported", 501, mimetype='text/plain')
|
return Response("--sidecar not supported", 501, mimetype='text/plain')
|
||||||
|
|
||||||
ocrmypdf_args = ["ocrmypdf", *cmd_args, up_file, down_file]
|
ocrmypdf_args = ["ocrmypdf", *cmd_args, up_file, down_file]
|
||||||
proc = run(ocrmypdf_args, stdout=PIPE, stderr=PIPE, encoding="utf-8")
|
proc = run(ocrmypdf_args, capture_output=True, encoding="utf-8")
|
||||||
if proc.returncode != 0:
|
if proc.returncode != 0:
|
||||||
stderr = proc.stderr
|
stderr = proc.stderr
|
||||||
return Response(stderr, 400, mimetype='text/plain')
|
return Response(stderr, 400, mimetype='text/plain')
|
||||||
|
|||||||
+53
-22
@@ -1,19 +1,16 @@
|
|||||||
[build-system]
|
[build-system]
|
||||||
requires = [
|
requires = [
|
||||||
"setuptools >= 30.3.0",
|
"setuptools >= 52",
|
||||||
"wheel",
|
"setuptools_scm[toml] >= 7.0.5",
|
||||||
"cffi",
|
"wheel"
|
||||||
"setuptools_scm[toml] >= 3.4",
|
|
||||||
"setuptools_scm_git_archive"
|
|
||||||
]
|
]
|
||||||
build-backend = "setuptools.build_meta"
|
build-backend = "setuptools.build_meta"
|
||||||
|
|
||||||
[tool.setuptools_scm]
|
[tool.setuptools_scm]
|
||||||
version_scheme = "post-release"
|
|
||||||
|
|
||||||
[tool.black]
|
[tool.black]
|
||||||
line-length = 88
|
line-length = 88
|
||||||
target-version = ["py36", "py37", "py38"]
|
target-version = ["py37", "py38"]
|
||||||
skip-string-normalization = true
|
skip-string-normalization = true
|
||||||
include = '\.pyi?$'
|
include = '\.pyi?$'
|
||||||
exclude = '''
|
exclude = '''
|
||||||
@@ -31,7 +28,6 @@ exclude = '''
|
|||||||
| docs
|
| docs
|
||||||
| misc
|
| misc
|
||||||
| \.egg-info
|
| \.egg-info
|
||||||
| src/ocrmypdf/lib/_leptonica.py
|
|
||||||
)/
|
)/
|
||||||
'''
|
'''
|
||||||
|
|
||||||
@@ -46,24 +42,38 @@ source = ["src/ocrmypdf"]
|
|||||||
[tool.coverage.report]
|
[tool.coverage.report]
|
||||||
# Regexes for lines to exclude from consideration
|
# Regexes for lines to exclude from consideration
|
||||||
exclude_lines = [
|
exclude_lines = [
|
||||||
# Have to re-enable the standard pragma
|
# Have to re-enable the standard pragma
|
||||||
"pragma: no cover",
|
"pragma: no cover",
|
||||||
|
# Don't complain if tests don't hit defensive assertion code:
|
||||||
# Don't complain if tests don't hit defensive assertion code:
|
"raise AssertionError",
|
||||||
"raise AssertionError",
|
"raise NotImplementedError",
|
||||||
"raise NotImplementedError",
|
# Don't complain if non-runnable code isn't run:
|
||||||
|
"if 0:",
|
||||||
# Don't complain if non-runnable code isn't run:
|
"if False:",
|
||||||
"if 0:",
|
"if __name__ == .__main__.:",
|
||||||
"if False:",
|
"if TYPE_CHECKING:"
|
||||||
"if __name__ == .__main__.:",
|
|
||||||
"if TYPE_CHECKING:"
|
|
||||||
]
|
]
|
||||||
|
|
||||||
[tool.isort]
|
[tool.isort]
|
||||||
profile = "black"
|
profile = "black"
|
||||||
known_first_party = "ocrmypdf"
|
known_first_party = "ocrmypdf"
|
||||||
known_third_party = ["PIL","_cffi_backend","cffi","flask","img2pdf","pdfminer","pikepdf","pkg_resources","pluggy","pytest","reportlab","setuptools","sphinx_rtd_theme","tqdm","watchdog","werkzeug"]
|
known_third_party = [
|
||||||
|
"PIL",
|
||||||
|
"flask",
|
||||||
|
"img2pdf",
|
||||||
|
"ocrmypdf",
|
||||||
|
"pdfminer",
|
||||||
|
"pikepdf",
|
||||||
|
"pkg_resources",
|
||||||
|
"pluggy",
|
||||||
|
"pytest",
|
||||||
|
"reportlab",
|
||||||
|
"setuptools",
|
||||||
|
"sphinx_rtd_theme",
|
||||||
|
"tqdm",
|
||||||
|
"watchdog",
|
||||||
|
"werkzeug"
|
||||||
|
]
|
||||||
|
|
||||||
[tool.pytest.ini_options]
|
[tool.pytest.ini_options]
|
||||||
minversion = "6.0"
|
minversion = "6.0"
|
||||||
@@ -71,4 +81,25 @@ norecursedirs = ["lib", ".pc", ".git", "venv", "output", "cache", "resources"]
|
|||||||
testpaths = ["tests"]
|
testpaths = ["tests"]
|
||||||
addopts = "-n auto"
|
addopts = "-n auto"
|
||||||
markers = ["slow"]
|
markers = ["slow"]
|
||||||
filterwarnings = ["ignore:.*XMLParser.*:DeprecationWarning"]
|
filterwarnings = ["ignore:.*XMLParser.*:DeprecationWarning"]
|
||||||
|
|
||||||
|
[tool.mypy]
|
||||||
|
|
||||||
|
[[tool.mypy.overrides]]
|
||||||
|
module = [
|
||||||
|
'pluggy',
|
||||||
|
'tqdm',
|
||||||
|
'coloredlogs',
|
||||||
|
'img2pdf',
|
||||||
|
'pdfminer.*',
|
||||||
|
'reportlab.*',
|
||||||
|
'fitz',
|
||||||
|
'libxmp.utils',
|
||||||
|
'importlib_metadata'
|
||||||
|
]
|
||||||
|
ignore_missing_imports = true
|
||||||
|
|
||||||
|
[tool.pylint.basic]
|
||||||
|
good-names = ["i", "j", "k", "ex", "Run", "_", "e", "p", "im", "w", "h", "m", "x", "y", "a", "b", "fp", "n", "f", "s", "v", "q", "dx", "dy"]
|
||||||
|
logging-format-style = "old"
|
||||||
|
disable = ["raw-checker-failed", "bad-inline-option", "locally-disabled", "file-ignored", "suppressed-message", "useless-suppression", "deprecated-pragma", "use-symbolic-message-instead", "logging-fstring-interpolation", "missing-function-docstring", "too-few-public-methods"]
|
||||||
|
|||||||
@@ -1,10 +0,0 @@
|
|||||||
# Deprecated and not maintained; use "pip install ocrmypdf" instead
|
|
||||||
cffi == 1.14.5
|
|
||||||
coloredlogs == 15.0 # technically optional
|
|
||||||
img2pdf == 0.4.0
|
|
||||||
pdfminer.six == 20201018
|
|
||||||
pikepdf == 2.10.0
|
|
||||||
pluggy == 0.13.1
|
|
||||||
Pillow == 8.2.0
|
|
||||||
reportlab == 3.5.66
|
|
||||||
tqdm == 4.59.0
|
|
||||||
@@ -1,7 +0,0 @@
|
|||||||
# Deprecated and not maintained; use "pip install ocrmypdf[test]" instead
|
|
||||||
pytest >= 6.0.0
|
|
||||||
pytest-xdist >= 2.2.0
|
|
||||||
pytest-cov >= 2.11.1
|
|
||||||
python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3
|
|
||||||
# or brew install exempi
|
|
||||||
#PyMuPDF == 1.13.4 # optional
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
# Deprecated and not maintained; use "pip install ocrmypdf[watcher]" instead
|
|
||||||
watchdog == 1.0.2
|
|
||||||
@@ -1,2 +0,0 @@
|
|||||||
# Deprecated and not maintained; use "pip install ocrmypdf[webservice]" instead
|
|
||||||
Flask >= 1, < 2
|
|
||||||
@@ -2,23 +2,15 @@
|
|||||||
name = ocrmypdf
|
name = ocrmypdf
|
||||||
description = OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched
|
description = OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched
|
||||||
long_description = file: README.md
|
long_description = file: README.md
|
||||||
long_description_content_type = text/markdown; charset=UTF-8
|
long_description_content_type = text/markdown
|
||||||
url = https://github.com/jbarlow83/OCRmyPDF
|
url = https://github.com/ocrmypdf/OCRmyPDF
|
||||||
author = James R. Barlow
|
author = James R. Barlow
|
||||||
author_email = james@purplerock.ca
|
author_email = james@purplerock.ca
|
||||||
|
license = MPL-2.0
|
||||||
|
license_file = LICENSE
|
||||||
license_files =
|
license_files =
|
||||||
LICENSE
|
LICENSE
|
||||||
keywords =
|
|
||||||
PDF
|
|
||||||
OCR
|
|
||||||
optical character recognition
|
|
||||||
PDF/A
|
|
||||||
scanning
|
|
||||||
classifiers =
|
classifiers =
|
||||||
Programming Language :: Python :: 3.6
|
|
||||||
Programming Language :: Python :: 3.7
|
|
||||||
Programming Language :: Python :: 3.8
|
|
||||||
Programming Language :: Python :: 3.9
|
|
||||||
Development Status :: 5 - Production/Stable
|
Development Status :: 5 - Production/Stable
|
||||||
Environment :: Console
|
Environment :: Console
|
||||||
Intended Audience :: End Users/Desktop
|
Intended Audience :: End Users/Desktop
|
||||||
@@ -30,75 +22,95 @@ classifiers =
|
|||||||
Operating System :: POSIX
|
Operating System :: POSIX
|
||||||
Operating System :: POSIX :: BSD
|
Operating System :: POSIX :: BSD
|
||||||
Operating System :: POSIX :: Linux
|
Operating System :: POSIX :: Linux
|
||||||
|
Programming Language :: Python :: 3
|
||||||
|
Programming Language :: Python :: 3 :: Only
|
||||||
|
Programming Language :: Python :: 3.7
|
||||||
|
Programming Language :: Python :: 3.8
|
||||||
|
Programming Language :: Python :: 3.9
|
||||||
|
Programming Language :: Python :: 3.10
|
||||||
Topic :: Scientific/Engineering :: Image Recognition
|
Topic :: Scientific/Engineering :: Image Recognition
|
||||||
Topic :: Text Processing :: Indexing
|
Topic :: Text Processing :: Indexing
|
||||||
Topic :: Text Processing :: Linguistic
|
Topic :: Text Processing :: Linguistic
|
||||||
|
keywords =
|
||||||
|
PDF
|
||||||
|
OCR
|
||||||
|
optical character recognition
|
||||||
|
PDF/A
|
||||||
|
scanning
|
||||||
project_urls =
|
project_urls =
|
||||||
Documentation = https://ocrmypdf.readthedocs.io/
|
Documentation = https://ocrmypdf.readthedocs.io/
|
||||||
Source = https://github.com/jbarlow83/ocrmypdf
|
Source = https://github.com/ocrmypdf/OCRmyPDF
|
||||||
Tracker = https://github.com/jbarlow83/ocrmypdf/issues
|
Tracker = https://github.com/ocrmypdf/OCRmyPDF/issues
|
||||||
|
|
||||||
[options]
|
[options]
|
||||||
zip_safe = False
|
|
||||||
packages = find:
|
packages = find:
|
||||||
|
install_requires =
|
||||||
|
Pillow>=8.2.0
|
||||||
|
coloredlogs>=14.0 # strictly optional
|
||||||
|
img2pdf>=0.3.0 # pure Python
|
||||||
|
packaging>=20
|
||||||
|
pdfminer.six!=20200720,>=20191110
|
||||||
|
pikepdf!=5.0.0,>=4.0.0
|
||||||
|
pluggy>=0.13.0
|
||||||
|
reportlab>=3.5.66
|
||||||
|
tqdm>=4
|
||||||
|
importlib-metadata>=4;python_version<'3.8' # until Python 3.8
|
||||||
|
importlib-resources>=5;python_version<'3.9' # until Python 3.9
|
||||||
|
typing-extensions>=4;python_version<'3.10'
|
||||||
|
python_requires = >=3.7
|
||||||
|
include_package_data = True
|
||||||
package_dir =
|
package_dir =
|
||||||
=src
|
=src
|
||||||
platforms = any
|
platforms = any
|
||||||
include_package_data=True
|
setup_requires =
|
||||||
install_requires =
|
setuptools-scm
|
||||||
cffi >= 1.9.1 # must be a setup and install requirement
|
setuptools-scm-git-archive
|
||||||
coloredlogs >= 14.0 # strictly optional
|
zip_safe = False
|
||||||
img2pdf >= 0.3.0, < 0.5 # pure Python, so track HEAD closely
|
|
||||||
pdfminer.six >= 20191110, != 20200720, <= 20201018
|
[options.packages.find]
|
||||||
pikepdf >= 2.10.0
|
where = src
|
||||||
Pillow >= 8.2.0
|
|
||||||
pluggy >= 0.13.0, < 1.0
|
[options.entry_points]
|
||||||
reportlab >= 3.5.66
|
console_scripts =
|
||||||
setuptools
|
ocrmypdf = ocrmypdf.__main__:run
|
||||||
tqdm >= 4
|
|
||||||
python_requires = >= 3.6
|
[options.extras_require]
|
||||||
setup_requires = # can be removed whenever we can drop pip 9 support
|
docs =
|
||||||
cffi >= 1.9.1 # to build the leptonica module
|
sphinx
|
||||||
setuptools_scm # so that version will work
|
sphinx-issues
|
||||||
setuptools_scm_git_archive # enable version from github tarballs
|
sphinx-rtd-theme
|
||||||
|
extended_test =
|
||||||
|
PyMuPDF==1.19.1
|
||||||
|
test =
|
||||||
|
coverage[toml]>=5
|
||||||
|
pytest>=6.0.0
|
||||||
|
pytest-cov>=2.11.1
|
||||||
|
pytest-xdist>=2.2.0
|
||||||
|
python-xmp-toolkit==2.0.1 # also requires apt-get install libexempi3
|
||||||
|
types-Pillow
|
||||||
|
types-humanfriendly
|
||||||
|
watcher =
|
||||||
|
watchdog>=1.0.2
|
||||||
|
webservice =
|
||||||
|
Flask>=1
|
||||||
|
|
||||||
[options.package_data]
|
[options.package_data]
|
||||||
ocrmypdf =
|
ocrmypdf =
|
||||||
data/sRGB.icc
|
data/sRGB.icc
|
||||||
py.typed
|
py.typed
|
||||||
|
|
||||||
[options.packages.find]
|
|
||||||
where = src
|
|
||||||
|
|
||||||
[options.extras_require]
|
|
||||||
test =
|
|
||||||
coverage[toml] >= 5
|
|
||||||
pytest >= 6.0.0
|
|
||||||
pytest-xdist >= 2.2.0
|
|
||||||
pytest-cov >= 2.11.1
|
|
||||||
python-xmp-toolkit == 2.0.1 # also requires apt-get install libexempi3
|
|
||||||
# or brew install exempi
|
|
||||||
docs =
|
|
||||||
sphinx
|
|
||||||
sphinx-rtd-theme
|
|
||||||
sphinx-issues
|
|
||||||
extended_test =
|
|
||||||
PyMuPDF == 1.13.4
|
|
||||||
watcher =
|
|
||||||
watchdog >= 1.0.2, < 3
|
|
||||||
webservice =
|
|
||||||
Flask >= 1, < 3
|
|
||||||
|
|
||||||
[options.entry_points]
|
|
||||||
console_scripts =
|
|
||||||
ocrmypdf = ocrmypdf.__main__:run
|
|
||||||
|
|
||||||
[bdist_wheel]
|
[bdist_wheel]
|
||||||
python-tag = py36
|
python-tag = py37
|
||||||
|
|
||||||
[aliases]
|
[aliases]
|
||||||
test = pytest
|
test = pytest
|
||||||
|
|
||||||
[check-manifest]
|
[check-manifest]
|
||||||
ignore =
|
ignore =
|
||||||
.github
|
.github
|
||||||
|
|
||||||
|
[flake8]
|
||||||
|
ignore = D203,F401,W503,E501,E203,F841
|
||||||
|
exclude = .git,__pycache__,docs/conf.py,build,dist,.venv,.venvpp,.eggs,tmp,src/ocrmypdf/lib/
|
||||||
|
max-complexity = 10
|
||||||
|
max-line-length = 100
|
||||||
|
|||||||
@@ -4,16 +4,10 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""setup.py to support older setuptools and pip."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
from setuptools import setup
|
from setuptools import setup
|
||||||
|
|
||||||
# Minimal setup to support older setuptools/setuptools_scm
|
setup()
|
||||||
setup(
|
|
||||||
setup_requires=[ # can be removed whenever we can drop pip 9 support
|
|
||||||
'cffi >= 1.9.1', # to build the leptonica module
|
|
||||||
'setuptools_scm', # so that version will work
|
|
||||||
'setuptools_scm_git_archive', # enable version from github tarballs
|
|
||||||
],
|
|
||||||
use_scm_version={'version_scheme': 'post-release'},
|
|
||||||
cffi_modules=['src/ocrmypdf/lib/compile_leptonica.py:ffibuilder'],
|
|
||||||
)
|
|
||||||
|
|||||||
@@ -0,0 +1,72 @@
|
|||||||
|
name: ocrmypdf
|
||||||
|
title: OCRmyPDF
|
||||||
|
base: core20
|
||||||
|
version: git
|
||||||
|
summary: OCRmyPDF adds optical character recognition (OCR) to PDFs
|
||||||
|
description: OCRmyPDF packaged for snap
|
||||||
|
grade: stable
|
||||||
|
confinement: strict
|
||||||
|
icon: docs/images/logo-square-256.svg
|
||||||
|
license: MPL-2.0
|
||||||
|
|
||||||
|
architectures: [amd64]
|
||||||
|
|
||||||
|
environment:
|
||||||
|
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
||||||
|
GS_LIB: $SNAP/usr/share/ghostscript/9.50/Resource/Init
|
||||||
|
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.50/Resource/Font
|
||||||
|
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
||||||
|
|
||||||
|
apps:
|
||||||
|
ocrmypdf:
|
||||||
|
command: usr/bin/snapcraft-preload python3 -m ocrmypdf
|
||||||
|
plugs:
|
||||||
|
- desktop
|
||||||
|
- desktop-legacy
|
||||||
|
- wayland
|
||||||
|
- x11
|
||||||
|
- home
|
||||||
|
- removable-media
|
||||||
|
|
||||||
|
parts:
|
||||||
|
snapcraft-preload:
|
||||||
|
source: https://github.com/sergiusens/snapcraft-preload.git
|
||||||
|
plugin: cmake
|
||||||
|
cmake-parameters:
|
||||||
|
- -DCMAKE_INSTALL_PREFIX=/usr -DLIBPATH=/usr/lib
|
||||||
|
build-packages:
|
||||||
|
- on amd64:
|
||||||
|
- gcc-multilib
|
||||||
|
- g++-multilib
|
||||||
|
stage-packages:
|
||||||
|
- lib32stdc++6
|
||||||
|
|
||||||
|
ocrmypdf:
|
||||||
|
plugin: python
|
||||||
|
source: https://github.com/ocrmypdf/OCRmyPDF.git
|
||||||
|
|
||||||
|
stage-packages:
|
||||||
|
- ghostscript
|
||||||
|
- icc-profiles-free
|
||||||
|
- liblept5
|
||||||
|
- libxml2
|
||||||
|
- pngquant
|
||||||
|
- tesseract-ocr-all
|
||||||
|
- unpaper
|
||||||
|
- qpdf
|
||||||
|
- zlib1g
|
||||||
|
|
||||||
|
python-packages:
|
||||||
|
- cffi
|
||||||
|
- pdfminer.six
|
||||||
|
- pikepdf
|
||||||
|
- Pillow
|
||||||
|
- pluggy
|
||||||
|
- reportlab
|
||||||
|
- setuptools
|
||||||
|
- tqdm
|
||||||
|
- pipe
|
||||||
|
|
||||||
|
override-build: |
|
||||||
|
snapcraftctl build
|
||||||
|
ln -sf ../usr/lib/libsnapcraft-preload.so $SNAPCRAFT_PART_INSTALL/lib/libsnapcraft-preload.so
|
||||||
@@ -4,10 +4,13 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Adds OCR layer to PDFs."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
from pluggy import HookimplMarker as _HookimplMarker
|
from pluggy import HookimplMarker as _HookimplMarker
|
||||||
|
|
||||||
from ocrmypdf import helpers, hocrtransform, leptonica, pdfa, pdfinfo
|
from ocrmypdf import helpers, hocrtransform, pdfa, pdfinfo
|
||||||
from ocrmypdf._concurrent import Executor
|
from ocrmypdf._concurrent import Executor
|
||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._version import PROGRAM_NAME, __version__
|
from ocrmypdf._version import PROGRAM_NAME, __version__
|
||||||
|
|||||||
@@ -5,17 +5,21 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""ocrmypdf command line entrypoint."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import signal
|
import signal
|
||||||
import sys
|
import sys
|
||||||
|
from contextlib import suppress
|
||||||
from multiprocessing import set_start_method
|
from multiprocessing import set_start_method
|
||||||
|
|
||||||
from ocrmypdf import __version__
|
from ocrmypdf import __version__
|
||||||
from ocrmypdf._plugin_manager import get_parser_options_plugins
|
from ocrmypdf._plugin_manager import get_parser_options_plugins
|
||||||
from ocrmypdf._sync import run_pipeline
|
from ocrmypdf._sync import run_pipeline
|
||||||
from ocrmypdf._validation import check_closed_streams, check_options
|
from ocrmypdf._validation import check_options
|
||||||
from ocrmypdf.api import Verbosity, configure_logging
|
from ocrmypdf.api import Verbosity, configure_logging
|
||||||
from ocrmypdf.exceptions import (
|
from ocrmypdf.exceptions import (
|
||||||
BadArgsError,
|
BadArgsError,
|
||||||
@@ -34,10 +38,7 @@ def sigbus(*args):
|
|||||||
def run(args=None):
|
def run(args=None):
|
||||||
_parser, options, plugin_manager = get_parser_options_plugins(args=args)
|
_parser, options, plugin_manager = get_parser_options_plugins(args=args)
|
||||||
|
|
||||||
if not check_closed_streams(options):
|
with suppress(AttributeError, PermissionError):
|
||||||
return ExitCode.bad_args
|
|
||||||
|
|
||||||
if hasattr(os, 'nice'):
|
|
||||||
os.nice(5)
|
os.nice(5)
|
||||||
|
|
||||||
verbosity = options.verbose
|
verbosity = options.verbose
|
||||||
@@ -65,7 +66,7 @@ def run(args=None):
|
|||||||
log.error(e)
|
log.error(e)
|
||||||
return ExitCode.missing_dependency
|
return ExitCode.missing_dependency
|
||||||
|
|
||||||
if hasattr(signal, 'SIGBUS'):
|
with suppress(AttributeError, OSError):
|
||||||
signal.signal(signal.SIGBUS, sigbus)
|
signal.signal(signal.SIGBUS, sigbus)
|
||||||
|
|
||||||
result = run_pipeline(options=options, plugin_manager=plugin_manager)
|
result = run_pipeline(options=options, plugin_manager=plugin_manager)
|
||||||
|
|||||||
@@ -4,9 +4,13 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""OCRmyPDF concurrency abstractions."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import threading
|
import threading
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
from typing import Callable, Iterable, Optional
|
from typing import Callable, Iterable
|
||||||
|
|
||||||
|
|
||||||
def _task_noop(*_args, **_kwargs):
|
def _task_noop(*_args, **_kwargs):
|
||||||
@@ -14,6 +18,8 @@ def _task_noop(*_args, **_kwargs):
|
|||||||
|
|
||||||
|
|
||||||
class NullProgressBar:
|
class NullProgressBar:
|
||||||
|
"""Progress bar API that takes no actions."""
|
||||||
|
|
||||||
def __init__(self, **kwargs):
|
def __init__(self, **kwargs):
|
||||||
pass
|
pass
|
||||||
|
|
||||||
@@ -28,6 +34,8 @@ class NullProgressBar:
|
|||||||
|
|
||||||
|
|
||||||
class Executor(ABC):
|
class Executor(ABC):
|
||||||
|
"""Abstract concurrent executor."""
|
||||||
|
|
||||||
pool_lock = threading.Lock()
|
pool_lock = threading.Lock()
|
||||||
pbar_class = NullProgressBar
|
pbar_class = NullProgressBar
|
||||||
|
|
||||||
@@ -41,10 +49,10 @@ class Executor(ABC):
|
|||||||
use_threads: bool,
|
use_threads: bool,
|
||||||
max_workers: int,
|
max_workers: int,
|
||||||
tqdm_kwargs: dict,
|
tqdm_kwargs: dict,
|
||||||
worker_initializer: Optional[Callable] = None,
|
worker_initializer: Callable | None = None,
|
||||||
task: Optional[Callable] = None,
|
task: Callable | None = None,
|
||||||
task_arguments: Optional[Iterable] = None,
|
task_arguments: Iterable | None = None,
|
||||||
task_finished: Optional[Callable] = None,
|
task_finished: Callable | None = None,
|
||||||
) -> None:
|
) -> None:
|
||||||
"""
|
"""
|
||||||
Set up parallel execution and progress reporting.
|
Set up parallel execution and progress reporting.
|
||||||
|
|||||||
@@ -6,3 +6,5 @@
|
|||||||
|
|
||||||
|
|
||||||
"""Manage third party executables"""
|
"""Manage third party executables"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|||||||
@@ -7,46 +7,44 @@
|
|||||||
|
|
||||||
"""Interface to Ghostscript executable"""
|
"""Interface to Ghostscript executable"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
|
import sys
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import which
|
|
||||||
from subprocess import PIPE, CalledProcessError
|
from subprocess import PIPE, CalledProcessError
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image, UnidentifiedImageError
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
from ocrmypdf.exceptions import SubprocessOutputError
|
||||||
from ocrmypdf.helpers import Resolution
|
from ocrmypdf.helpers import Resolution
|
||||||
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
||||||
|
|
||||||
|
# Remove this workaround when we require Pillow >= 10
|
||||||
|
try:
|
||||||
|
Transpose = Image.Transpose # type: ignore
|
||||||
|
except AttributeError:
|
||||||
|
# Pillow 9 shim
|
||||||
|
Transpose = Image # type: ignore
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
missing_gs_error = """
|
# Most reliable what to get the bitness of Python interpreter, according to Python docs
|
||||||
---------------------------------------------------------------------
|
_IS_64BIT = sys.maxsize > 2**32
|
||||||
This error normally occurs when ocrmypdf find can't Ghostscript.
|
|
||||||
Please ensure Ghostscript is installed and its location is added to
|
|
||||||
the system PATH environment variable.
|
|
||||||
|
|
||||||
For details see:
|
_GSWIN = None
|
||||||
https://ocrmypdf.readthedocs.io/en/latest/installation.html
|
|
||||||
---------------------------------------------------------------------
|
|
||||||
"""
|
|
||||||
|
|
||||||
_gswin = None
|
|
||||||
if os.name == 'nt':
|
if os.name == 'nt':
|
||||||
_gswin = which('gswin64c')
|
if _IS_64BIT:
|
||||||
if not _gswin:
|
_GSWIN = 'gswin64c'
|
||||||
_gswin = which('gswin32c')
|
else:
|
||||||
if not _gswin:
|
_GSWIN = 'gswin32c'
|
||||||
raise MissingDependencyError(missing_gs_error)
|
|
||||||
_gswin = Path(_gswin).stem
|
|
||||||
|
|
||||||
GS = _gswin if _gswin else 'gs'
|
GS = _GSWIN if _GSWIN else 'gs'
|
||||||
del _gswin
|
del _GSWIN
|
||||||
|
|
||||||
|
|
||||||
def version():
|
def version():
|
||||||
@@ -69,7 +67,8 @@ def jpeg_passthrough_available() -> bool:
|
|||||||
|
|
||||||
|
|
||||||
def _gs_error_reported(stream) -> bool:
|
def _gs_error_reported(stream) -> bool:
|
||||||
return True if re.search(r'error', stream, flags=re.IGNORECASE) else False
|
match = re.search(r'error', stream, flags=re.IGNORECASE)
|
||||||
|
return bool(match)
|
||||||
|
|
||||||
|
|
||||||
def rasterize_pdf(
|
def rasterize_pdf(
|
||||||
@@ -79,8 +78,8 @@ def rasterize_pdf(
|
|||||||
raster_device: str,
|
raster_device: str,
|
||||||
raster_dpi: Resolution,
|
raster_dpi: Resolution,
|
||||||
pageno: int = 1,
|
pageno: int = 1,
|
||||||
page_dpi: Optional[Resolution] = None,
|
page_dpi: Resolution | None = None,
|
||||||
rotation: Optional[int] = None,
|
rotation: int | None = None,
|
||||||
filter_vector: bool = False,
|
filter_vector: bool = False,
|
||||||
):
|
):
|
||||||
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units."""
|
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units."""
|
||||||
@@ -95,6 +94,7 @@ def rasterize_pdf(
|
|||||||
'-dSAFER',
|
'-dSAFER',
|
||||||
'-dBATCH',
|
'-dBATCH',
|
||||||
'-dNOPAUSE',
|
'-dNOPAUSE',
|
||||||
|
'-dInterpolateControl=-1',
|
||||||
f'-sDEVICE={raster_device}',
|
f'-sDEVICE={raster_device}',
|
||||||
f'-dFirstPage={pageno}',
|
f'-dFirstPage={pageno}',
|
||||||
f'-dLastPage={pageno}',
|
f'-dLastPage={pageno}',
|
||||||
@@ -104,7 +104,7 @@ def rasterize_pdf(
|
|||||||
+ [
|
+ [
|
||||||
'-o',
|
'-o',
|
||||||
'-',
|
'-',
|
||||||
'-sstdout=%stderr',
|
'-sstdout=%stderr', # Literal %s, not string interpolation
|
||||||
'-dAutoRotatePages=/None', # Probably has no effect on raster
|
'-dAutoRotatePages=/None', # Probably has no effect on raster
|
||||||
'-f',
|
'-f',
|
||||||
fspath(input_file),
|
fspath(input_file),
|
||||||
@@ -115,29 +115,38 @@ def rasterize_pdf(
|
|||||||
p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True)
|
p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
log.error(e.stderr.decode(errors='replace'))
|
log.error(e.stderr.decode(errors='replace'))
|
||||||
raise SubprocessOutputError('Ghostscript rasterizing failed')
|
raise SubprocessOutputError('Ghostscript rasterizing failed') from e
|
||||||
else:
|
else:
|
||||||
stderr = p.stderr.decode(errors='replace')
|
stderr = p.stderr.decode(errors='replace')
|
||||||
if _gs_error_reported(stderr):
|
if _gs_error_reported(stderr):
|
||||||
log.error(stderr)
|
log.error(stderr)
|
||||||
|
|
||||||
with Image.open(BytesIO(p.stdout)) as im:
|
try:
|
||||||
if rotation is not None:
|
with Image.open(BytesIO(p.stdout)) as im:
|
||||||
log.debug("Rotating output by %i", rotation)
|
if rotation is not None:
|
||||||
# rotation is a clockwise angle and Image.ROTATE_* is
|
log.debug("Rotating output by %i", rotation)
|
||||||
# counterclockwise so this cancels out the rotation
|
# rotation is a clockwise angle and Image.ROTATE_* is
|
||||||
if rotation == 90:
|
# counterclockwise so this cancels out the rotation
|
||||||
im = im.transpose(Image.ROTATE_90)
|
if rotation == 90:
|
||||||
elif rotation == 180:
|
im = im.transpose(Transpose.ROTATE_90)
|
||||||
im = im.transpose(Image.ROTATE_180)
|
elif rotation == 180:
|
||||||
elif rotation == 270:
|
im = im.transpose(Transpose.ROTATE_180)
|
||||||
im = im.transpose(Image.ROTATE_270)
|
elif rotation == 270:
|
||||||
if rotation % 180 == 90:
|
im = im.transpose(Transpose.ROTATE_270)
|
||||||
page_dpi = page_dpi.flip_axis()
|
if rotation % 180 == 90:
|
||||||
im.save(fspath(output_file), dpi=page_dpi)
|
page_dpi = page_dpi.flip_axis()
|
||||||
|
im.save(fspath(output_file), dpi=page_dpi)
|
||||||
|
except UnidentifiedImageError:
|
||||||
|
log.error(
|
||||||
|
f"Ghostscript (using {raster_device} at {raster_dpi} dpi) produced "
|
||||||
|
"an invalid page image file."
|
||||||
|
)
|
||||||
|
raise
|
||||||
|
|
||||||
|
|
||||||
class GhostscriptFollower:
|
class GhostscriptFollower:
|
||||||
|
"""Parses the output of Ghostscript and uses it to update the progress bar."""
|
||||||
|
|
||||||
re_process = re.compile(r"Processing pages \d+ through (\d+).")
|
re_process = re.compile(r"Processing pages \d+ through (\d+).")
|
||||||
re_page = re.compile(r"Page (\d+)")
|
re_page = re.compile(r"Page (\d+)")
|
||||||
|
|
||||||
@@ -158,8 +167,7 @@ class GhostscriptFollower:
|
|||||||
)
|
)
|
||||||
return
|
return
|
||||||
else:
|
else:
|
||||||
m = self.re_page.match(line.strip())
|
if self.re_page.match(line.strip()):
|
||||||
if m:
|
|
||||||
self.progressbar.update()
|
self.progressbar.update()
|
||||||
|
|
||||||
|
|
||||||
@@ -200,14 +208,18 @@ def generate_pdfa(
|
|||||||
# Older versions of Ghostscript expect a leading slash in
|
# Older versions of Ghostscript expect a leading slash in
|
||||||
# sColorConversionStrategy, newer ones should not have it. See Ghostscript
|
# sColorConversionStrategy, newer ones should not have it. See Ghostscript
|
||||||
# git commit fe1c025d.
|
# git commit fe1c025d.
|
||||||
strategy = ('/' + strategy) if version() < '9.19' else strategy
|
gs_version = version()
|
||||||
|
strategy = ('/' + strategy) if gs_version < '9.19' else strategy
|
||||||
|
|
||||||
if version() == '9.23':
|
if gs_version == '9.23':
|
||||||
# 9.23: added JPEG passthrough as a new feature, but with a bug that
|
# 9.23: added JPEG passthrough as a new feature, but with a bug that
|
||||||
# incorrectly formats some images. Fixed as of 9.24. So we disable this
|
# incorrectly formats some images. Fixed as of 9.24. So we disable this
|
||||||
# feature for 9.23.
|
# feature for 9.23.
|
||||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
# https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
||||||
compression_args.append('-dPassThroughJPEGImages=false')
|
compression_args.append('-dPassThroughJPEGImages=false')
|
||||||
|
elif gs_version == '9.56.0':
|
||||||
|
# 9.56.0 breaks our OCR...?
|
||||||
|
compression_args.append('-dNEWPDF=false')
|
||||||
|
|
||||||
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
||||||
# is set; see:
|
# is set; see:
|
||||||
@@ -230,7 +242,7 @@ def generate_pdfa(
|
|||||||
"-dPDFACompatibilityPolicy=1",
|
"-dPDFACompatibilityPolicy=1",
|
||||||
"-o",
|
"-o",
|
||||||
"-",
|
"-",
|
||||||
"-sstdout=%stderr",
|
"-sstdout=%stderr", # Literal %s, not string interpolation
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||||
|
|||||||
@@ -7,6 +7,8 @@
|
|||||||
|
|
||||||
"""Interface to jbig2 executable"""
|
"""Interface to jbig2 executable"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
from subprocess import PIPE
|
from subprocess import PIPE
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
|
|||||||
@@ -7,6 +7,8 @@
|
|||||||
|
|
||||||
"""Interface to pngquant executable"""
|
"""Interface to pngquant executable"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|||||||
+147
-62
@@ -7,17 +7,16 @@
|
|||||||
|
|
||||||
"""Interface to Tesseract executable"""
|
"""Interface to Tesseract executable"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
|
||||||
import re
|
import re
|
||||||
import shutil
|
from math import pi
|
||||||
from collections import namedtuple
|
|
||||||
from distutils.version import StrictVersion
|
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
||||||
from typing import List, Optional
|
|
||||||
|
|
||||||
|
from packaging.version import Version
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf.exceptions import (
|
from ocrmypdf.exceptions import (
|
||||||
@@ -25,11 +24,11 @@ from ocrmypdf.exceptions import (
|
|||||||
SubprocessOutputError,
|
SubprocessOutputError,
|
||||||
TesseractConfigError,
|
TesseractConfigError,
|
||||||
)
|
)
|
||||||
|
from ocrmypdf.pluginspec import OrientationConfidence
|
||||||
from ocrmypdf.subprocess import get_version, run
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence'))
|
|
||||||
|
|
||||||
HOCR_TEMPLATE = """<?xml version="1.0" encoding="UTF-8"?>
|
HOCR_TEMPLATE = """<?xml version="1.0" encoding="UTF-8"?>
|
||||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||||
@@ -48,35 +47,74 @@ HOCR_TEMPLATE = """<?xml version="1.0" encoding="UTF-8"?>
|
|||||||
</html>
|
</html>
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
TESSERACT_THRESHOLDING_METHODS: dict[str, int] = {
|
||||||
|
'auto': 0,
|
||||||
|
'otsu': 0,
|
||||||
|
'adaptive-otsu': 1,
|
||||||
|
'sauvola': 2,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
class TesseractLoggerAdapter(logging.LoggerAdapter):
|
class TesseractLoggerAdapter(logging.LoggerAdapter):
|
||||||
|
"Prepend [tesseract] to messages emitted from tesseract"
|
||||||
|
|
||||||
def process(self, msg, kwargs):
|
def process(self, msg, kwargs):
|
||||||
kwargs['extra'] = self.extra
|
kwargs['extra'] = self.extra
|
||||||
return '[tesseract] %s' % (msg), kwargs
|
return f'[tesseract] {msg}', kwargs
|
||||||
|
|
||||||
|
|
||||||
class TesseractVersion(StrictVersion):
|
TESSERACT_VERSION_PATTERN = r"""
|
||||||
|
v?
|
||||||
|
(?:
|
||||||
|
(?:(?P<epoch>[0-9]+)!)? # epoch
|
||||||
|
(?P<release>[0-9]+(?:\.[0-9]+)*) # release segment
|
||||||
|
(?P<pre> # pre-release
|
||||||
|
[-_\.]?
|
||||||
|
(?P<pre_l>(a|b|c|rc|alpha|beta|pre|preview))
|
||||||
|
[-_\.]?
|
||||||
|
(?P<pre_n>[0-9]+)?
|
||||||
|
)?
|
||||||
|
(?P<post> # post release
|
||||||
|
(?:-(?P<post_n1>[0-9]+))
|
||||||
|
|
|
||||||
|
(?:
|
||||||
|
[-_\.]?
|
||||||
|
(?P<post_l>post|rev|r)
|
||||||
|
[-_\.]?
|
||||||
|
(?P<post_n2>[0-9]+)?
|
||||||
|
)
|
||||||
|
)?
|
||||||
|
(?P<dev> # dev release
|
||||||
|
[-_\.]?
|
||||||
|
(?P<dev_l>dev)
|
||||||
|
[-_\.]?
|
||||||
|
(?P<dev_n>[0-9]+)?
|
||||||
|
)?
|
||||||
|
(?P<date>
|
||||||
|
[-_\.]
|
||||||
|
(?:20[0-9][0-9] [0-1][0-9] [0-3][0-9]) # yyyy mm dd
|
||||||
|
)?
|
||||||
|
(?P<gitcount>
|
||||||
|
[-_\.]?
|
||||||
|
[0-9]+
|
||||||
|
)?
|
||||||
|
(?P<gitcommit>
|
||||||
|
[-_\.]?
|
||||||
|
g[0-9a-f]{2,10}
|
||||||
|
)?
|
||||||
|
)
|
||||||
|
(?:\+(?P<local>[a-z0-9]+(?:[-_\.][a-z0-9]+)*))? # local version
|
||||||
|
"""
|
||||||
|
|
||||||
version_re = re.compile(
|
|
||||||
r'''
|
class TesseractVersion(Version):
|
||||||
^(\d+) \. (\d+) (\. (\d+))? # groups: 1/major, 2/minor, 3/[skip], 4/patch
|
"Modify standard packaging.Version regex to support Tesseract idiosyncracies."
|
||||||
[-]? # optional hyphen separator
|
_regex = re.compile(
|
||||||
(?:(alpha|beta|rc|dev)?[.\-\ ]?(\d+)?)? # 5/prerelease, 6/prerelease_num
|
r"^\s*" + TESSERACT_VERSION_PATTERN + r"\s*$", re.VERBOSE | re.IGNORECASE
|
||||||
(?:-(\d+)-g[0-9a-f]+)? # untagged git version
|
|
||||||
$
|
|
||||||
''',
|
|
||||||
re.VERBOSE | re.ASCII,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
def parse(self, vstring):
|
|
||||||
try:
|
|
||||||
super().parse(vstring)
|
|
||||||
except TypeError as e:
|
|
||||||
if 'int() argument must be a string' in str(e):
|
|
||||||
super().parse(vstring + '0')
|
|
||||||
|
|
||||||
|
def version() -> str:
|
||||||
def version():
|
|
||||||
return get_version('tesseract', regex=r'tesseract\s(.+)')
|
return get_version('tesseract', regex=r'tesseract\s(.+)')
|
||||||
|
|
||||||
|
|
||||||
@@ -89,6 +127,11 @@ def has_user_words():
|
|||||||
return version() >= '4.1'
|
return version() >= '4.1'
|
||||||
|
|
||||||
|
|
||||||
|
def has_thresholding():
|
||||||
|
"""Does Tesseract have -c thresholding method capability?"""
|
||||||
|
return version() >= '5.0'
|
||||||
|
|
||||||
|
|
||||||
def get_languages():
|
def get_languages():
|
||||||
def lang_error(output):
|
def lang_error(output):
|
||||||
msg = (
|
msg = (
|
||||||
@@ -117,10 +160,10 @@ def get_languages():
|
|||||||
if line.startswith('Error'):
|
if line.startswith('Error'):
|
||||||
raise MissingDependencyError(lang_error(output))
|
raise MissingDependencyError(lang_error(output))
|
||||||
_header, *rest = output.splitlines()
|
_header, *rest = output.splitlines()
|
||||||
return set(lang.strip() for lang in rest)
|
return {lang.strip() for lang in rest}
|
||||||
|
|
||||||
|
|
||||||
def tess_base_args(langs: List[str], engine_mode: Optional[int]) -> List[str]:
|
def tess_base_args(langs: list[str], engine_mode: int | None) -> list[str]:
|
||||||
args = ['tesseract']
|
args = ['tesseract']
|
||||||
if langs:
|
if langs:
|
||||||
args.extend(['-l', '+'.join(langs)])
|
args.extend(['-l', '+'.join(langs)])
|
||||||
@@ -129,7 +172,20 @@ def tess_base_args(langs: List[str], engine_mode: Optional[int]) -> List[str]:
|
|||||||
return args
|
return args
|
||||||
|
|
||||||
|
|
||||||
def get_orientation(input_file: Path, engine_mode: Optional[int], timeout: float):
|
def _parse_tesseract_output(binary_output: bytes) -> dict[str, str]:
|
||||||
|
def gen():
|
||||||
|
for line in binary_output.decode().splitlines():
|
||||||
|
line = line.strip()
|
||||||
|
parts = line.split(':', maxsplit=2)
|
||||||
|
if len(parts) == 2:
|
||||||
|
yield parts[0].strip(), parts[1].strip()
|
||||||
|
|
||||||
|
return dict(gen())
|
||||||
|
|
||||||
|
|
||||||
|
def get_orientation(
|
||||||
|
input_file: Path, engine_mode: int | None, timeout: float
|
||||||
|
) -> OrientationConfidence:
|
||||||
args_tesseract = tess_base_args(['osd'], engine_mode) + [
|
args_tesseract = tess_base_args(['osd'], engine_mode) + [
|
||||||
'--psm',
|
'--psm',
|
||||||
'0',
|
'0',
|
||||||
@@ -139,7 +195,6 @@ def get_orientation(input_file: Path, engine_mode: Optional[int], timeout: float
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||||
stdout = p.stdout
|
|
||||||
except TimeoutExpired:
|
except TimeoutExpired:
|
||||||
return OrientationConfidence(angle=0, confidence=0.0)
|
return OrientationConfidence(angle=0, confidence=0.0)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
@@ -151,19 +206,44 @@ def get_orientation(input_file: Path, engine_mode: Optional[int], timeout: float
|
|||||||
):
|
):
|
||||||
return OrientationConfidence(0, 0)
|
return OrientationConfidence(0, 0)
|
||||||
raise SubprocessOutputError() from e
|
raise SubprocessOutputError() from e
|
||||||
else:
|
|
||||||
osd = {}
|
|
||||||
for line in stdout.decode().splitlines():
|
|
||||||
line = line.strip()
|
|
||||||
parts = line.split(':', maxsplit=2)
|
|
||||||
if len(parts) == 2:
|
|
||||||
osd[parts[0].strip()] = parts[1].strip()
|
|
||||||
|
|
||||||
angle = int(osd.get('Orientation in degrees', 0))
|
osd = _parse_tesseract_output(p.stdout)
|
||||||
oc = OrientationConfidence(
|
angle = int(osd.get('Orientation in degrees', 0))
|
||||||
angle=angle, confidence=float(osd.get('Orientation confidence', 0))
|
orient_conf = OrientationConfidence(
|
||||||
)
|
angle=angle, confidence=float(osd.get('Orientation confidence', 0))
|
||||||
return oc
|
)
|
||||||
|
return orient_conf
|
||||||
|
|
||||||
|
|
||||||
|
def get_deskew(
|
||||||
|
input_file: Path, languages: list[str], engine_mode: int | None, timeout: float
|
||||||
|
) -> float:
|
||||||
|
"""Gets angle to deskew this page, in degrees."""
|
||||||
|
args_tesseract = tess_base_args(languages, engine_mode) + [
|
||||||
|
'--psm',
|
||||||
|
'2',
|
||||||
|
fspath(input_file),
|
||||||
|
'stdout',
|
||||||
|
]
|
||||||
|
|
||||||
|
try:
|
||||||
|
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||||
|
except TimeoutExpired:
|
||||||
|
return 0.0
|
||||||
|
except CalledProcessError as e:
|
||||||
|
tesseract_log_output(e.stdout)
|
||||||
|
tesseract_log_output(e.stderr)
|
||||||
|
if b'Empty page!!' in e.output or (
|
||||||
|
e.output == b'' and e.returncode == 1
|
||||||
|
): # Not enough info for a skew angle - Tess 4 and 5 return different errors
|
||||||
|
return 0.0
|
||||||
|
|
||||||
|
raise SubprocessOutputError() from e
|
||||||
|
|
||||||
|
parsed = _parse_tesseract_output(p.stdout)
|
||||||
|
deskew_radians = float(parsed.get('Deskew angle', 0))
|
||||||
|
deskew_degrees = 180 / pi * deskew_radians
|
||||||
|
return deskew_degrees
|
||||||
|
|
||||||
|
|
||||||
def tesseract_log_output(stream):
|
def tesseract_log_output(stream):
|
||||||
@@ -227,14 +307,16 @@ def generate_hocr(
|
|||||||
input_file: Path,
|
input_file: Path,
|
||||||
output_hocr: Path,
|
output_hocr: Path,
|
||||||
output_text: Path,
|
output_text: Path,
|
||||||
languages: List[str],
|
languages: list[str],
|
||||||
engine_mode: int,
|
engine_mode: int,
|
||||||
tessconfig: List[str],
|
tessconfig: list[str],
|
||||||
timeout: float,
|
timeout: float,
|
||||||
pagesegmode: int,
|
pagesegmode: int,
|
||||||
|
thresholding: int,
|
||||||
user_words,
|
user_words,
|
||||||
user_patterns,
|
user_patterns,
|
||||||
):
|
):
|
||||||
|
"""Generate a hOCR file, which must be converted to PDF."""
|
||||||
prefix = output_hocr.with_suffix('')
|
prefix = output_hocr.with_suffix('')
|
||||||
|
|
||||||
args_tesseract = tess_base_args(languages, engine_mode)
|
args_tesseract = tess_base_args(languages, engine_mode)
|
||||||
@@ -242,6 +324,9 @@ def generate_hocr(
|
|||||||
if pagesegmode is not None:
|
if pagesegmode is not None:
|
||||||
args_tesseract.extend(['--psm', str(pagesegmode)])
|
args_tesseract.extend(['--psm', str(pagesegmode)])
|
||||||
|
|
||||||
|
if thresholding != 0 and has_thresholding():
|
||||||
|
args_tesseract.extend(['-c', f'thresholding_method={thresholding}'])
|
||||||
|
|
||||||
if user_words:
|
if user_words:
|
||||||
args_tesseract.extend(['--user-words', user_words])
|
args_tesseract.extend(['--user-words', user_words])
|
||||||
|
|
||||||
@@ -250,7 +335,8 @@ def generate_hocr(
|
|||||||
|
|
||||||
# Reminder: test suite tesseract test plugins will break after any changes
|
# Reminder: test suite tesseract test plugins will break after any changes
|
||||||
# to the number of order parameters here
|
# to the number of order parameters here
|
||||||
args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig)
|
args_tesseract.extend([fspath(input_file), fspath(prefix), 'hocr', 'txt'])
|
||||||
|
args_tesseract.extend(tessconfig)
|
||||||
try:
|
try:
|
||||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||||
stdout = p.stdout
|
stdout = p.stdout
|
||||||
@@ -262,7 +348,7 @@ def generate_hocr(
|
|||||||
_generate_null_hocr(output_hocr, output_text, input_file)
|
_generate_null_hocr(output_hocr, output_text, input_file)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
tesseract_log_output(e.output)
|
tesseract_log_output(e.output)
|
||||||
if b'Image too large' in e.output:
|
if b'Image too large' in e.output or b'Empty page!!' in e.output:
|
||||||
_generate_null_hocr(output_hocr, output_text, input_file)
|
_generate_null_hocr(output_hocr, output_text, input_file)
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -272,7 +358,7 @@ def generate_hocr(
|
|||||||
# The sidecar text file will get the suffix .txt; rename it to
|
# The sidecar text file will get the suffix .txt; rename it to
|
||||||
# whatever caller wants it named
|
# whatever caller wants it named
|
||||||
if prefix.with_suffix('.txt').exists():
|
if prefix.with_suffix('.txt').exists():
|
||||||
shutil.move(prefix.with_suffix('.txt'), output_text)
|
prefix.with_suffix('.txt').replace(output_text)
|
||||||
|
|
||||||
|
|
||||||
def use_skip_page(output_pdf, output_text):
|
def use_skip_page(output_pdf, output_text):
|
||||||
@@ -287,25 +373,20 @@ def generate_pdf(
|
|||||||
input_file: Path,
|
input_file: Path,
|
||||||
output_pdf: Path,
|
output_pdf: Path,
|
||||||
output_text: Path,
|
output_text: Path,
|
||||||
languages: List[str],
|
languages: list[str],
|
||||||
engine_mode: int,
|
engine_mode: int,
|
||||||
tessconfig: List[str],
|
tessconfig: list[str],
|
||||||
timeout: float,
|
timeout: float,
|
||||||
pagesegmode: int,
|
pagesegmode: int,
|
||||||
|
thresholding: int,
|
||||||
user_words,
|
user_words,
|
||||||
user_patterns,
|
user_patterns,
|
||||||
):
|
):
|
||||||
"""Use Tesseract to render a PDF.
|
"""Generate a PDF using Tesseract's internal PDF generator.
|
||||||
|
|
||||||
input_file -- image to analyze
|
We specifically a text-only PDF which is more suitable for combining with
|
||||||
output_pdf -- file to generate
|
the input page.
|
||||||
output_text -- OCR text file
|
|
||||||
languages -- list of languages to consider
|
|
||||||
engine_mode -- engine mode argument for tess v4
|
|
||||||
tessconfig -- tesseract configuration
|
|
||||||
timeout -- timeout (seconds)
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
args_tesseract = tess_base_args(languages, engine_mode)
|
args_tesseract = tess_base_args(languages, engine_mode)
|
||||||
|
|
||||||
if pagesegmode is not None:
|
if pagesegmode is not None:
|
||||||
@@ -313,29 +394,33 @@ def generate_pdf(
|
|||||||
|
|
||||||
args_tesseract.extend(['-c', 'textonly_pdf=1'])
|
args_tesseract.extend(['-c', 'textonly_pdf=1'])
|
||||||
|
|
||||||
|
if thresholding != 0 and has_thresholding():
|
||||||
|
args_tesseract.extend(['-c', f'thresholding_method={thresholding}'])
|
||||||
|
|
||||||
if user_words:
|
if user_words:
|
||||||
args_tesseract.extend(['--user-words', user_words])
|
args_tesseract.extend(['--user-words', user_words])
|
||||||
|
|
||||||
if user_patterns:
|
if user_patterns:
|
||||||
args_tesseract.extend(['--user-patterns', user_patterns])
|
args_tesseract.extend(['--user-patterns', user_patterns])
|
||||||
|
|
||||||
prefix = os.path.splitext(output_pdf)[0] # Tesseract appends suffixes
|
prefix = output_pdf.parent / Path(output_pdf.stem)
|
||||||
|
|
||||||
# Reminder: test suite tesseract test plugins might break after any changes
|
# Reminder: test suite tesseract test plugins might break after any changes
|
||||||
# to the number of order parameters here
|
# to the number of order parameters here
|
||||||
|
|
||||||
args_tesseract.extend([input_file, prefix, 'pdf', 'txt'] + tessconfig)
|
args_tesseract.extend([fspath(input_file), fspath(prefix), 'pdf', 'txt'])
|
||||||
|
args_tesseract.extend(tessconfig)
|
||||||
try:
|
try:
|
||||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||||
stdout = p.stdout
|
stdout = p.stdout
|
||||||
if os.path.exists(prefix + '.txt'):
|
if prefix.with_suffix('.txt').exists():
|
||||||
shutil.move(prefix + '.txt', output_text)
|
prefix.with_suffix('.txt').replace(output_text)
|
||||||
except TimeoutExpired:
|
except TimeoutExpired:
|
||||||
page_timedout(timeout)
|
page_timedout(timeout)
|
||||||
use_skip_page(output_pdf, output_text)
|
use_skip_page(output_pdf, output_text)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
tesseract_log_output(e.output)
|
tesseract_log_output(e.output)
|
||||||
if b'Image too large' in e.output:
|
if b'Image too large' in e.output or b'Empty page!!' in e.output:
|
||||||
use_skip_page(output_pdf, output_text)
|
use_skip_page(output_pdf, output_text)
|
||||||
return
|
return
|
||||||
raise SubprocessOutputError() from e
|
raise SubprocessOutputError() from e
|
||||||
|
|||||||
+102
-46
@@ -5,77 +5,128 @@
|
|||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
# unpaper documentation:
|
# unpaper documentation:
|
||||||
# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md
|
# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md
|
||||||
|
|
||||||
"""Interface to unpaper executable"""
|
"""Interface to unpaper executable"""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import shlex
|
import shlex
|
||||||
|
import sys
|
||||||
|
from contextlib import contextmanager
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT
|
from subprocess import PIPE, STDOUT
|
||||||
from tempfile import TemporaryDirectory
|
from typing import Iterator, Union
|
||||||
from typing import List, Optional, Tuple, Union
|
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
||||||
from ocrmypdf.subprocess import get_version
|
from ocrmypdf.subprocess import get_version, run
|
||||||
from ocrmypdf.subprocess import run as external_run
|
|
||||||
|
if sys.version_info >= (3, 10):
|
||||||
|
from tempfile import TemporaryDirectory
|
||||||
|
else:
|
||||||
|
from tempfile import TemporaryDirectory as _TemporaryDirectory
|
||||||
|
|
||||||
|
class TemporaryDirectory(_TemporaryDirectory):
|
||||||
|
"""Shim to consume ignore_cleanup_errors kwarg on Python 3.9 and older.
|
||||||
|
|
||||||
|
The argument is consumed without action. If users are getting errors related
|
||||||
|
to temporary file cleanup, they should upgrade to Python 3.10 which properly
|
||||||
|
cleans up temporary directories on Windows.
|
||||||
|
|
||||||
|
See: https://github.com/python/cpython/pull/24793
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, ignore_cleanup_errors=False, **kwargs):
|
||||||
|
super().__init__(**kwargs)
|
||||||
|
|
||||||
|
del _TemporaryDirectory
|
||||||
|
|
||||||
|
|
||||||
|
UNPAPER_IMAGE_PIXEL_LIMIT = 256 * 1024 * 1024
|
||||||
|
|
||||||
DecFloat = Union[Decimal, float]
|
DecFloat = Union[Decimal, float]
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
class UnpaperImageTooLargeError(Exception):
|
||||||
|
"""To capture details when an image is too large for unpaper."""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
w,
|
||||||
|
h,
|
||||||
|
message="Image with size {}x{} is too large for cleaning with 'unpaper'.",
|
||||||
|
):
|
||||||
|
self.w = w
|
||||||
|
self.h = h
|
||||||
|
self.message = message.format(w, h)
|
||||||
|
super().__init__(self.message)
|
||||||
|
|
||||||
|
|
||||||
def version() -> str:
|
def version() -> str:
|
||||||
return get_version('unpaper')
|
return get_version('unpaper')
|
||||||
|
|
||||||
|
|
||||||
def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]:
|
SUPPORTED_MODES = {'1', 'L', 'RGB'}
|
||||||
SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'}
|
|
||||||
with Image.open(input_file) as im:
|
|
||||||
im_modified = False
|
def _convert_image(im: Image.Image) -> tuple[Image.Image, bool]:
|
||||||
if im.mode not in SUFFIXES:
|
im_modified = False
|
||||||
log.info("Converting image to other colorspace")
|
|
||||||
try:
|
if im.mode not in SUPPORTED_MODES:
|
||||||
if im.mode == 'P' and len(im.getcolors()) == 2:
|
log.info("Converting image to other colorspace")
|
||||||
im = im.convert(mode='1')
|
|
||||||
else:
|
|
||||||
im = im.convert(mode='RGB')
|
|
||||||
except IOError as e:
|
|
||||||
raise MissingDependencyError(
|
|
||||||
"Could not convert image with type " + im.mode
|
|
||||||
) from e
|
|
||||||
else:
|
|
||||||
im_modified = True
|
|
||||||
try:
|
try:
|
||||||
suffix = SUFFIXES[im.mode]
|
if im.mode == 'P' and len(im.getcolors()) == 2:
|
||||||
except KeyError:
|
im = im.convert(mode='1')
|
||||||
|
else:
|
||||||
|
im = im.convert(mode='RGB')
|
||||||
|
except OSError as e:
|
||||||
|
raise MissingDependencyError(
|
||||||
|
"Could not convert image with type " + im.mode
|
||||||
|
) from e
|
||||||
|
else:
|
||||||
|
im_modified = True
|
||||||
|
if im.mode not in SUPPORTED_MODES:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
"Failed to convert image to a supported format."
|
"Failed to convert image to a supported format."
|
||||||
) from None
|
) from None
|
||||||
|
return im, im_modified
|
||||||
if im_modified or input_file.suffix != '.pnm':
|
|
||||||
input_pnm = tmpdir / 'input.pnm'
|
|
||||||
im.save(input_pnm, format='PPM')
|
|
||||||
else:
|
|
||||||
# No changes, PNG input, just use the file we already have
|
|
||||||
input_pnm = input_file
|
|
||||||
output_pnm = tmpdir / f'output{suffix}'
|
|
||||||
return input_pnm, output_pnm
|
|
||||||
|
|
||||||
|
|
||||||
def run(
|
@contextmanager
|
||||||
input_file: Path, output_file: Path, *, dpi: DecFloat, mode_args: List[str]
|
def _setup_unpaper_io(input_file: Path) -> Iterator[tuple[Path, Path, Path]]:
|
||||||
|
with Image.open(input_file) as im:
|
||||||
|
if im.width * im.height >= UNPAPER_IMAGE_PIXEL_LIMIT:
|
||||||
|
raise UnpaperImageTooLargeError(w=im.width, h=im.height)
|
||||||
|
im, im_modified = _convert_image(im)
|
||||||
|
|
||||||
|
with TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
||||||
|
tmppath = Path(tmpdir)
|
||||||
|
if im_modified or input_file.suffix != '.png':
|
||||||
|
input_png = tmppath / 'input.png'
|
||||||
|
im.save(input_png, format='PNG')
|
||||||
|
else:
|
||||||
|
# No changes, PNG input, just use the file we already have
|
||||||
|
input_png = input_file
|
||||||
|
|
||||||
|
# unpaper can write .png too, but it seems to write them slowly
|
||||||
|
# adds a few seconds to test suite - so just use pnm
|
||||||
|
output_pnm = tmppath / 'output.pnm'
|
||||||
|
yield input_png, output_pnm, tmppath
|
||||||
|
|
||||||
|
|
||||||
|
def run_unpaper(
|
||||||
|
input_file: Path, output_file: Path, *, dpi: DecFloat, mode_args: list[str]
|
||||||
) -> None:
|
) -> None:
|
||||||
args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args
|
args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args
|
||||||
|
|
||||||
with TemporaryDirectory() as tmpdir:
|
with _setup_unpaper_io(input_file) as (input_png, output_pnm, tmpdir):
|
||||||
input_pnm, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file)
|
|
||||||
|
|
||||||
# To prevent any shenanigans from accepting arbitrary parameters in
|
# To prevent any shenanigans from accepting arbitrary parameters in
|
||||||
# --unpaper-args, we:
|
# --unpaper-args, we:
|
||||||
# 1) run with cwd set to a tmpdir with only unpaper's files
|
# 1) run with cwd set to a tmpdir with only unpaper's files
|
||||||
@@ -83,8 +134,8 @@ def run(
|
|||||||
# 3) append absolute paths for the input and output file
|
# 3) append absolute paths for the input and output file
|
||||||
# This should ensure that a user cannot clobber some other file with
|
# This should ensure that a user cannot clobber some other file with
|
||||||
# their unpaper arguments (whether intentionally or otherwise)
|
# their unpaper arguments (whether intentionally or otherwise)
|
||||||
args_unpaper.extend([os.fspath(input_pnm), os.fspath(output_pnm)])
|
args_unpaper.extend([os.fspath(input_png), os.fspath(output_pnm)])
|
||||||
external_run(
|
run(
|
||||||
args_unpaper,
|
args_unpaper,
|
||||||
close_fds=True,
|
close_fds=True,
|
||||||
check=True,
|
check=True,
|
||||||
@@ -96,15 +147,15 @@ def run(
|
|||||||
try:
|
try:
|
||||||
with Image.open(output_pnm) as imout:
|
with Image.open(output_pnm) as imout:
|
||||||
imout.save(output_file, dpi=(dpi, dpi))
|
imout.save(output_file, dpi=(dpi, dpi))
|
||||||
except (FileNotFoundError, OSError):
|
except OSError as e:
|
||||||
raise SubprocessOutputError(
|
raise SubprocessOutputError(
|
||||||
"unpaper: failed to produce the expected output file. "
|
"unpaper: failed to produce the expected output file. "
|
||||||
+ " Called with: "
|
+ " Called with: "
|
||||||
+ str(args_unpaper)
|
+ str(args_unpaper)
|
||||||
) from None
|
) from e
|
||||||
|
|
||||||
|
|
||||||
def validate_custom_args(args: str) -> List[str]:
|
def validate_custom_args(args: str) -> list[str]:
|
||||||
unpaper_args = shlex.split(args)
|
unpaper_args = shlex.split(args)
|
||||||
if any(('/' in arg or arg == '.' or arg == '..') for arg in unpaper_args):
|
if any(('/' in arg or arg == '.' or arg == '..') for arg in unpaper_args):
|
||||||
raise ValueError('No filenames allowed in --unpaper-args')
|
raise ValueError('No filenames allowed in --unpaper-args')
|
||||||
@@ -116,8 +167,8 @@ def clean(
|
|||||||
output_file: Path,
|
output_file: Path,
|
||||||
*,
|
*,
|
||||||
dpi: DecFloat,
|
dpi: DecFloat,
|
||||||
unpaper_args: Optional[List[str]] = None,
|
unpaper_args: list[str] | None = None,
|
||||||
):
|
) -> Path:
|
||||||
default_args = [
|
default_args = [
|
||||||
'--layout',
|
'--layout',
|
||||||
'none',
|
'none',
|
||||||
@@ -131,4 +182,9 @@ def clean(
|
|||||||
]
|
]
|
||||||
if not unpaper_args:
|
if not unpaper_args:
|
||||||
unpaper_args = default_args
|
unpaper_args = default_args
|
||||||
run(input_file, output_file, dpi=dpi, mode_args=unpaper_args)
|
try:
|
||||||
|
run_unpaper(input_file, output_file, dpi=dpi, mode_args=unpaper_args)
|
||||||
|
return output_file
|
||||||
|
except UnpaperImageTooLargeError as e:
|
||||||
|
log.warning(str(e))
|
||||||
|
return input_file
|
||||||
|
|||||||
+45
-55
@@ -4,15 +4,26 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""For grafting text-only PDF pages onto freeform PDF pages."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import uuid
|
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
import pikepdf
|
from pikepdf import (
|
||||||
from pikepdf.objects import Dictionary, Name
|
Dictionary,
|
||||||
|
Name,
|
||||||
|
Object,
|
||||||
|
Operator,
|
||||||
|
Pdf,
|
||||||
|
PdfError,
|
||||||
|
PdfMatrix,
|
||||||
|
Stream,
|
||||||
|
parse_content_stream,
|
||||||
|
unparse_content_stream,
|
||||||
|
)
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
MAX_REPLACE_PAGES = 100
|
MAX_REPLACE_PAGES = 100
|
||||||
@@ -47,59 +58,43 @@ def strip_invisible_text(pdf, page):
|
|||||||
render_mode = 0
|
render_mode = 0
|
||||||
text_objects = []
|
text_objects = []
|
||||||
|
|
||||||
rich_page = pikepdf.Page(page)
|
for operands, operator in parse_content_stream(page, ''):
|
||||||
rich_page.contents_coalesce()
|
|
||||||
for operands, operator in pikepdf.parse_content_stream(page, ''):
|
|
||||||
if not in_text_obj:
|
if not in_text_obj:
|
||||||
if operator == pikepdf.Operator('BT'):
|
if operator == Operator('BT'):
|
||||||
in_text_obj = True
|
in_text_obj = True
|
||||||
render_mode = 0
|
render_mode = 0
|
||||||
text_objects.append((operands, operator))
|
text_objects.append((operands, operator))
|
||||||
else:
|
else:
|
||||||
stream.append((operands, operator))
|
stream.append((operands, operator))
|
||||||
else:
|
else:
|
||||||
if operator == pikepdf.Operator('Tr'):
|
if operator == Operator('Tr'):
|
||||||
render_mode = operands[0]
|
render_mode = operands[0]
|
||||||
text_objects.append((operands, operator))
|
text_objects.append((operands, operator))
|
||||||
if operator == pikepdf.Operator('ET'):
|
if operator == Operator('ET'):
|
||||||
in_text_obj = False
|
in_text_obj = False
|
||||||
if render_mode != 3:
|
if render_mode != 3:
|
||||||
stream.extend(text_objects)
|
stream.extend(text_objects)
|
||||||
text_objects.clear()
|
text_objects.clear()
|
||||||
|
|
||||||
def convert(op):
|
content_stream = unparse_content_stream(stream)
|
||||||
try:
|
page.Contents = Stream(pdf, content_stream)
|
||||||
return op.unparse()
|
|
||||||
except AttributeError:
|
|
||||||
return str(op).encode('ascii')
|
|
||||||
|
|
||||||
lines = []
|
|
||||||
|
|
||||||
for operands, operator in stream:
|
|
||||||
if operator == pikepdf.Operator('INLINE IMAGE'):
|
|
||||||
iim = operands[0]
|
|
||||||
line = iim.unparse()
|
|
||||||
else:
|
|
||||||
line = b' '.join(convert(op) for op in operands) + b' ' + operator.unparse()
|
|
||||||
lines.append(line)
|
|
||||||
|
|
||||||
content_stream = b'\n'.join(lines)
|
|
||||||
page.Contents = pikepdf.Stream(pdf, content_stream)
|
|
||||||
|
|
||||||
|
|
||||||
class OcrGrafter:
|
class OcrGrafter:
|
||||||
|
"""Manages grafting text-only PDFs onto regular PDFs."""
|
||||||
|
|
||||||
def __init__(self, context):
|
def __init__(self, context):
|
||||||
self.context = context
|
self.context = context
|
||||||
self.path_base = context.origin
|
self.path_base = context.origin
|
||||||
|
|
||||||
self.pdf_base = pikepdf.open(self.path_base)
|
self.pdf_base = Pdf.open(self.path_base)
|
||||||
self.font, self.font_key = None, None
|
self.font, self.font_key = None, None
|
||||||
|
|
||||||
self.pdfinfo = context.pdfinfo
|
self.pdfinfo = context.pdfinfo
|
||||||
self.output_file = context.get_path('graft_layers.pdf')
|
self.output_file = context.get_path('graft_layers.pdf')
|
||||||
|
|
||||||
self.procset = self.pdf_base.make_indirect(
|
self.procset = self.pdf_base.make_indirect(
|
||||||
pikepdf.Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]')
|
Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]')
|
||||||
)
|
)
|
||||||
|
|
||||||
self.emplacements = 1
|
self.emplacements = 1
|
||||||
@@ -109,8 +104,8 @@ class OcrGrafter:
|
|||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
pageno: int,
|
pageno: int,
|
||||||
image: Optional[Path],
|
image: Path | None,
|
||||||
textpdf: Optional[Path],
|
textpdf: Path | None,
|
||||||
autorotate_correction: int,
|
autorotate_correction: int,
|
||||||
):
|
):
|
||||||
if textpdf and not self.font:
|
if textpdf and not self.font:
|
||||||
@@ -123,7 +118,7 @@ class OcrGrafter:
|
|||||||
# We are updating the old page with a rasterized PDF of the new
|
# We are updating the old page with a rasterized PDF of the new
|
||||||
# page (without changing objgen, to preserve references)
|
# page (without changing objgen, to preserve references)
|
||||||
log.debug("Emplacement update")
|
log.debug("Emplacement update")
|
||||||
with pikepdf.open(image) as pdf_image:
|
with Pdf.open(path_image) as pdf_image:
|
||||||
self.emplacements += 1
|
self.emplacements += 1
|
||||||
foreign_image_page = pdf_image.pages[0]
|
foreign_image_page = pdf_image.pages[0]
|
||||||
self.pdf_base.pages.append(foreign_image_page)
|
self.pdf_base.pages.append(foreign_image_page)
|
||||||
@@ -196,7 +191,7 @@ class OcrGrafter:
|
|||||||
self.pdf_base.save(next_file)
|
self.pdf_base.save(next_file)
|
||||||
self.pdf_base.close()
|
self.pdf_base.close()
|
||||||
|
|
||||||
self.pdf_base = pikepdf.open(next_file)
|
self.pdf_base = Pdf.open(next_file)
|
||||||
self.procset = self.pdf_base.pages[0].Resources.ProcSet
|
self.procset = self.pdf_base.pages[0].Resources.ProcSet
|
||||||
self.font, self.font_key = None, None # Ensure we reacquire this information
|
self.font, self.font_key = None, None # Ensure we reacquire this information
|
||||||
self.interim_count += 1
|
self.interim_count += 1
|
||||||
@@ -212,7 +207,7 @@ class OcrGrafter:
|
|||||||
font, font_key = None, None
|
font, font_key = None, None
|
||||||
possible_font_names = ('/f-0-0', '/F1')
|
possible_font_names = ('/f-0-0', '/F1')
|
||||||
try:
|
try:
|
||||||
with pikepdf.open(text) as pdf_text:
|
with Pdf.open(text) as pdf_text:
|
||||||
try:
|
try:
|
||||||
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
||||||
except (AttributeError, IndexError, KeyError):
|
except (AttributeError, IndexError, KeyError):
|
||||||
@@ -226,7 +221,7 @@ class OcrGrafter:
|
|||||||
if pdf_text_font:
|
if pdf_text_font:
|
||||||
font = self.pdf_base.copy_foreign(pdf_text_font)
|
font = self.pdf_base.copy_foreign(pdf_text_font)
|
||||||
return font, font_key
|
return font, font_key
|
||||||
except (FileNotFoundError, pikepdf.PdfError):
|
except (FileNotFoundError, PdfError):
|
||||||
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
||||||
return None, None
|
return None, None
|
||||||
|
|
||||||
@@ -235,20 +230,22 @@ class OcrGrafter:
|
|||||||
*,
|
*,
|
||||||
page_num: int,
|
page_num: int,
|
||||||
textpdf: Path,
|
textpdf: Path,
|
||||||
font: pikepdf.Object,
|
font: Object,
|
||||||
font_key: pikepdf.Object,
|
font_key: Object,
|
||||||
procset: pikepdf.Object,
|
procset: Object,
|
||||||
text_rotation: int,
|
text_rotation: int,
|
||||||
strip_old_text: bool,
|
strip_old_text: bool,
|
||||||
):
|
):
|
||||||
"""Insert the text layer from text page 0 on to pdf_base at page_num"""
|
"""Insert the text layer from text page 0 on to pdf_base at page_num"""
|
||||||
|
|
||||||
|
# pylint: disable=invalid-name
|
||||||
|
|
||||||
log.debug("Grafting")
|
log.debug("Grafting")
|
||||||
if Path(textpdf).stat().st_size == 0:
|
if Path(textpdf).stat().st_size == 0:
|
||||||
return
|
return
|
||||||
|
|
||||||
# This is a pointer indicating a specific page in the base file
|
# This is a pointer indicating a specific page in the base file
|
||||||
with pikepdf.open(textpdf) as pdf_text:
|
with Pdf.open(textpdf) as pdf_text:
|
||||||
pdf_text_contents = pdf_text.pages[0].Contents.read_bytes()
|
pdf_text_contents = pdf_text.pages[0].Contents.read_bytes()
|
||||||
|
|
||||||
base_page = self.pdf_base.pages.p(page_num)
|
base_page = self.pdf_base.pages.p(page_num)
|
||||||
@@ -263,13 +260,13 @@ class OcrGrafter:
|
|||||||
mediabox = [float(base_page.MediaBox[v]) for v in range(4)]
|
mediabox = [float(base_page.MediaBox[v]) for v in range(4)]
|
||||||
wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
||||||
|
|
||||||
translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2)
|
translate = PdfMatrix().translated(-wt / 2, -ht / 2)
|
||||||
untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2)
|
untranslate = PdfMatrix().translated(wp / 2, hp / 2)
|
||||||
corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1])
|
corner = PdfMatrix().translated(mediabox[0], mediabox[1])
|
||||||
# -rotation because the input is a clockwise angle and this formula
|
# -rotation because the input is a clockwise angle and this formula
|
||||||
# uses CCW
|
# uses CCW
|
||||||
text_rotation = -text_rotation % 360
|
text_rotation = -text_rotation % 360
|
||||||
rotate = pikepdf.PdfMatrix().rotated(text_rotation)
|
rotate = PdfMatrix().rotated(text_rotation)
|
||||||
|
|
||||||
# Because of rounding of DPI, we might get a text layer that is not
|
# Because of rounding of DPI, we might get a text layer that is not
|
||||||
# identically sized to the target page. Scale to adjust. Normally this
|
# identically sized to the target page. Scale to adjust. Normally this
|
||||||
@@ -280,7 +277,7 @@ class OcrGrafter:
|
|||||||
scale_y = hp / ht
|
scale_y = hp / ht
|
||||||
|
|
||||||
# log.debug('%r', scale_x, scale_y)
|
# log.debug('%r', scale_x, scale_y)
|
||||||
scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y)
|
scale = PdfMatrix().scaled(scale_x, scale_y)
|
||||||
|
|
||||||
# Translate the text so it is centered at (0, 0), rotate it there, adjust
|
# Translate the text so it is centered at (0, 0), rotate it there, adjust
|
||||||
# for a size different between initial and text PDF, then untranslate, and
|
# for a size different between initial and text PDF, then untranslate, and
|
||||||
@@ -289,7 +286,7 @@ class OcrGrafter:
|
|||||||
|
|
||||||
base_resources = _ensure_dictionary(base_page, Name.Resources)
|
base_resources = _ensure_dictionary(base_page, Name.Resources)
|
||||||
base_xobjs = _ensure_dictionary(base_resources, Name.XObject)
|
base_xobjs = _ensure_dictionary(base_resources, Name.XObject)
|
||||||
text_xobj_name = Name('/' + str(uuid.uuid4()))
|
text_xobj_name = Name.random(prefix="OCR-")
|
||||||
xobj = self.pdf_base.make_stream(pdf_text_contents)
|
xobj = self.pdf_base.make_stream(pdf_text_contents)
|
||||||
base_xobjs[text_xobj_name] = xobj
|
base_xobjs[text_xobj_name] = xobj
|
||||||
xobj.Type = Name.XObject
|
xobj.Type = Name.XObject
|
||||||
@@ -303,19 +300,12 @@ class OcrGrafter:
|
|||||||
pdf_draw_xobj = (
|
pdf_draw_xobj = (
|
||||||
(b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
(b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
||||||
)
|
)
|
||||||
new_text_layer = pikepdf.Stream(self.pdf_base, pdf_draw_xobj)
|
new_text_layer = Stream(self.pdf_base, pdf_draw_xobj)
|
||||||
|
|
||||||
if strip_old_text:
|
if strip_old_text:
|
||||||
strip_invisible_text(self.pdf_base, base_page)
|
strip_invisible_text(self.pdf_base, base_page)
|
||||||
|
|
||||||
if hasattr(pikepdf.Page, 'contents_add'):
|
base_page.contents_add(new_text_layer, prepend=True)
|
||||||
# pikepdf >= 2.14 adds this method and deprecates the one below
|
|
||||||
pikepdf.Page(base_page).contents_add(new_text_layer, prepend=True)
|
|
||||||
else:
|
|
||||||
# pikepdf < 2.14
|
|
||||||
base_page.page_contents_add(
|
|
||||||
new_text_layer, prepend=True
|
|
||||||
) # pragma: no cover
|
|
||||||
|
|
||||||
_update_resources(
|
_update_resources(
|
||||||
obj=base_page, font=font, font_key=font_key, procset=procset
|
obj=base_page, font=font, font_key=font_key, procset=procset
|
||||||
|
|||||||
@@ -4,6 +4,9 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Defines context objects that are passed to child processes/threads."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
@@ -49,7 +52,7 @@ class PdfContext:
|
|||||||
"""
|
"""
|
||||||
return self.work_folder / name
|
return self.work_folder / name
|
||||||
|
|
||||||
def get_page_contexts(self) -> Iterator['PageContext']:
|
def get_page_contexts(self) -> Iterator[PageContext]:
|
||||||
"""Get all ``PageContext`` for this PDF."""
|
"""Get all ``PageContext`` for this PDF."""
|
||||||
npages = len(self.pdfinfo)
|
npages = len(self.pdfinfo)
|
||||||
for n in range(npages):
|
for n in range(npages):
|
||||||
@@ -83,7 +86,7 @@ class PageContext:
|
|||||||
The path will be based in a common temporary folder and have a prefix based
|
The path will be based in a common temporary folder and have a prefix based
|
||||||
on the page number.
|
on the page number.
|
||||||
"""
|
"""
|
||||||
return self.work_folder / ("%06d_%s" % (self.pageno + 1, name))
|
return self.work_folder / f"{(self.pageno + 1):06d}_{name}"
|
||||||
|
|
||||||
def __getstate__(self):
|
def __getstate__(self):
|
||||||
state = self.__dict__.copy()
|
state = self.__dict__.copy()
|
||||||
|
|||||||
@@ -4,15 +4,19 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Logging support classes."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import sys
|
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
|
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
|
|
||||||
|
|
||||||
class PageNumberFilter(logging.Filter):
|
class PageNumberFilter(logging.Filter):
|
||||||
|
"""Insert PDF page number that emitted log message to log record."""
|
||||||
|
|
||||||
def filter(self, record):
|
def filter(self, record):
|
||||||
pageno = getattr(record, 'pageno', None)
|
pageno = getattr(record, 'pageno', None)
|
||||||
if isinstance(pageno, int):
|
if isinstance(pageno, int):
|
||||||
@@ -28,22 +32,14 @@ class TqdmConsole:
|
|||||||
This routes log messages through tqdm so that it can print them above the
|
This routes log messages through tqdm so that it can print them above the
|
||||||
progress bar, and then refresh the progress bar, rather than overwriting
|
progress bar, and then refresh the progress bar, rather than overwriting
|
||||||
it which looks messy.
|
it which looks messy.
|
||||||
|
|
||||||
For some reason Python 3.6 prints extra empty messages from time to time,
|
|
||||||
so we suppress those.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, file):
|
def __init__(self, file):
|
||||||
self.file = file
|
self.file = file
|
||||||
self.py36 = sys.version_info[0:2] == (3, 6)
|
|
||||||
|
|
||||||
def write(self, msg):
|
def write(self, msg):
|
||||||
# When no progress bar is active, tqdm.write() routes to print()
|
# When no progress bar is active, tqdm.write() routes to print()
|
||||||
if self.py36:
|
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
||||||
if msg.strip() != '':
|
|
||||||
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
|
||||||
else:
|
|
||||||
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
|
||||||
|
|
||||||
def flush(self):
|
def flush(self):
|
||||||
with suppress(AttributeError):
|
with suppress(AttributeError):
|
||||||
|
|||||||
+78
-59
@@ -4,6 +4,9 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""OCRmyPDF page processing pipeline functions."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -13,14 +16,13 @@ from contextlib import suppress
|
|||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj
|
from shutil import copyfileobj
|
||||||
from typing import Dict, Iterable, Optional
|
from typing import Iterable
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
import pikepdf
|
import pikepdf
|
||||||
from pikepdf.models.metadata import encode_pdf_date
|
from pikepdf.models.metadata import encode_pdf_date
|
||||||
from PIL import Image, ImageColor, ImageDraw
|
from PIL import Image, ImageColor, ImageDraw
|
||||||
|
|
||||||
from ocrmypdf import leptonica
|
|
||||||
from ocrmypdf._concurrent import Executor
|
from ocrmypdf._concurrent import Executor
|
||||||
from ocrmypdf._exec import unpaper
|
from ocrmypdf._exec import unpaper
|
||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
@@ -33,12 +35,18 @@ from ocrmypdf.exceptions import (
|
|||||||
PriorOcrFoundError,
|
PriorOcrFoundError,
|
||||||
UnsupportedImageFormatError,
|
UnsupportedImageFormatError,
|
||||||
)
|
)
|
||||||
from ocrmypdf.helpers import Resolution, safe_symlink
|
from ocrmypdf.helpers import IMG2PDF_KWARGS, Resolution, safe_symlink
|
||||||
from ocrmypdf.hocrtransform import HocrTransform
|
from ocrmypdf.hocrtransform import HocrTransform
|
||||||
from ocrmypdf.optimize import optimize
|
|
||||||
from ocrmypdf.pdfa import generate_pdfa_ps
|
from ocrmypdf.pdfa import generate_pdfa_ps
|
||||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
||||||
|
|
||||||
|
# Remove this workaround when we require Pillow >= 10
|
||||||
|
try:
|
||||||
|
BICUBIC = Image.Resampling.BICUBIC # type: ignore
|
||||||
|
except AttributeError: # pragma: no cover
|
||||||
|
# Pillow 9 shim
|
||||||
|
BICUBIC = Image.BICUBIC # type: ignore
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
VECTOR_PAGE_DPI = 400
|
VECTOR_PAGE_DPI = 400
|
||||||
@@ -48,7 +56,7 @@ def triage_image_file(input_file, output_file, options):
|
|||||||
log.info("Input file is not a PDF, checking if it is an image...")
|
log.info("Input file is not a PDF, checking if it is an image...")
|
||||||
try:
|
try:
|
||||||
im = Image.open(input_file)
|
im = Image.open(input_file)
|
||||||
except EnvironmentError as e:
|
except OSError as e:
|
||||||
# Recover the original filename
|
# Recover the original filename
|
||||||
log.error(str(e).replace(str(input_file), str(options.input_file)))
|
log.error(str(e).replace(str(input_file), str(options.input_file)))
|
||||||
raise UnsupportedImageFormatError() from e
|
raise UnsupportedImageFormatError() from e
|
||||||
@@ -99,8 +107,8 @@ def triage_image_file(input_file, output_file, options):
|
|||||||
img2pdf.convert(
|
img2pdf.convert(
|
||||||
os.fspath(input_file),
|
os.fspath(input_file),
|
||||||
layout_fun=layout_fun,
|
layout_fun=layout_fun,
|
||||||
with_pdfrw=False,
|
|
||||||
outputstream=outf,
|
outputstream=outf,
|
||||||
|
**IMG2PDF_KWARGS,
|
||||||
)
|
)
|
||||||
log.info("Successfully converted to PDF, processing...")
|
log.info("Successfully converted to PDF, processing...")
|
||||||
except img2pdf.ImageOpenError as e:
|
except img2pdf.ImageOpenError as e:
|
||||||
@@ -135,7 +143,7 @@ def triage(original_filename, input_file, output_file, options):
|
|||||||
# Origin file is a pdf create a symlink with pdf extension
|
# Origin file is a pdf create a symlink with pdf extension
|
||||||
safe_symlink(input_file, output_file)
|
safe_symlink(input_file, output_file)
|
||||||
return output_file
|
return output_file
|
||||||
except EnvironmentError as e:
|
except OSError as e:
|
||||||
log.debug(f"Temporary file was at: {input_file}")
|
log.debug(f"Temporary file was at: {input_file}")
|
||||||
msg = str(e).replace(str(input_file), original_filename)
|
msg = str(e).replace(str(input_file), original_filename)
|
||||||
raise InputFileError(msg) from e
|
raise InputFileError(msg) from e
|
||||||
@@ -326,7 +334,8 @@ def is_ocr_required(page_context: PageContext):
|
|||||||
ocr_required = False
|
ocr_required = False
|
||||||
log.warning(
|
log.warning(
|
||||||
"page too big, skipping OCR "
|
"page too big, skipping OCR "
|
||||||
f"({(pixel_count / 1_000_000):.1f} MPixels > {options.skip_big:.1f} MPixels --skip-big)"
|
f"({(pixel_count / 1_000_000):.1f} MPixels > "
|
||||||
|
f"{options.skip_big:.1f} MPixels --skip-big)"
|
||||||
)
|
)
|
||||||
return ocr_required
|
return ocr_required
|
||||||
|
|
||||||
@@ -424,8 +433,8 @@ def rasterize(
|
|||||||
output_file = page_context.get_path(f'rasterize{output_tag}.png')
|
output_file = page_context.get_path(f'rasterize{output_tag}.png')
|
||||||
pageinfo = page_context.pageinfo
|
pageinfo = page_context.pageinfo
|
||||||
|
|
||||||
def at_least(cs):
|
def at_least(colorspace):
|
||||||
return max(device_idx, colorspaces.index(cs))
|
return max(device_idx, colorspaces.index(colorspace))
|
||||||
|
|
||||||
for image in pageinfo.images:
|
for image in pageinfo.images:
|
||||||
if image.type_ != 'image':
|
if image.type_ != 'image':
|
||||||
@@ -465,9 +474,10 @@ def rasterize(
|
|||||||
|
|
||||||
def preprocess_remove_background(input_file: Path, page_context: PageContext):
|
def preprocess_remove_background(input_file: Path, page_context: PageContext):
|
||||||
if any(image.bpc > 1 for image in page_context.pageinfo.images):
|
if any(image.bpc > 1 for image in page_context.pageinfo.images):
|
||||||
output_file = page_context.get_path('pp_rm_bg.png')
|
raise NotImplementedError("--remove-background is temporarily not implemented")
|
||||||
leptonica.remove_background(input_file, output_file)
|
# output_file = page_context.get_path('pp_rm_bg.png')
|
||||||
return output_file
|
# leptonica.remove_background(input_file, output_file)
|
||||||
|
# return output_file
|
||||||
else:
|
else:
|
||||||
log.info("background removal skipped on mono page")
|
log.info("background removal skipped on mono page")
|
||||||
return input_file
|
return input_file
|
||||||
@@ -476,20 +486,32 @@ def preprocess_remove_background(input_file: Path, page_context: PageContext):
|
|||||||
def preprocess_deskew(input_file: Path, page_context: PageContext):
|
def preprocess_deskew(input_file: Path, page_context: PageContext):
|
||||||
output_file = page_context.get_path('pp_deskew.png')
|
output_file = page_context.get_path('pp_deskew.png')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
leptonica.deskew(input_file, output_file, dpi.x)
|
|
||||||
|
ocr_engine = page_context.plugin_manager.hook.get_ocr_engine()
|
||||||
|
deskew_angle_degrees = ocr_engine.get_deskew(input_file, page_context.options)
|
||||||
|
|
||||||
|
with Image.open(input_file) as im:
|
||||||
|
# According to Pillow docs, .rotate() will automatically use Image.NEAREST
|
||||||
|
# resampling if image is mode '1' or 'P'
|
||||||
|
deskewed = im.rotate(
|
||||||
|
deskew_angle_degrees,
|
||||||
|
resample=BICUBIC,
|
||||||
|
fillcolor=ImageColor.getcolor('white', mode=im.mode), # type: ignore
|
||||||
|
)
|
||||||
|
deskewed.save(output_file, dpi=dpi)
|
||||||
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def preprocess_clean(input_file: Path, page_context: PageContext):
|
def preprocess_clean(input_file: Path, page_context: PageContext):
|
||||||
output_file = page_context.get_path('pp_clean.png')
|
output_file = page_context.get_path('pp_clean.png')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
unpaper.clean(
|
return unpaper.clean(
|
||||||
input_file,
|
input_file,
|
||||||
output_file,
|
output_file,
|
||||||
dpi=dpi.x,
|
dpi=dpi.x,
|
||||||
unpaper_args=page_context.options.unpaper_args,
|
unpaper_args=page_context.options.unpaper_args,
|
||||||
)
|
)
|
||||||
return output_file
|
|
||||||
|
|
||||||
|
|
||||||
def create_ocr_image(image: Path, page_context: PageContext):
|
def create_ocr_image(image: Path, page_context: PageContext):
|
||||||
@@ -500,10 +522,6 @@ def create_ocr_image(image: Path, page_context: PageContext):
|
|||||||
output_file = page_context.get_path('ocr.png')
|
output_file = page_context.get_path('ocr.png')
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
with Image.open(image) as im:
|
with Image.open(image) as im:
|
||||||
white = ImageColor.getcolor('#ffffff', im.mode)
|
|
||||||
# pink = ImageColor.getcolor('#ff0080', im.mode)
|
|
||||||
draw = ImageDraw.ImageDraw(im)
|
|
||||||
|
|
||||||
log.debug('resolution %r', im.info['dpi'])
|
log.debug('resolution %r', im.info['dpi'])
|
||||||
|
|
||||||
if not options.force_ocr:
|
if not options.force_ocr:
|
||||||
@@ -513,6 +531,7 @@ def create_ocr_image(image: Path, page_context: PageContext):
|
|||||||
if options.redo_ocr:
|
if options.redo_ocr:
|
||||||
mask = True # Mask visible text, but not invisible text
|
mask = True # Mask visible text, but not invisible text
|
||||||
|
|
||||||
|
draw = ImageDraw.ImageDraw(im)
|
||||||
for textarea in page_context.pageinfo.get_textareas(
|
for textarea in page_context.pageinfo.get_textareas(
|
||||||
visible=mask, corrupt=None
|
visible=mask, corrupt=None
|
||||||
):
|
):
|
||||||
@@ -521,25 +540,15 @@ def create_ocr_image(image: Path, page_context: PageContext):
|
|||||||
# be None)
|
# be None)
|
||||||
bbox = [float(v) for v in textarea]
|
bbox = [float(v) for v in textarea]
|
||||||
xyscale = tuple(float(coord) / 72.0 for coord in im.info['dpi'])
|
xyscale = tuple(float(coord) / 72.0 for coord in im.info['dpi'])
|
||||||
pixcoords = [
|
pixcoords = (
|
||||||
bbox[0] * xyscale[0],
|
bbox[0] * xyscale[0],
|
||||||
im.height - bbox[3] * xyscale[1],
|
im.height - bbox[3] * xyscale[1],
|
||||||
bbox[2] * xyscale[0],
|
bbox[2] * xyscale[0],
|
||||||
im.height - bbox[1] * xyscale[1],
|
im.height - bbox[1] * xyscale[1],
|
||||||
]
|
)
|
||||||
pixcoords = [int(round(c)) for c in pixcoords]
|
|
||||||
log.debug('blanking %r', pixcoords)
|
log.debug('blanking %r', pixcoords)
|
||||||
draw.rectangle(pixcoords, fill=white)
|
draw.rectangle(pixcoords, fill='white')
|
||||||
# draw.rectangle(pixcoords, outline=pink)
|
# draw.rectangle(pixcoords, outline='pink')
|
||||||
|
|
||||||
if options.threshold:
|
|
||||||
pix = leptonica.Pix.frompil(im)
|
|
||||||
pix = pix.masked_threshold_on_background_norm()
|
|
||||||
im_pix = pix.topil()
|
|
||||||
im_pix.info['dpi'] = im.info['dpi']
|
|
||||||
im = im_pix
|
|
||||||
|
|
||||||
del draw
|
|
||||||
|
|
||||||
filter_im = page_context.plugin_manager.hook.filter_ocr_image(
|
filter_im = page_context.plugin_manager.hook.filter_ocr_image(
|
||||||
page=page_context, image=im
|
page=page_context, image=im
|
||||||
@@ -615,7 +624,7 @@ def create_pdf_page_from_image(
|
|||||||
|
|
||||||
layout_fun = img2pdf.get_layout_fun(pagesize)
|
layout_fun = img2pdf.get_layout_fun(pagesize)
|
||||||
img2pdf.convert(
|
img2pdf.convert(
|
||||||
imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf
|
imfile, layout_fun=layout_fun, outputstream=pdf, **IMG2PDF_KWARGS
|
||||||
)
|
)
|
||||||
log.debug('convert done')
|
log.debug('convert done')
|
||||||
|
|
||||||
@@ -658,7 +667,7 @@ def ocr_engine_textonly_pdf(input_image: Path, page_context: PageContext):
|
|||||||
return (output_pdf, output_text)
|
return (output_pdf, output_text)
|
||||||
|
|
||||||
|
|
||||||
def get_docinfo(base_pdf: pikepdf.Pdf, context: PdfContext) -> Dict[str, str]:
|
def get_docinfo(base_pdf: pikepdf.Pdf, context: PdfContext) -> dict[str, str]:
|
||||||
options = context.options
|
options = context.options
|
||||||
|
|
||||||
def from_document_info(key):
|
def from_document_info(key):
|
||||||
@@ -672,15 +681,14 @@ def get_docinfo(base_pdf: pikepdf.Pdf, context: PdfContext) -> Dict[str, str]:
|
|||||||
k: from_document_info(k)
|
k: from_document_info(k)
|
||||||
for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate')
|
for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate')
|
||||||
}
|
}
|
||||||
if options is not None:
|
if options.title:
|
||||||
if options.title:
|
pdfmark['/Title'] = options.title
|
||||||
pdfmark['/Title'] = options.title
|
if options.author:
|
||||||
if options.author:
|
pdfmark['/Author'] = options.author
|
||||||
pdfmark['/Author'] = options.author
|
if options.keywords:
|
||||||
if options.keywords:
|
pdfmark['/Keywords'] = options.keywords
|
||||||
pdfmark['/Keywords'] = options.keywords
|
if options.subject:
|
||||||
if options.subject:
|
pdfmark['/Subject'] = options.subject
|
||||||
pdfmark['/Subject'] = options.subject
|
|
||||||
|
|
||||||
creator_tag = context.plugin_manager.hook.get_ocr_engine().creator_tag(options)
|
creator_tag = context.plugin_manager.hook.get_ocr_engine().creator_tag(options)
|
||||||
|
|
||||||
@@ -810,13 +818,14 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
|||||||
missing = set(meta_original.keys()) - set(meta.keys())
|
missing = set(meta_original.keys()) - set(meta.keys())
|
||||||
report_on_metadata(missing)
|
report_on_metadata(missing)
|
||||||
|
|
||||||
|
optimizing = context.plugin_manager.hook.is_optimization_enabled(
|
||||||
|
context=context
|
||||||
|
)
|
||||||
pdf.save(
|
pdf.save(
|
||||||
output_file,
|
output_file,
|
||||||
**get_pdf_save_settings(options.output_type),
|
**get_pdf_save_settings(options.output_type),
|
||||||
linearize=( # Don't linearize if optimize() will be linearizing too
|
linearize=( # Don't linearize if optimize() will be linearizing too
|
||||||
should_linearize(working_file, context)
|
not optimizing and should_linearize(working_file, context)
|
||||||
if options.optimize == 0
|
|
||||||
else False
|
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -825,12 +834,22 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
|||||||
|
|
||||||
def optimize_pdf(input_file: Path, context: PdfContext, executor: Executor):
|
def optimize_pdf(input_file: Path, context: PdfContext, executor: Executor):
|
||||||
output_file = context.get_path('optimize.pdf')
|
output_file = context.get_path('optimize.pdf')
|
||||||
save_settings = dict(
|
output_pdf, messages = context.plugin_manager.hook.optimize_pdf(
|
||||||
|
input_pdf=input_file,
|
||||||
|
output_pdf=output_file,
|
||||||
|
context=context,
|
||||||
|
executor=executor,
|
||||||
linearize=should_linearize(input_file, context),
|
linearize=should_linearize(input_file, context),
|
||||||
**get_pdf_save_settings(context.options.output_type),
|
|
||||||
)
|
)
|
||||||
optimize(input_file, output_file, context, save_settings, executor)
|
|
||||||
return output_file
|
input_size = input_file.stat().st_size
|
||||||
|
output_size = output_file.stat().st_size
|
||||||
|
if output_size > 0:
|
||||||
|
ratio = input_size / output_size
|
||||||
|
savings = 1 - output_size / input_size
|
||||||
|
log.info(f"Optimize ratio: {ratio:.2f} savings: {(savings):.1%}")
|
||||||
|
|
||||||
|
return output_pdf, messages
|
||||||
|
|
||||||
|
|
||||||
def enumerate_compress_ranges(iterable):
|
def enumerate_compress_ranges(iterable):
|
||||||
@@ -849,14 +868,14 @@ def enumerate_compress_ranges(iterable):
|
|||||||
yield (skipped_from, index), None
|
yield (skipped_from, index), None
|
||||||
|
|
||||||
|
|
||||||
def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
def merge_sidecars(txt_files: Iterable[Path | None], context: PdfContext):
|
||||||
output_file = context.get_path('sidecar.txt')
|
output_file = context.get_path('sidecar.txt')
|
||||||
with open(output_file, 'w', encoding="utf-8") as stream:
|
with open(output_file, 'w', encoding="utf-8") as stream:
|
||||||
for (frm, to), txt_file in enumerate_compress_ranges(txt_files):
|
for (from_, to_), txt_file in enumerate_compress_ranges(txt_files):
|
||||||
if frm != 1:
|
if from_ != 1:
|
||||||
stream.write('\f') # Form feed between pages
|
stream.write('\f') # Form feed between pages
|
||||||
if txt_file:
|
if txt_file:
|
||||||
with open(txt_file, 'r', encoding="utf-8") as in_:
|
with open(txt_file, encoding="utf-8") as in_:
|
||||||
txt = in_.read()
|
txt = in_.read()
|
||||||
# Some OCR engines (e.g. Tesseract v4 alpha) add form feeds
|
# Some OCR engines (e.g. Tesseract v4 alpha) add form feeds
|
||||||
# between pages, and some do not. For consistency, we ignore
|
# between pages, and some do not. For consistency, we ignore
|
||||||
@@ -866,10 +885,10 @@ def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
|||||||
else:
|
else:
|
||||||
stream.write(txt)
|
stream.write(txt)
|
||||||
else:
|
else:
|
||||||
if frm != to:
|
if from_ != to_:
|
||||||
pages = f'{frm}-{to}'
|
pages = f'{from_}-{to_}'
|
||||||
else:
|
else:
|
||||||
pages = f'{frm}'
|
pages = f'{from_}'
|
||||||
stream.write(f'[OCR skipped on page(s) {pages}]')
|
stream.write(f'[OCR skipped on page(s) {pages}]')
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|||||||
@@ -4,6 +4,9 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Plugin manager using pluggy."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import importlib
|
import importlib
|
||||||
@@ -11,7 +14,7 @@ import importlib.util
|
|||||||
import pkgutil
|
import pkgutil
|
||||||
import sys
|
import sys
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import List, Tuple, Union
|
from typing import Sequence
|
||||||
|
|
||||||
import pluggy
|
import pluggy
|
||||||
|
|
||||||
@@ -33,7 +36,7 @@ class OcrmypdfPluginManager(pluggy.PluginManager):
|
|||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
*args,
|
*args,
|
||||||
plugins: List[Union[str, Path]],
|
plugins: list[str | Path],
|
||||||
builtins: bool = True,
|
builtins: bool = True,
|
||||||
**kwargs,
|
**kwargs,
|
||||||
):
|
):
|
||||||
@@ -100,23 +103,28 @@ class OcrmypdfPluginManager(pluggy.PluginManager):
|
|||||||
self.register(module)
|
self.register(module)
|
||||||
|
|
||||||
|
|
||||||
def get_plugin_manager(plugins: List[Union[str, Path]], builtins=True):
|
def get_plugin_manager(plugins: list[str | Path], builtins=True):
|
||||||
pm = OcrmypdfPluginManager(
|
return OcrmypdfPluginManager(
|
||||||
project_name='ocrmypdf',
|
project_name='ocrmypdf',
|
||||||
plugins=plugins,
|
plugins=plugins,
|
||||||
builtins=builtins,
|
builtins=builtins,
|
||||||
)
|
)
|
||||||
return pm
|
|
||||||
|
|
||||||
|
|
||||||
def get_parser_options_plugins(
|
def get_parser_options_plugins(
|
||||||
args,
|
args: Sequence[str],
|
||||||
) -> Tuple[argparse.ArgumentParser, argparse.Namespace, pluggy.PluginManager]:
|
) -> tuple[argparse.ArgumentParser, argparse.Namespace, pluggy.PluginManager]:
|
||||||
pre_options, _unused = plugins_only_parser.parse_known_args(args=args)
|
pre_options, _unused = plugins_only_parser.parse_known_args(args=args)
|
||||||
plugin_manager = get_plugin_manager(pre_options.plugins)
|
plugin_manager = get_plugin_manager(pre_options.plugins)
|
||||||
|
|
||||||
parser = get_parser()
|
parser = get_parser()
|
||||||
|
plugin_manager.hook.initialize( # pylint: disable=no-member
|
||||||
|
plugin_manager=plugin_manager
|
||||||
|
)
|
||||||
plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
|
plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
|
||||||
|
|
||||||
options = parser.parse_args(args=args)
|
options = parser.parse_args(args=args)
|
||||||
return parser, options, plugin_manager
|
return parser, options, plugin_manager
|
||||||
|
|
||||||
|
|
||||||
|
__all__ = ['OcrmypdfPluginManager', 'get_plugin_manager', 'get_parser_options_plugins']
|
||||||
|
|||||||
+67
-25
@@ -4,16 +4,23 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Implements the concurrent and page synchronous parts of the pipeline."""
|
||||||
|
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
import logging
|
import logging
|
||||||
import logging.handlers
|
import logging.handlers
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
|
from concurrent.futures.process import BrokenProcessPool
|
||||||
|
from concurrent.futures.thread import BrokenThreadPool
|
||||||
from functools import partial
|
from functools import partial
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from tempfile import mkdtemp
|
from tempfile import mkdtemp
|
||||||
from typing import List, NamedTuple, Optional, Tuple
|
from typing import NamedTuple, Sequence, cast
|
||||||
|
|
||||||
import PIL
|
import PIL
|
||||||
|
|
||||||
@@ -46,7 +53,7 @@ from ocrmypdf._pipeline import (
|
|||||||
triage,
|
triage,
|
||||||
validate_pdfinfo_options,
|
validate_pdfinfo_options,
|
||||||
)
|
)
|
||||||
from ocrmypdf._plugin_manager import get_plugin_manager
|
from ocrmypdf._plugin_manager import OcrmypdfPluginManager, get_plugin_manager
|
||||||
from ocrmypdf._validation import (
|
from ocrmypdf._validation import (
|
||||||
check_requested_output_file,
|
check_requested_output_file,
|
||||||
create_input_file,
|
create_input_file,
|
||||||
@@ -65,11 +72,13 @@ from ocrmypdf.pdfa import file_claims_pdfa
|
|||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
class PageResult(NamedTuple): # pylint: disable=inherit-non-class
|
class PageResult(NamedTuple):
|
||||||
|
"""Result when a page is finished processing."""
|
||||||
|
|
||||||
pageno: int
|
pageno: int
|
||||||
pdf_page_from_image: Optional[Path]
|
pdf_page_from_image: Path | None
|
||||||
ocr: Optional[Path]
|
ocr: Path | None
|
||||||
text: Optional[Path]
|
text: Path | None
|
||||||
orientation_correction: int
|
orientation_correction: int
|
||||||
|
|
||||||
|
|
||||||
@@ -108,7 +117,7 @@ def preprocess(
|
|||||||
|
|
||||||
def make_intermediate_images(
|
def make_intermediate_images(
|
||||||
page_context: PageContext, orientation_correction: int
|
page_context: PageContext, orientation_correction: int
|
||||||
) -> Tuple[Path, Optional[Path]]:
|
) -> tuple[Path, Path | None]:
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
|
|
||||||
ocr_image = preprocess_out = None
|
ocr_image = preprocess_out = None
|
||||||
@@ -165,7 +174,7 @@ def make_intermediate_images(
|
|||||||
return ocr_image, preprocess_out
|
return ocr_image, preprocess_out
|
||||||
|
|
||||||
|
|
||||||
def exec_page_sync(page_context: PageContext):
|
def exec_page_sync(page_context: PageContext) -> PageResult:
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
tls.pageno = page_context.pageno + 1
|
tls.pageno = page_context.pageno + 1
|
||||||
|
|
||||||
@@ -223,7 +232,9 @@ def exec_page_sync(page_context: PageContext):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def post_process(pdf_file, context: PdfContext, executor: Executor):
|
def post_process(
|
||||||
|
pdf_file: Path, context: PdfContext, executor: Executor
|
||||||
|
) -> tuple[Path, Sequence[str]]:
|
||||||
pdf_out = pdf_file
|
pdf_out = pdf_file
|
||||||
if context.options.output_type.startswith('pdfa'):
|
if context.options.output_type.startswith('pdfa'):
|
||||||
ps_stub_out = generate_postscript_stub(context)
|
ps_stub_out = generate_postscript_stub(context)
|
||||||
@@ -233,7 +244,7 @@ def post_process(pdf_file, context: PdfContext, executor: Executor):
|
|||||||
return optimize_pdf(pdf_out, context, executor)
|
return optimize_pdf(pdf_out, context, executor)
|
||||||
|
|
||||||
|
|
||||||
def worker_init(max_pixels: int):
|
def worker_init(max_pixels: int) -> None:
|
||||||
# In Windows, child process will not inherit our change to this value in
|
# In Windows, child process will not inherit our change to this value in
|
||||||
# the parent process, so ensure workers get it set. Not needed when running
|
# the parent process, so ensure workers get it set. Not needed when running
|
||||||
# threaded, but harmless to set again.
|
# threaded, but harmless to set again.
|
||||||
@@ -241,7 +252,7 @@ def worker_init(max_pixels: int):
|
|||||||
pikepdf_enable_mmap()
|
pikepdf_enable_mmap()
|
||||||
|
|
||||||
|
|
||||||
def exec_concurrent(context: PdfContext, executor: Executor):
|
def exec_concurrent(context: PdfContext, executor: Executor) -> Sequence[str]:
|
||||||
"""Execute the pipeline concurrently"""
|
"""Execute the pipeline concurrently"""
|
||||||
|
|
||||||
# Run exec_page_sync on every page context
|
# Run exec_page_sync on every page context
|
||||||
@@ -250,7 +261,7 @@ def exec_concurrent(context: PdfContext, executor: Executor):
|
|||||||
if max_workers > 1:
|
if max_workers > 1:
|
||||||
log.info("Start processing %d pages concurrently", max_workers)
|
log.info("Start processing %d pages concurrently", max_workers)
|
||||||
|
|
||||||
sidecars: List[Optional[Path]] = [None] * len(context.pdfinfo)
|
sidecars: list[Path | None] = [None] * len(context.pdfinfo)
|
||||||
ocrgraft = OcrGrafter(context)
|
ocrgraft = OcrGrafter(context)
|
||||||
|
|
||||||
def update_page(result: PageResult, pbar):
|
def update_page(result: PageResult, pbar):
|
||||||
@@ -293,15 +304,20 @@ def exec_concurrent(context: PdfContext, executor: Executor):
|
|||||||
# Merge layers to one single pdf
|
# Merge layers to one single pdf
|
||||||
pdf = ocrgraft.finalize()
|
pdf = ocrgraft.finalize()
|
||||||
|
|
||||||
# PDF/A and metadata
|
messages: Sequence[str] = []
|
||||||
log.info("Postprocessing...")
|
if options.output_type != 'none':
|
||||||
pdf = post_process(pdf, context, executor)
|
# PDF/A and metadata
|
||||||
|
log.info("Postprocessing...")
|
||||||
|
pdf, messages = post_process(pdf, context, executor)
|
||||||
|
|
||||||
# Copy PDF file to destination
|
# Copy PDF file to destination
|
||||||
copy_final(pdf, options.output_file, context)
|
copy_final(pdf, options.output_file, context)
|
||||||
|
return messages
|
||||||
|
|
||||||
|
|
||||||
def configure_debug_logging(log_filename: Path, prefix: str = ''):
|
def configure_debug_logging(
|
||||||
|
log_filename: Path, prefix: str = ''
|
||||||
|
) -> logging.FileHandler:
|
||||||
"""
|
"""
|
||||||
Create a debug log file at a specified location.
|
Create a debug log file at a specified location.
|
||||||
|
|
||||||
@@ -320,7 +336,12 @@ def configure_debug_logging(log_filename: Path, prefix: str = ''):
|
|||||||
return log_file_handler
|
return log_file_handler
|
||||||
|
|
||||||
|
|
||||||
def run_pipeline(options, *, plugin_manager, api=False):
|
def run_pipeline(
|
||||||
|
options: argparse.Namespace,
|
||||||
|
*,
|
||||||
|
plugin_manager: OcrmypdfPluginManager | None,
|
||||||
|
api: bool = False,
|
||||||
|
) -> ExitCode:
|
||||||
# Any changes to options will not take effect for options that are already
|
# Any changes to options will not take effect for options that are already
|
||||||
# bound to function parameters in the pipeline. (For example
|
# bound to function parameters in the pipeline. (For example
|
||||||
# options.input_file, options.pdf_renderer are already bound.)
|
# options.input_file, options.pdf_renderer are already bound.)
|
||||||
@@ -371,7 +392,7 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
validate_pdfinfo_options(context)
|
validate_pdfinfo_options(context)
|
||||||
|
|
||||||
# Execute the pipeline
|
# Execute the pipeline
|
||||||
exec_concurrent(context, executor)
|
optimize_messages = exec_concurrent(context, executor)
|
||||||
|
|
||||||
if options.output_file == '-':
|
if options.output_file == '-':
|
||||||
log.info("Output sent to stdout")
|
log.info("Output sent to stdout")
|
||||||
@@ -379,7 +400,7 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
hasattr(options.output_file, 'writable') and options.output_file.writable()
|
hasattr(options.output_file, 'writable') and options.output_file.writable()
|
||||||
):
|
):
|
||||||
log.info("Output written to stream")
|
log.info("Output written to stream")
|
||||||
elif samefile(options.output_file, os.devnull):
|
elif samefile(options.output_file, Path(os.devnull)):
|
||||||
pass # Say nothing when sending to dev null
|
pass # Say nothing when sending to dev null
|
||||||
else:
|
else:
|
||||||
if options.output_type.startswith('pdfa'):
|
if options.output_type.startswith('pdfa'):
|
||||||
@@ -397,15 +418,18 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
if not check_pdf(options.output_file):
|
if not check_pdf(options.output_file):
|
||||||
log.warning('Output file: The generated PDF is INVALID')
|
log.warning('Output file: The generated PDF is INVALID')
|
||||||
return ExitCode.invalid_output_pdf
|
return ExitCode.invalid_output_pdf
|
||||||
report_output_file_size(options, start_input_file, options.output_file)
|
report_output_file_size(
|
||||||
|
options, start_input_file, options.output_file, optimize_messages
|
||||||
|
)
|
||||||
|
|
||||||
except (KeyboardInterrupt if not api else NeverRaise) as e:
|
except (KeyboardInterrupt if not api else NeverRaise):
|
||||||
if options.verbose >= 1:
|
if options.verbose >= 1:
|
||||||
log.exception("KeyboardInterrupt")
|
log.exception("KeyboardInterrupt")
|
||||||
else:
|
else:
|
||||||
log.error("KeyboardInterrupt")
|
log.error("KeyboardInterrupt")
|
||||||
return ExitCode.ctrl_c
|
return ExitCode.ctrl_c
|
||||||
except (ExitCodeException if not api else NeverRaise) as e:
|
except (ExitCodeException if not api else NeverRaise) as e:
|
||||||
|
e = cast(ExitCodeException, e)
|
||||||
if options.verbose >= 1:
|
if options.verbose >= 1:
|
||||||
log.exception("ExitCodeException")
|
log.exception("ExitCodeException")
|
||||||
elif str(e):
|
elif str(e):
|
||||||
@@ -413,7 +437,25 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
else:
|
else:
|
||||||
log.error(type(e).__name__)
|
log.error(type(e).__name__)
|
||||||
return e.exit_code
|
return e.exit_code
|
||||||
except (Exception if not api else NeverRaise) as e: # pylint: disable=broad-except
|
except (PIL.Image.DecompressionBombError if not api else NeverRaise):
|
||||||
|
log.exception(
|
||||||
|
"A decompression bomb error was encountered while executing the "
|
||||||
|
"pipeline. Use the argument --max-image-mpixels to raise the maximum "
|
||||||
|
"image pixel limit."
|
||||||
|
)
|
||||||
|
return ExitCode.other_error
|
||||||
|
except (
|
||||||
|
BrokenProcessPool if not api else NeverRaise,
|
||||||
|
BrokenThreadPool if not api else NeverRaise,
|
||||||
|
):
|
||||||
|
log.exception(
|
||||||
|
"A worker process was terminated unexpectedly. This is known to occur if "
|
||||||
|
"processing your file takes all available swap space and RAM. It may "
|
||||||
|
"help to try again with a smaller number of jobs, using the --jobs "
|
||||||
|
"argument."
|
||||||
|
)
|
||||||
|
return ExitCode.child_process_error
|
||||||
|
except (Exception if not api else NeverRaise): # pylint: disable=broad-except
|
||||||
log.exception("An exception occurred while executing the pipeline")
|
log.exception("An exception occurred while executing the pipeline")
|
||||||
return ExitCode.other_error
|
return ExitCode.other_error
|
||||||
finally:
|
finally:
|
||||||
@@ -421,7 +463,7 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
|||||||
try:
|
try:
|
||||||
debug_log_handler.close()
|
debug_log_handler.close()
|
||||||
log.removeHandler(debug_log_handler)
|
log.removeHandler(debug_log_handler)
|
||||||
except EnvironmentError as e:
|
except OSError as e:
|
||||||
print(e, file=sys.stderr)
|
print(e, file=sys.stderr)
|
||||||
cleanup_working_files(work_folder, options)
|
cleanup_working_files(work_folder, options)
|
||||||
|
|
||||||
|
|||||||
@@ -1,118 +0,0 @@
|
|||||||
# Copyright (c) 2014, Armin Ronacher
|
|
||||||
#
|
|
||||||
# Copyright (c) 2017, James R Barlow
|
|
||||||
#
|
|
||||||
# Some rights reserved.
|
|
||||||
#
|
|
||||||
# Redistribution and use in source and binary forms, with or without
|
|
||||||
# modification, are permitted provided that the following conditions are
|
|
||||||
# met:
|
|
||||||
#
|
|
||||||
# * Redistributions of source code must retain the above copyright
|
|
||||||
# notice, this list of conditions and the following disclaimer.
|
|
||||||
#
|
|
||||||
# * Redistributions in binary form must reproduce the above
|
|
||||||
# copyright notice, this list of conditions and the following
|
|
||||||
# disclaimer in the documentation and/or other materials provided
|
|
||||||
# with the distribution.
|
|
||||||
#
|
|
||||||
# * The names of the contributors may not be used to endorse or
|
|
||||||
# promote products derived from this software without specific
|
|
||||||
# prior written permission.
|
|
||||||
#
|
|
||||||
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
|
||||||
# "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
|
||||||
# LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
|
||||||
# A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
|
||||||
# OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
|
||||||
# SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
|
||||||
# LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
|
||||||
# DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
|
||||||
# THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
|
||||||
# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
|
||||||
# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
|
||||||
|
|
||||||
|
|
||||||
import codecs
|
|
||||||
import os
|
|
||||||
import sys
|
|
||||||
|
|
||||||
|
|
||||||
def verify_python3_env(): # pragma: no cover
|
|
||||||
"""Ensures that the environment is good for unicode on Python 3."""
|
|
||||||
|
|
||||||
# PEP 538 changes in Python 3.7 should make this wrangling unnecessary
|
|
||||||
if sys.version_info[0:3] >= (3, 7, 0):
|
|
||||||
return
|
|
||||||
|
|
||||||
try:
|
|
||||||
import locale
|
|
||||||
|
|
||||||
fs_enc = codecs.lookup(locale.getpreferredencoding()).name
|
|
||||||
except Exception:
|
|
||||||
fs_enc = 'ascii'
|
|
||||||
if fs_enc != 'ascii':
|
|
||||||
return
|
|
||||||
|
|
||||||
extra = ''
|
|
||||||
if os.name == 'posix':
|
|
||||||
import subprocess
|
|
||||||
|
|
||||||
rv = subprocess.run(
|
|
||||||
['locale', '-a'], stdout=subprocess.PIPE, stderr=subprocess.PIPE
|
|
||||||
).stdout
|
|
||||||
good_locales = set()
|
|
||||||
has_c_utf8 = False
|
|
||||||
|
|
||||||
# Make sure we're operating on text here.
|
|
||||||
if isinstance(rv, bytes):
|
|
||||||
rv = rv.decode('ascii', 'replace')
|
|
||||||
|
|
||||||
for line in rv.splitlines():
|
|
||||||
locale = line.strip()
|
|
||||||
if locale.lower().endswith(('.utf-8', '.utf8')):
|
|
||||||
good_locales.add(locale)
|
|
||||||
if locale.lower() in ('c.utf8', 'c.utf-8'):
|
|
||||||
has_c_utf8 = True
|
|
||||||
|
|
||||||
extra += '\n\n'
|
|
||||||
if not good_locales:
|
|
||||||
extra += (
|
|
||||||
'Additional information: on this system no suitable UTF-8\n'
|
|
||||||
'locales were discovered. This most likely requires resolving\n'
|
|
||||||
'by reconfiguring the locale system.'
|
|
||||||
)
|
|
||||||
elif has_c_utf8:
|
|
||||||
extra += (
|
|
||||||
'This system supports the C.UTF-8 locale which is recommended.\n'
|
|
||||||
'You might be able to resolve your issue by exporting the\n'
|
|
||||||
'following environment variables:\n\n'
|
|
||||||
' export LC_ALL=C.UTF-8\n'
|
|
||||||
' export LANG=C.UTF-8'
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
extra += (
|
|
||||||
'This system lists a couple of UTF-8 supporting locales that\n'
|
|
||||||
'you can pick from. The following suitable locales were\n'
|
|
||||||
'discovered: %s'
|
|
||||||
) % ', '.join(sorted(good_locales))
|
|
||||||
|
|
||||||
bad_locale = None
|
|
||||||
for locale in os.environ.get('LC_ALL'), os.environ.get('LANG'):
|
|
||||||
if locale and locale.lower().endswith(('.utf-8', '.utf8')):
|
|
||||||
bad_locale = locale
|
|
||||||
if locale is not None:
|
|
||||||
break
|
|
||||||
if bad_locale is not None:
|
|
||||||
extra += (
|
|
||||||
'\nocrmypdf discovered that you exported a UTF-8 locale\n'
|
|
||||||
'but the locale system could not pick up from it because\n'
|
|
||||||
'it does not exist. The exported locale is "%s" but it\n'
|
|
||||||
'is not supported'
|
|
||||||
) % bad_locale
|
|
||||||
|
|
||||||
raise RuntimeError(
|
|
||||||
'ocrmypdf will abort further execution because Python 3 '
|
|
||||||
'was configured to use ASCII as encoding for the '
|
|
||||||
'environment.' + extra
|
|
||||||
)
|
|
||||||
+54
-135
@@ -5,6 +5,9 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Validate a work order from API or command line."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import locale
|
import locale
|
||||||
import logging
|
import logging
|
||||||
@@ -13,25 +16,19 @@ import sys
|
|||||||
import unicodedata
|
import unicodedata
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj
|
from shutil import copyfileobj
|
||||||
from typing import List, Set, Tuple, Union
|
from typing import Sequence
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
import PIL
|
import PIL
|
||||||
|
|
||||||
from ocrmypdf._exec import jbig2enc, pngquant, unpaper
|
from ocrmypdf._exec import unpaper
|
||||||
from ocrmypdf._unicodefun import verify_python3_env
|
|
||||||
from ocrmypdf.exceptions import (
|
from ocrmypdf.exceptions import (
|
||||||
BadArgsError,
|
BadArgsError,
|
||||||
InputFileError,
|
InputFileError,
|
||||||
MissingDependencyError,
|
MissingDependencyError,
|
||||||
OutputFileAccessError,
|
OutputFileAccessError,
|
||||||
)
|
)
|
||||||
from ocrmypdf.helpers import (
|
from ocrmypdf.helpers import is_file_writable, monotonic, safe_symlink
|
||||||
is_file_writable,
|
|
||||||
is_iterable_notstr,
|
|
||||||
monotonic,
|
|
||||||
safe_symlink,
|
|
||||||
)
|
|
||||||
from ocrmypdf.hocrtransform import HOCR_OK_LANGS
|
from ocrmypdf.hocrtransform import HOCR_OK_LANGS
|
||||||
from ocrmypdf.subprocess import check_external_program
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
@@ -44,12 +41,10 @@ log = logging.getLogger(__name__)
|
|||||||
|
|
||||||
|
|
||||||
# --------
|
# --------
|
||||||
# Critical environment tests
|
|
||||||
verify_python3_env()
|
|
||||||
|
|
||||||
|
|
||||||
def check_platform():
|
def check_platform():
|
||||||
if os.name == 'nt' and sys.maxsize <= 2 ** 32: # pragma: no cover
|
if os.name == 'nt' and sys.maxsize <= 2**32: # pragma: no cover
|
||||||
# 32-bit interpreter on Windows
|
# 32-bit interpreter on Windows
|
||||||
log.error(
|
log.error(
|
||||||
"You are running OCRmyPDF in a 32-bit (x86) Python interpreter."
|
"You are running OCRmyPDF in a 32-bit (x86) Python interpreter."
|
||||||
@@ -68,7 +63,7 @@ def check_options_languages(options, ocr_engine_languages):
|
|||||||
missing_languages = options.languages - ocr_engine_languages
|
missing_languages = options.languages - ocr_engine_languages
|
||||||
if missing_languages:
|
if missing_languages:
|
||||||
msg = (
|
msg = (
|
||||||
f"OCR engine does not have language data for the following "
|
"OCR engine does not have language data for the following "
|
||||||
"requested languages: \n"
|
"requested languages: \n"
|
||||||
)
|
)
|
||||||
msg += '\n'.join(lang for lang in missing_languages)
|
msg += '\n'.join(lang for lang in missing_languages)
|
||||||
@@ -80,12 +75,18 @@ def check_options_output(options):
|
|||||||
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
||||||
|
|
||||||
if options.pdf_renderer.startswith('hocr') and not is_latin:
|
if options.pdf_renderer.startswith('hocr') and not is_latin:
|
||||||
msg = (
|
log.warning(
|
||||||
"The 'hocr' PDF renderer is known to cause problems with one "
|
"The 'hocr' PDF renderer is known to cause problems with one "
|
||||||
"or more of the languages in your document. Use "
|
"or more of the languages in your document. Use "
|
||||||
"--pdf-renderer auto (the default) to avoid this issue."
|
"`--pdf-renderer auto` (the default) to avoid this issue."
|
||||||
|
)
|
||||||
|
|
||||||
|
if options.output_type == 'none' and options.output_file not in (os.devnull, '-'):
|
||||||
|
raise BadArgsError(
|
||||||
|
"Since you specified `--output-type none`, the output file "
|
||||||
|
f"{options.output_file} cannot be produced. Set the output file to "
|
||||||
|
f"`-` to suppress this message."
|
||||||
)
|
)
|
||||||
log.warning(msg)
|
|
||||||
|
|
||||||
lossless_reconstruction = False
|
lossless_reconstruction = False
|
||||||
if not any(
|
if not any(
|
||||||
@@ -112,6 +113,10 @@ def check_options_sidecar(options):
|
|||||||
raise BadArgsError(
|
raise BadArgsError(
|
||||||
"--sidecar filename must be specified when output file is stdout."
|
"--sidecar filename must be specified when output file is stdout."
|
||||||
)
|
)
|
||||||
|
elif options.output_file == os.devnull:
|
||||||
|
raise BadArgsError(
|
||||||
|
"--sidecar filename must be specified when output file is /dev/null or NUL."
|
||||||
|
)
|
||||||
options.sidecar = options.output_file + '.txt'
|
options.sidecar = options.output_file + '.txt'
|
||||||
if options.sidecar == options.input_file or options.sidecar == options.output_file:
|
if options.sidecar == options.input_file or options.sidecar == options.output_file:
|
||||||
raise BadArgsError(
|
raise BadArgsError(
|
||||||
@@ -141,26 +146,26 @@ def check_options_preprocessing(options):
|
|||||||
raise BadArgsError("--unpaper-args: " + str(e)) from e
|
raise BadArgsError("--unpaper-args: " + str(e)) from e
|
||||||
|
|
||||||
|
|
||||||
def _pages_from_ranges(ranges: str) -> Set[int]:
|
def _pages_from_ranges(ranges: str) -> set[int]:
|
||||||
if is_iterable_notstr(ranges):
|
pages: list[int] = []
|
||||||
return set(ranges)
|
|
||||||
pages: List[int] = []
|
|
||||||
page_groups = ranges.replace(' ', '').split(',')
|
page_groups = ranges.replace(' ', '').split(',')
|
||||||
for g in page_groups:
|
for group in page_groups:
|
||||||
if not g:
|
if not group:
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
start, end = g.split('-')
|
start, end = group.split('-')
|
||||||
except ValueError:
|
except ValueError:
|
||||||
pages.append(int(g) - 1)
|
pages.append(int(group) - 1)
|
||||||
else:
|
else:
|
||||||
try:
|
try:
|
||||||
new_pages = list(range(int(start) - 1, int(end)))
|
new_pages = list(range(int(start) - 1, int(end)))
|
||||||
if not new_pages:
|
if not new_pages:
|
||||||
raise BadArgsError(f"invalid page subrange '{start}-{end}'")
|
raise BadArgsError(
|
||||||
|
f"invalid page subrange '{start}-{end}'"
|
||||||
|
) from None
|
||||||
pages.extend(new_pages)
|
pages.extend(new_pages)
|
||||||
except ValueError:
|
except ValueError:
|
||||||
raise BadArgsError("invalid page range") from None
|
raise BadArgsError(f"invalid page subrange '{group}'") from None
|
||||||
|
|
||||||
if not pages:
|
if not pages:
|
||||||
raise BadArgsError(
|
raise BadArgsError(
|
||||||
@@ -182,10 +187,8 @@ def _pages_from_ranges(ranges: str) -> Set[int]:
|
|||||||
|
|
||||||
def check_options_ocr_behavior(options):
|
def check_options_ocr_behavior(options):
|
||||||
exclusive_options = sum(
|
exclusive_options = sum(
|
||||||
[
|
(1 if opt else 0)
|
||||||
(1 if opt else 0)
|
for opt in (options.force_ocr, options.skip_text, options.redo_ocr)
|
||||||
for opt in (options.force_ocr, options.skip_text, options.redo_ocr)
|
|
||||||
]
|
|
||||||
)
|
)
|
||||||
if exclusive_options >= 2:
|
if exclusive_options >= 2:
|
||||||
raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.")
|
raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.")
|
||||||
@@ -193,37 +196,6 @@ def check_options_ocr_behavior(options):
|
|||||||
options.pages = _pages_from_ranges(options.pages)
|
options.pages = _pages_from_ranges(options.pages)
|
||||||
|
|
||||||
|
|
||||||
def check_options_optimizing(options):
|
|
||||||
if options.optimize >= 2:
|
|
||||||
check_external_program(
|
|
||||||
program='pngquant',
|
|
||||||
package='pngquant',
|
|
||||||
version_checker=pngquant.version,
|
|
||||||
need_version='2.0.1',
|
|
||||||
required_for='--optimize {2,3}',
|
|
||||||
)
|
|
||||||
|
|
||||||
if options.optimize >= 2:
|
|
||||||
# Although we use JBIG2 for optimize=1, don't nag about it unless the
|
|
||||||
# user is asking for more optimization
|
|
||||||
check_external_program(
|
|
||||||
program='jbig2',
|
|
||||||
package='jbig2enc',
|
|
||||||
version_checker=jbig2enc.version,
|
|
||||||
need_version='0.28',
|
|
||||||
required_for='--optimize {2,3} | --jbig2-lossy',
|
|
||||||
recommended=True if not options.jbig2_lossy else False,
|
|
||||||
)
|
|
||||||
|
|
||||||
if options.optimize == 0 and any(
|
|
||||||
[options.jbig2_lossy, options.png_quality, options.jpeg_quality]
|
|
||||||
):
|
|
||||||
log.warning(
|
|
||||||
"The arguments --jbig2-lossy, --png-quality, and --jpeg-quality "
|
|
||||||
"will be ignored because --optimize=0."
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def check_options_advanced(options):
|
def check_options_advanced(options):
|
||||||
if options.pdfa_image_compression != 'auto' and not options.output_type.startswith(
|
if options.pdfa_image_compression != 'auto' and not options.output_type.startswith(
|
||||||
'pdfa'
|
'pdfa'
|
||||||
@@ -237,13 +209,13 @@ def check_options_advanced(options):
|
|||||||
def check_options_metadata(options):
|
def check_options_metadata(options):
|
||||||
docinfo = [options.title, options.author, options.keywords, options.subject]
|
docinfo = [options.title, options.author, options.keywords, options.subject]
|
||||||
for s in (m for m in docinfo if m):
|
for s in (m for m in docinfo if m):
|
||||||
for c in s:
|
for char in s:
|
||||||
if unicodedata.category(c) == 'Co' or ord(c) >= 0x10000:
|
if unicodedata.category(char) == 'Co' or ord(char) >= 0x10000:
|
||||||
|
hexchar = hex(ord(char))[2:].upper()
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
"One of the metadata strings contains "
|
"One of the metadata strings contains "
|
||||||
"an unsupported Unicode character: '{}' (U+{})".format(
|
"an unsupported Unicode character: "
|
||||||
c, hex(ord(c))[2:].upper()
|
f"{char} (U+{hexchar})"
|
||||||
)
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
@@ -261,7 +233,6 @@ def _check_options(options, plugin_manager, ocr_engine_languages):
|
|||||||
check_options_sidecar(options)
|
check_options_sidecar(options)
|
||||||
check_options_preprocessing(options)
|
check_options_preprocessing(options)
|
||||||
check_options_ocr_behavior(options)
|
check_options_ocr_behavior(options)
|
||||||
check_options_optimizing(options)
|
|
||||||
check_options_advanced(options)
|
check_options_advanced(options)
|
||||||
check_options_pillow(options)
|
check_options_pillow(options)
|
||||||
plugin_manager.hook.check_options(options=options)
|
plugin_manager.hook.check_options(options=options)
|
||||||
@@ -272,55 +243,7 @@ def check_options(options, plugin_manager):
|
|||||||
_check_options(options, plugin_manager, ocr_engine_languages)
|
_check_options(options, plugin_manager, ocr_engine_languages)
|
||||||
|
|
||||||
|
|
||||||
def check_closed_streams(options): # pragma: no cover
|
def create_input_file(options, work_folder: Path) -> tuple[Path, str]:
|
||||||
"""Work around Python issue with multiprocessing forking on closed streams
|
|
||||||
|
|
||||||
https://bugs.python.org/issue28326
|
|
||||||
|
|
||||||
Attempting to a fork/exec a new Python process when any of std{in,out,err}
|
|
||||||
are closed or not flushable for some reason may raise an exception.
|
|
||||||
Fix this by opening devnull if the handle seems to be closed. Do this
|
|
||||||
globally to avoid tracking all places that fork.
|
|
||||||
|
|
||||||
Seems to be specific to multiprocessing.Process not all Python process
|
|
||||||
forkers.
|
|
||||||
|
|
||||||
The error actually occurs when the stream object is not flushable,
|
|
||||||
but replacing an open stream object that is not flushable with
|
|
||||||
/dev/null is a bad idea since it will create a silent failure. Replacing
|
|
||||||
a closed handle with /dev/null seems safe.
|
|
||||||
|
|
||||||
"""
|
|
||||||
|
|
||||||
if sys.version_info[0:3] >= (3, 6, 4):
|
|
||||||
return True # Issued fixed in Python 3.6.4+
|
|
||||||
|
|
||||||
if sys.stderr is None:
|
|
||||||
sys.stderr = open(os.devnull, 'w')
|
|
||||||
|
|
||||||
if sys.stdin is None:
|
|
||||||
if options.input_file == '-':
|
|
||||||
log.error("Trying to read from stdin but stdin seems closed")
|
|
||||||
return False
|
|
||||||
sys.stdin = open(os.devnull, 'r')
|
|
||||||
|
|
||||||
if sys.stdout is None:
|
|
||||||
if options.output_file == '-':
|
|
||||||
# Can't replace stdout if the user is piping
|
|
||||||
# If this case can even happen, it must be some kind of weird
|
|
||||||
# stream.
|
|
||||||
log.error(
|
|
||||||
"Output was set to stdout '-' but the stream attached to "
|
|
||||||
"stdout does not support the flush() system call. This "
|
|
||||||
"will fail."
|
|
||||||
)
|
|
||||||
return False
|
|
||||||
sys.stdout = open(os.devnull, 'w')
|
|
||||||
|
|
||||||
return True
|
|
||||||
|
|
||||||
|
|
||||||
def create_input_file(options, work_folder: Path) -> Tuple[Path, str]:
|
|
||||||
if options.input_file == '-':
|
if options.input_file == '-':
|
||||||
# stdin
|
# stdin
|
||||||
log.info('reading file from standard input')
|
log.info('reading file from standard input')
|
||||||
@@ -341,7 +264,7 @@ def create_input_file(options, work_folder: Path) -> Tuple[Path, str]:
|
|||||||
target = work_folder / 'origin'
|
target = work_folder / 'origin'
|
||||||
safe_symlink(options.input_file, target)
|
safe_symlink(options.input_file, target)
|
||||||
return target, os.fspath(options.input_file)
|
return target, os.fspath(options.input_file)
|
||||||
except FileNotFoundError:
|
except FileNotFoundError as e:
|
||||||
msg = f"File not found - {options.input_file}"
|
msg = f"File not found - {options.input_file}"
|
||||||
if Path('/.dockerenv').exists(): # pragma: no cover
|
if Path('/.dockerenv').exists(): # pragma: no cover
|
||||||
msg += (
|
msg += (
|
||||||
@@ -352,7 +275,7 @@ def create_input_file(options, work_folder: Path) -> Tuple[Path, str]:
|
|||||||
"\n"
|
"\n"
|
||||||
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf\n"
|
"\tdocker run -i --rm jbarlow83/ocrmypdf - - <input.pdf >output.pdf\n"
|
||||||
)
|
)
|
||||||
raise InputFileError(msg)
|
raise InputFileError(msg) from e
|
||||||
|
|
||||||
|
|
||||||
def check_requested_output_file(options):
|
def check_requested_output_file(options):
|
||||||
@@ -372,7 +295,16 @@ def check_requested_output_file(options):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def report_output_file_size(options, input_file, output_file):
|
def report_output_file_size(
|
||||||
|
options,
|
||||||
|
input_file: Path,
|
||||||
|
output_file: Path,
|
||||||
|
optimize_messages: Sequence[str] | None = None,
|
||||||
|
file_overhead: int = 4000,
|
||||||
|
page_overhead: int = 3000,
|
||||||
|
):
|
||||||
|
if optimize_messages is None:
|
||||||
|
optimize_messages = []
|
||||||
try:
|
try:
|
||||||
output_size = Path(output_file).stat().st_size
|
output_size = Path(output_file).stat().st_size
|
||||||
input_size = Path(input_file).stat().st_size
|
input_size = Path(input_file).stat().st_size
|
||||||
@@ -381,9 +313,7 @@ def report_output_file_size(options, input_file, output_file):
|
|||||||
with pikepdf.open(output_file) as p:
|
with pikepdf.open(output_file) as p:
|
||||||
# Overhead constants obtained by estimating amount of data added by OCR
|
# Overhead constants obtained by estimating amount of data added by OCR
|
||||||
# PDF/A conversion, and possible XMP metadata addition, with compression
|
# PDF/A conversion, and possible XMP metadata addition, with compression
|
||||||
FILE_OVERHEAD = 4000
|
reasonable_overhead = file_overhead + page_overhead * len(p.pages)
|
||||||
OCR_PER_PAGE_OVERHEAD = 3000
|
|
||||||
reasonable_overhead = FILE_OVERHEAD + OCR_PER_PAGE_OVERHEAD * len(p.pages)
|
|
||||||
ratio = output_size / input_size
|
ratio = output_size / input_size
|
||||||
reasonable_ratio = output_size / (input_size + reasonable_overhead)
|
reasonable_ratio = output_size / (input_size + reasonable_overhead)
|
||||||
if reasonable_ratio < 1.35 or input_size < 25000:
|
if reasonable_ratio < 1.35 or input_size < 25000:
|
||||||
@@ -403,19 +333,8 @@ def report_output_file_size(options, input_file, output_file):
|
|||||||
f"The argument --{arg.replace('_', '-')} was issued, causing transcoding."
|
f"The argument --{arg.replace('_', '-')} was issued, causing transcoding."
|
||||||
)
|
)
|
||||||
|
|
||||||
if options.optimize == 0:
|
reasons.extend(optimize_messages)
|
||||||
reasons.append("Optimization was disabled.")
|
|
||||||
else:
|
|
||||||
image_optimizers = {
|
|
||||||
'jbig2': jbig2enc.available(),
|
|
||||||
'pngquant': pngquant.available(),
|
|
||||||
}
|
|
||||||
for name, available in image_optimizers.items():
|
|
||||||
if not available:
|
|
||||||
reasons.append(
|
|
||||||
f"The optional dependency '{name}' was not found, so some image "
|
|
||||||
f"optimizations could not be attempted."
|
|
||||||
)
|
|
||||||
if options.output_type.startswith('pdfa'):
|
if options.output_type.startswith('pdfa'):
|
||||||
reasons.append("PDF/A conversion was enabled. (Try `--output-type pdf`.)")
|
reasons.append("PDF/A conversion was enabled. (Try `--output-type pdf`.)")
|
||||||
if options.plugins:
|
if options.plugins:
|
||||||
|
|||||||
@@ -4,10 +4,19 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Get version by introspecting package information.
|
||||||
|
|
||||||
import pkg_resources
|
OCRmyPDF uses setuptools_scm to derive version from git tags.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
try:
|
||||||
|
from importlib.metadata import version as _package_version
|
||||||
|
except ImportError:
|
||||||
|
from importlib_metadata import version as _package_version # type: ignore
|
||||||
|
|
||||||
PROGRAM_NAME = 'ocrmypdf'
|
PROGRAM_NAME = 'ocrmypdf'
|
||||||
|
|
||||||
# Official PEP 396
|
# Official PEP 396
|
||||||
__version__ = pkg_resources.get_distribution('ocrmypdf').version
|
__version__ = _package_version('ocrmypdf')
|
||||||
|
|||||||
+36
-19
@@ -4,6 +4,9 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Functions for using ocrmypdf as an API."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
@@ -12,13 +15,10 @@ import threading
|
|||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
from io import IOBase
|
from io import IOBase
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import AnyStr, BinaryIO, Iterable, Optional, Union
|
from typing import AnyStr, BinaryIO, Iterable, Union
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
from ocrmypdf._logging import ( # pylint: disable=unused-import
|
from ocrmypdf._logging import PageNumberFilter, TqdmConsole
|
||||||
PageNumberFilter,
|
|
||||||
TqdmConsole,
|
|
||||||
)
|
|
||||||
from ocrmypdf._plugin_manager import get_plugin_manager
|
from ocrmypdf._plugin_manager import get_plugin_manager
|
||||||
from ocrmypdf._sync import run_pipeline
|
from ocrmypdf._sync import run_pipeline
|
||||||
from ocrmypdf._validation import check_options
|
from ocrmypdf._validation import check_options
|
||||||
@@ -28,10 +28,13 @@ from ocrmypdf.helpers import is_iterable_notstr
|
|||||||
try:
|
try:
|
||||||
import coloredlogs
|
import coloredlogs
|
||||||
except ModuleNotFoundError:
|
except ModuleNotFoundError:
|
||||||
coloredlogs = None
|
coloredlogs = None # pylint: disable=invalid-name
|
||||||
|
|
||||||
|
if coloredlogs:
|
||||||
|
from humanfriendly.terminal import enable_ansi_support
|
||||||
|
|
||||||
|
|
||||||
StrPath = Union[os.PathLike, AnyStr]
|
StrPath = Union[Path, AnyStr]
|
||||||
PathOrIO = Union[BinaryIO, StrPath]
|
PathOrIO = Union[BinaryIO, StrPath]
|
||||||
|
|
||||||
_api_lock = threading.Lock()
|
_api_lock = threading.Lock()
|
||||||
@@ -40,6 +43,7 @@ _api_lock = threading.Lock()
|
|||||||
class Verbosity(IntEnum):
|
class Verbosity(IntEnum):
|
||||||
"""Verbosity level for configure_logging."""
|
"""Verbosity level for configure_logging."""
|
||||||
|
|
||||||
|
# pylint: disable=invalid-name
|
||||||
quiet = -1 #: Suppress most messages
|
quiet = -1 #: Suppress most messages
|
||||||
default = 0 #: Default level of logging
|
default = 0 #: Default level of logging
|
||||||
debug = 1 #: Output ocrmypdf debug messages
|
debug = 1 #: Output ocrmypdf debug messages
|
||||||
@@ -119,16 +123,15 @@ def configure_logging(
|
|||||||
fmt = '%(pageno)s%(message)s'
|
fmt = '%(pageno)s%(message)s'
|
||||||
|
|
||||||
use_colors = progress_bar_friendly
|
use_colors = progress_bar_friendly
|
||||||
if not coloredlogs:
|
formatter = None
|
||||||
use_colors = False
|
if coloredlogs and use_colors:
|
||||||
if use_colors:
|
use_colors = enable_ansi_support()
|
||||||
if os.name == 'nt':
|
|
||||||
use_colors = coloredlogs.enable_ansi_support()
|
|
||||||
if use_colors:
|
if use_colors:
|
||||||
use_colors = coloredlogs.terminal_supports_colors()
|
use_colors = coloredlogs.terminal_supports_colors()
|
||||||
if use_colors:
|
if use_colors:
|
||||||
formatter = coloredlogs.ColoredFormatter(fmt=fmt)
|
formatter = coloredlogs.ColoredFormatter(fmt=fmt)
|
||||||
else:
|
|
||||||
|
if not formatter:
|
||||||
formatter = logging.Formatter(fmt=fmt)
|
formatter = logging.Formatter(fmt=fmt)
|
||||||
|
|
||||||
console.setFormatter(formatter)
|
console.setFormatter(formatter)
|
||||||
@@ -196,7 +199,7 @@ def create_options(
|
|||||||
else:
|
else:
|
||||||
cmdline.append(os.fspath(output_file))
|
cmdline.append(os.fspath(output_file))
|
||||||
|
|
||||||
parser._api_mode = True
|
parser.enable_api_mode()
|
||||||
options = parser.parse_args(cmdline)
|
options = parser.parse_args(cmdline)
|
||||||
for keyword, val in deferred:
|
for keyword, val in deferred:
|
||||||
setattr(options, keyword, val)
|
setattr(options, keyword, val)
|
||||||
@@ -216,7 +219,7 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
language: Iterable[str] = None,
|
language: Iterable[str] = None,
|
||||||
image_dpi: int = None,
|
image_dpi: int = None,
|
||||||
output_type=None,
|
output_type=None,
|
||||||
sidecar: Optional[StrPath] = None,
|
sidecar: StrPath | None = None,
|
||||||
jobs: int = None,
|
jobs: int = None,
|
||||||
use_threads: bool = None,
|
use_threads: bool = None,
|
||||||
title: str = None,
|
title: str = None,
|
||||||
@@ -231,7 +234,6 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
unpaper_args: str = None,
|
unpaper_args: str = None,
|
||||||
oversample: int = None,
|
oversample: int = None,
|
||||||
remove_vectors: bool = None,
|
remove_vectors: bool = None,
|
||||||
threshold: bool = None,
|
|
||||||
force_ocr: bool = None,
|
force_ocr: bool = None,
|
||||||
skip_text: bool = None,
|
skip_text: bool = None,
|
||||||
redo_ocr: bool = None,
|
redo_ocr: bool = None,
|
||||||
@@ -246,6 +248,7 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
tesseract_config: Iterable[str] = None,
|
tesseract_config: Iterable[str] = None,
|
||||||
tesseract_pagesegmode: int = None,
|
tesseract_pagesegmode: int = None,
|
||||||
tesseract_oem: int = None,
|
tesseract_oem: int = None,
|
||||||
|
tesseract_thresholding: int = None,
|
||||||
pdf_renderer=None,
|
pdf_renderer=None,
|
||||||
tesseract_timeout: float = None,
|
tesseract_timeout: float = None,
|
||||||
rotate_pages_threshold: float = None,
|
rotate_pages_threshold: float = None,
|
||||||
@@ -298,7 +301,7 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
text already, and settings did not tell us to proceed.
|
text already, and settings did not tell us to proceed.
|
||||||
ocrmypdf.InputFileError: Any other problem with the input file.
|
ocrmypdf.InputFileError: Any other problem with the input file.
|
||||||
ocrmypdf.SubprocessOutputError: Any error related to executing a subprocess.
|
ocrmypdf.SubprocessOutputError: Any error related to executing a subprocess.
|
||||||
ocrmypdf.EncryptedPdfERror: If the input PDF is encrypted (password protected).
|
ocrmypdf.EncryptedPdfError: If the input PDF is encrypted (password protected).
|
||||||
OCRmyPDF does not remove passwords.
|
OCRmyPDF does not remove passwords.
|
||||||
ocrmypdf.TesseractConfigError: If Tesseract reported its configuration was not
|
ocrmypdf.TesseractConfigError: If Tesseract reported its configuration was not
|
||||||
valid.
|
valid.
|
||||||
@@ -338,3 +341,17 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
options = create_options(**create_options_kwargs)
|
options = create_options(**create_options_kwargs)
|
||||||
check_options(options, plugin_manager)
|
check_options(options, plugin_manager)
|
||||||
return run_pipeline(options=options, plugin_manager=plugin_manager, api=True)
|
return run_pipeline(options=options, plugin_manager=plugin_manager, api=True)
|
||||||
|
|
||||||
|
|
||||||
|
__all__ = [
|
||||||
|
'PageNumberFilter',
|
||||||
|
'TqdmConsole',
|
||||||
|
'Verbosity',
|
||||||
|
'check_options',
|
||||||
|
'configure_logging',
|
||||||
|
'create_options',
|
||||||
|
'get_parser',
|
||||||
|
'get_plugin_manager',
|
||||||
|
'ocr',
|
||||||
|
'run_pipeline',
|
||||||
|
]
|
||||||
|
|||||||
@@ -4,6 +4,8 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
# This file exists only mark builtin_plugins as a package.
|
# This file exists only mark builtin_plugins as a package.
|
||||||
# The plugin manager will not load it, so anything defined here may not be
|
# The plugin manager will not load it, so anything defined here may not be
|
||||||
# processed as a module.
|
# processed as a module.
|
||||||
|
|||||||
@@ -4,12 +4,15 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""OCRmyPDF's multiprocessing/multithreading abstraction layer."""
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import logging.handlers
|
import logging.handlers
|
||||||
@@ -19,10 +22,9 @@ import queue
|
|||||||
import signal
|
import signal
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
|
from concurrent.futures import ProcessPoolExecutor, ThreadPoolExecutor, as_completed
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from multiprocessing import Pool as ProcessPool
|
from typing import Callable, Iterable, Type, Union
|
||||||
from multiprocessing.pool import ThreadPool
|
|
||||||
from typing import Callable, Iterable, Union
|
|
||||||
|
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
|
|
||||||
@@ -31,7 +33,10 @@ from ocrmypdf._logging import TqdmConsole
|
|||||||
from ocrmypdf.exceptions import InputFileError
|
from ocrmypdf.exceptions import InputFileError
|
||||||
from ocrmypdf.helpers import remove_all_log_handlers
|
from ocrmypdf.helpers import remove_all_log_handlers
|
||||||
|
|
||||||
|
FuturesExecutorClass = Union[Type[ThreadPoolExecutor], Type[ProcessPoolExecutor]]
|
||||||
Queue = Union[multiprocessing.Queue, queue.Queue]
|
Queue = Union[multiprocessing.Queue, queue.Queue]
|
||||||
|
UserInit = Callable[[], None]
|
||||||
|
WorkerInit = Callable[[Queue, UserInit, int], None]
|
||||||
|
|
||||||
|
|
||||||
def log_listener(q: Queue):
|
def log_listener(q: Queue):
|
||||||
@@ -41,7 +46,8 @@ def log_listener(q: Queue):
|
|||||||
should actually write to sys.stderr or whatever we're using, so if this is
|
should actually write to sys.stderr or whatever we're using, so if this is
|
||||||
made into a process the main application needs to be directed to it.
|
made into a process the main application needs to be directed to it.
|
||||||
|
|
||||||
See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes
|
See:
|
||||||
|
https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes
|
||||||
"""
|
"""
|
||||||
|
|
||||||
while True:
|
while True:
|
||||||
@@ -62,7 +68,7 @@ def process_sigbus(*args):
|
|||||||
raise InputFileError("A worker process lost access to an input file")
|
raise InputFileError("A worker process lost access to an input file")
|
||||||
|
|
||||||
|
|
||||||
def process_init(q: Queue, user_init: Callable[[], None], loglevel):
|
def process_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
||||||
"""Initialize a process pool worker"""
|
"""Initialize a process pool worker"""
|
||||||
|
|
||||||
# Ignore SIGINT (our parent process will kill us gracefully)
|
# Ignore SIGINT (our parent process will kill us gracefully)
|
||||||
@@ -85,7 +91,9 @@ def process_init(q: Queue, user_init: Callable[[], None], loglevel):
|
|||||||
return
|
return
|
||||||
|
|
||||||
|
|
||||||
def thread_init(_queue: Queue, user_init: Callable[[], None], _loglevel):
|
def thread_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
||||||
|
del q # unused but required argument
|
||||||
|
del loglevel # unused but required argument
|
||||||
# As a thread, block SIGBUS so the main thread deals with it...
|
# As a thread, block SIGBUS so the main thread deals with it...
|
||||||
with suppress(AttributeError):
|
with suppress(AttributeError):
|
||||||
signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS})
|
signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS})
|
||||||
@@ -95,6 +103,17 @@ def thread_init(_queue: Queue, user_init: Callable[[], None], _loglevel):
|
|||||||
|
|
||||||
|
|
||||||
class StandardExecutor(Executor):
|
class StandardExecutor(Executor):
|
||||||
|
"""Standard OCRmyPDF concurrent task executor."""
|
||||||
|
|
||||||
|
def _cancel_futures_kwargs(self):
|
||||||
|
"""Shim older Pythons that do not have Executor.shutdown(...cancel_futures=).
|
||||||
|
|
||||||
|
Remove this code when support for Python 3.8 is dropped.
|
||||||
|
"""
|
||||||
|
if sys.version_info[:2] < (3, 9):
|
||||||
|
return {}
|
||||||
|
return dict(cancel_futures=True)
|
||||||
|
|
||||||
def _execute(
|
def _execute(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
@@ -107,12 +126,12 @@ class StandardExecutor(Executor):
|
|||||||
task_finished: Callable,
|
task_finished: Callable,
|
||||||
):
|
):
|
||||||
if use_threads:
|
if use_threads:
|
||||||
log_queue = queue.Queue(-1)
|
log_queue: Queue = queue.Queue(-1)
|
||||||
pool_class = ThreadPool
|
executor_class: FuturesExecutorClass = ThreadPoolExecutor
|
||||||
initializer = thread_init
|
initializer: WorkerInit = thread_init
|
||||||
else:
|
else:
|
||||||
log_queue = multiprocessing.Queue(-1)
|
log_queue = multiprocessing.Queue(-1)
|
||||||
pool_class = ProcessPool
|
executor_class = ProcessPoolExecutor
|
||||||
initializer = process_init
|
initializer = process_init
|
||||||
|
|
||||||
# Regardless of whether we use_threads for worker processes, the log_listener
|
# Regardless of whether we use_threads for worker processes, the log_listener
|
||||||
@@ -121,39 +140,35 @@ class StandardExecutor(Executor):
|
|||||||
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
||||||
listener.start()
|
listener.start()
|
||||||
|
|
||||||
with self.pbar_class(**tqdm_kwargs) as pbar:
|
with self.pbar_class(**tqdm_kwargs) as pbar, executor_class(
|
||||||
pool = pool_class(
|
max_workers=max_workers,
|
||||||
processes=max_workers,
|
initializer=initializer,
|
||||||
initializer=initializer,
|
initargs=(log_queue, worker_initializer, logging.getLogger("").level),
|
||||||
initargs=(log_queue, worker_initializer, logging.getLogger("").level),
|
) as executor:
|
||||||
)
|
futures = [executor.submit(task, args) for args in task_arguments]
|
||||||
try:
|
try:
|
||||||
results = pool.imap_unordered(task, task_arguments)
|
for future in as_completed(futures):
|
||||||
for result in results:
|
result = future.result()
|
||||||
if task_finished:
|
task_finished(result, pbar)
|
||||||
task_finished(result, pbar)
|
|
||||||
else:
|
|
||||||
pbar.update()
|
|
||||||
except KeyboardInterrupt:
|
except KeyboardInterrupt:
|
||||||
# Terminate pool so we exit instantly
|
# Terminate pool so we exit instantly
|
||||||
pool.terminate()
|
executor.shutdown(wait=False, **self._cancel_futures_kwargs())
|
||||||
# Don't try listener.join() here, will deadlock
|
|
||||||
raise
|
raise
|
||||||
except Exception:
|
except Exception:
|
||||||
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
||||||
# Unless inside pytest, exit immediately because no one wants
|
# Normally we shutdown without waiting for other child workers
|
||||||
# to wait for child processes to finalize results that will be
|
# on error, because there is no point in waiting for them. Their
|
||||||
# thrown away. Inside pytest, we want child processes to exit
|
# results will be discard. But if the condition above is True,
|
||||||
# cleanly so that they output an error messages or coverage data
|
# then we are running in pytest, and we want everything to exit
|
||||||
# we need from them.
|
# as cleanly as possible so that we get good error messages.
|
||||||
pool.terminate()
|
executor.shutdown(wait=False, **self._cancel_futures_kwargs())
|
||||||
raise
|
raise
|
||||||
finally:
|
finally:
|
||||||
# Terminate log listener
|
# Terminate log listener
|
||||||
log_queue.put_nowait(None)
|
log_queue.put_nowait(None)
|
||||||
pool.close()
|
|
||||||
pool.join()
|
|
||||||
|
|
||||||
|
# When the above succeeds, wait for the listener thread to exit. (If
|
||||||
|
# an exception occurs, we don't try to join, in case it deadlocks.)
|
||||||
listener.join()
|
listener.join()
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -4,6 +4,10 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""OCRmyPDF automatically installs these filters as plugins."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
from ocrmypdf import hookimpl
|
from ocrmypdf import hookimpl
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -5,6 +5,10 @@
|
|||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
|
"""Built-in plugin to implement PDF page rasterization and PDF/A production."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
from ocrmypdf import hookimpl
|
from ocrmypdf import hookimpl
|
||||||
@@ -18,13 +22,13 @@ log = logging.getLogger(__name__)
|
|||||||
|
|
||||||
@hookimpl
|
@hookimpl
|
||||||
def check_options(options):
|
def check_options(options):
|
||||||
gs_version = ghostscript.version()
|
|
||||||
check_external_program(
|
check_external_program(
|
||||||
program='gs',
|
program='gs',
|
||||||
package='ghostscript',
|
package='ghostscript',
|
||||||
version_checker=gs_version,
|
version_checker=ghostscript.version,
|
||||||
need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports
|
need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports
|
||||||
)
|
)
|
||||||
|
gs_version = ghostscript.version()
|
||||||
if gs_version in ('9.24', '9.51'):
|
if gs_version in ('9.24', '9.51'):
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
f"Ghostscript {gs_version} contains serious regressions and is not "
|
f"Ghostscript {gs_version} contains serious regressions and is not "
|
||||||
|
|||||||
@@ -0,0 +1,160 @@
|
|||||||
|
# © 2022 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
|
"""Built-in plugin to implement PDF page optimization."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import logging
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Sequence
|
||||||
|
|
||||||
|
from ocrmypdf import Executor, PdfContext, hookimpl
|
||||||
|
from ocrmypdf._exec import jbig2enc, pngquant
|
||||||
|
from ocrmypdf._pipeline import get_pdf_save_settings
|
||||||
|
from ocrmypdf.cli import numeric
|
||||||
|
from ocrmypdf.optimize import optimize
|
||||||
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def add_options(parser):
|
||||||
|
optimizing = parser.add_argument_group(
|
||||||
|
"Optimization options", "Control how the PDF is optimized after OCR"
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'-O',
|
||||||
|
'--optimize',
|
||||||
|
type=int,
|
||||||
|
choices=range(0, 4),
|
||||||
|
default=1,
|
||||||
|
help=(
|
||||||
|
"Control how PDF is optimized after processing:"
|
||||||
|
"0 - do not optimize; "
|
||||||
|
"1 - do safe, lossless optimizations (default); "
|
||||||
|
"2 - do lossy JPEG and JPEG2000 optimizations; "
|
||||||
|
"3 - do more aggressive lossy JPEG and JPEG2000 optimizations. "
|
||||||
|
"To enable lossy JBIG2, see --jbig2-lossy."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jpeg-quality',
|
||||||
|
type=numeric(int, 0, 100),
|
||||||
|
default=0,
|
||||||
|
metavar='Q',
|
||||||
|
help=(
|
||||||
|
"Adjust JPEG quality level for JPEG optimization. "
|
||||||
|
"100 is best quality and largest output size; "
|
||||||
|
"1 is lowest quality and smallest output; "
|
||||||
|
"0 uses the default."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jpg-quality',
|
||||||
|
type=numeric(int, 0, 100),
|
||||||
|
default=0,
|
||||||
|
metavar='Q',
|
||||||
|
dest='jpeg_quality',
|
||||||
|
help=argparse.SUPPRESS, # Alias for --jpeg-quality
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--png-quality',
|
||||||
|
type=numeric(int, 0, 100),
|
||||||
|
default=0,
|
||||||
|
metavar='Q',
|
||||||
|
help=(
|
||||||
|
"Adjust PNG quality level to use when quantizing PNGs. "
|
||||||
|
"Values have same meaning as with --jpeg-quality"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jbig2-lossy',
|
||||||
|
action='store_true',
|
||||||
|
help=(
|
||||||
|
"Enable JBIG2 lossy mode (better compression, not suitable for some "
|
||||||
|
"use cases - see documentation). Only takes effect if --optimize 1 or "
|
||||||
|
"higher is also enabled."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jbig2-page-group-size',
|
||||||
|
type=numeric(int, 1, 10000),
|
||||||
|
default=0,
|
||||||
|
metavar='N',
|
||||||
|
# Adjust number of pages to consider at once for JBIG2 compression
|
||||||
|
help=argparse.SUPPRESS,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def check_options(options):
|
||||||
|
if options.optimize >= 2:
|
||||||
|
check_external_program(
|
||||||
|
program='pngquant',
|
||||||
|
package='pngquant',
|
||||||
|
version_checker=pngquant.version,
|
||||||
|
need_version='2.0.1',
|
||||||
|
required_for='--optimize {2,3}',
|
||||||
|
)
|
||||||
|
|
||||||
|
if options.optimize >= 2:
|
||||||
|
# Although we use JBIG2 for optimize=1, don't nag about it unless the
|
||||||
|
# user is asking for more optimization
|
||||||
|
check_external_program(
|
||||||
|
program='jbig2',
|
||||||
|
package='jbig2enc',
|
||||||
|
version_checker=jbig2enc.version,
|
||||||
|
need_version='0.28',
|
||||||
|
required_for='--optimize {2,3} | --jbig2-lossy',
|
||||||
|
recommended=True if not options.jbig2_lossy else False,
|
||||||
|
)
|
||||||
|
|
||||||
|
if options.optimize == 0 and any(
|
||||||
|
[options.jbig2_lossy, options.png_quality, options.jpeg_quality]
|
||||||
|
):
|
||||||
|
log.warning(
|
||||||
|
"The arguments --jbig2-lossy, --png-quality, and --jpeg-quality "
|
||||||
|
"will be ignored because --optimize=0."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def optimize_pdf(
|
||||||
|
input_pdf: Path,
|
||||||
|
output_pdf: Path,
|
||||||
|
context: PdfContext,
|
||||||
|
executor: Executor,
|
||||||
|
linearize: bool,
|
||||||
|
) -> tuple[Path, Sequence[str]]:
|
||||||
|
save_settings = dict(
|
||||||
|
linearize=linearize,
|
||||||
|
**get_pdf_save_settings(context.options.output_type),
|
||||||
|
)
|
||||||
|
result_path = optimize(input_pdf, output_pdf, context, save_settings, executor)
|
||||||
|
messages = []
|
||||||
|
if context.options.optimize == 0:
|
||||||
|
messages.append("Optimization was disabled.")
|
||||||
|
else:
|
||||||
|
image_optimizers = {
|
||||||
|
'jbig2': jbig2enc.available(),
|
||||||
|
'pngquant': pngquant.available(),
|
||||||
|
}
|
||||||
|
for name, available in image_optimizers.items():
|
||||||
|
if not available:
|
||||||
|
messages.append(
|
||||||
|
f"The optional dependency '{name}' was not found, so some image "
|
||||||
|
f"optimizations could not be attempted."
|
||||||
|
)
|
||||||
|
return result_path, messages
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def is_optimization_enabled(context: PdfContext) -> bool:
|
||||||
|
return context.options.optimize != 0
|
||||||
@@ -4,14 +4,17 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Built-in plugin to implement OCR using Tesseract."""
|
||||||
|
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
|
|
||||||
from ocrmypdf import hookimpl
|
from ocrmypdf import hookimpl
|
||||||
from ocrmypdf._exec import tesseract
|
from ocrmypdf._exec import tesseract
|
||||||
from ocrmypdf.cli import numeric
|
from ocrmypdf.cli import numeric, str_to_int
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
|
||||||
from ocrmypdf.helpers import clamp
|
from ocrmypdf.helpers import clamp
|
||||||
from ocrmypdf.pluginspec import OcrEngine
|
from ocrmypdf.pluginspec import OcrEngine
|
||||||
from ocrmypdf.subprocess import check_external_program
|
from ocrmypdf.subprocess import check_external_program
|
||||||
@@ -44,13 +47,27 @@ def add_options(parser):
|
|||||||
metavar='MODE',
|
metavar='MODE',
|
||||||
choices=range(0, 4),
|
choices=range(0, 4),
|
||||||
help=(
|
help=(
|
||||||
"Set Tesseract 4.0 OCR engine mode: "
|
"Set Tesseract 4.0+ OCR engine mode: "
|
||||||
"0 - original Tesseract only; "
|
"0 - original Tesseract only; "
|
||||||
"1 - neural nets LSTM only; "
|
"1 - neural nets LSTM only; "
|
||||||
"2 - Tesseract + LSTM; "
|
"2 - Tesseract + LSTM; "
|
||||||
"3 - default."
|
"3 - default."
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--tesseract-thresholding',
|
||||||
|
action='store',
|
||||||
|
type=str_to_int(tesseract.TESSERACT_THRESHOLDING_METHODS),
|
||||||
|
default='auto',
|
||||||
|
metavar='METHOD',
|
||||||
|
help=(
|
||||||
|
"Set Tesseract 5.0+ input image thresholding mode. This may improve OCR "
|
||||||
|
"results on low quality images or those that contain high constrast color. "
|
||||||
|
"legacy-otsu is the Tesseract default; adaptive-otsu is an improved Otsu "
|
||||||
|
"algorithm with improved sort for background color changes; sauvola is "
|
||||||
|
"based on local standard deviation."
|
||||||
|
),
|
||||||
|
)
|
||||||
tess.add_argument(
|
tess.add_argument(
|
||||||
'--tesseract-timeout',
|
'--tesseract-timeout',
|
||||||
default=180.0,
|
default=180.0,
|
||||||
@@ -90,8 +107,14 @@ def check_options(options):
|
|||||||
|
|
||||||
if not tesseract.has_user_words() and (options.user_words or options.user_patterns):
|
if not tesseract.has_user_words() and (options.user_words or options.user_patterns):
|
||||||
log.warning(
|
log.warning(
|
||||||
"Tesseract 4.0 ignores --user-words and --user-patterns, so these "
|
"Tesseract 4.0 (which you have installed) ignores --user-words and "
|
||||||
"arguments have no effect."
|
"--user-patterns, so these arguments have no effect."
|
||||||
|
)
|
||||||
|
if not tesseract.has_thresholding() and options.tesseract_thresholding != 0:
|
||||||
|
log.warning(
|
||||||
|
"The installed version of Tesseract does not support changes to its "
|
||||||
|
"thresholding method. The --tesseract-threshold argument will be "
|
||||||
|
"ignored."
|
||||||
)
|
)
|
||||||
if options.tesseract_pagesegmode in (0, 2):
|
if options.tesseract_pagesegmode in (0, 2):
|
||||||
log.warning(
|
log.warning(
|
||||||
@@ -119,6 +142,8 @@ def validate(pdfinfo, options):
|
|||||||
|
|
||||||
|
|
||||||
class TesseractOcrEngine(OcrEngine):
|
class TesseractOcrEngine(OcrEngine):
|
||||||
|
"""Implements OCR with Tesseract."""
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def version():
|
def version():
|
||||||
return tesseract.version()
|
return tesseract.version()
|
||||||
@@ -143,6 +168,15 @@ class TesseractOcrEngine(OcrEngine):
|
|||||||
timeout=options.tesseract_timeout,
|
timeout=options.tesseract_timeout,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def get_deskew(input_file, options) -> float:
|
||||||
|
return tesseract.get_deskew(
|
||||||
|
input_file,
|
||||||
|
languages=options.languages,
|
||||||
|
engine_mode=options.tesseract_oem,
|
||||||
|
timeout=options.tesseract_timeout,
|
||||||
|
)
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||||
tesseract.generate_hocr(
|
tesseract.generate_hocr(
|
||||||
@@ -154,6 +188,7 @@ class TesseractOcrEngine(OcrEngine):
|
|||||||
tessconfig=options.tesseract_config,
|
tessconfig=options.tesseract_config,
|
||||||
timeout=options.tesseract_timeout,
|
timeout=options.tesseract_timeout,
|
||||||
pagesegmode=options.tesseract_pagesegmode,
|
pagesegmode=options.tesseract_pagesegmode,
|
||||||
|
thresholding=options.tesseract_thresholding,
|
||||||
user_words=options.user_words,
|
user_words=options.user_words,
|
||||||
user_patterns=options.user_patterns,
|
user_patterns=options.user_patterns,
|
||||||
)
|
)
|
||||||
@@ -169,6 +204,7 @@ class TesseractOcrEngine(OcrEngine):
|
|||||||
tessconfig=options.tesseract_config,
|
tessconfig=options.tesseract_config,
|
||||||
timeout=options.tesseract_timeout,
|
timeout=options.tesseract_timeout,
|
||||||
pagesegmode=options.tesseract_pagesegmode,
|
pagesegmode=options.tesseract_pagesegmode,
|
||||||
|
thresholding=options.tesseract_thresholding,
|
||||||
user_words=options.user_words,
|
user_words=options.user_words,
|
||||||
user_patterns=options.user_patterns,
|
user_patterns=options.user_patterns,
|
||||||
)
|
)
|
||||||
|
|||||||
+47
-84
@@ -4,42 +4,68 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Command line interface customization and validation."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
from typing import Optional, Type, TypeVar
|
from typing import Any, Callable, Mapping, TypeVar
|
||||||
|
|
||||||
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
||||||
from ocrmypdf._version import __version__ as _VERSION
|
from ocrmypdf._version import __version__ as _VERSION
|
||||||
|
|
||||||
T = TypeVar('T')
|
T = TypeVar('T', int, float)
|
||||||
|
|
||||||
|
|
||||||
def numeric(basetype: Type[T], min_: Optional[T] = None, max_: Optional[T] = None):
|
def numeric(basetype: Callable[[Any], T], min_: T | None = None, max_: T | None = None):
|
||||||
"""Validator for numeric params"""
|
"""Validator for numeric params"""
|
||||||
min_ = basetype(min_) if min_ is not None else None
|
min_ = basetype(min_) if min_ is not None else None
|
||||||
max_ = basetype(max_) if max_ is not None else None
|
max_ = basetype(max_) if max_ is not None else None
|
||||||
|
|
||||||
def _numeric(string):
|
def _numeric(s: str) -> T:
|
||||||
value = basetype(string)
|
value = basetype(s)
|
||||||
if (min_ is not None and value < min_) or (max_ is not None and value > max_):
|
if (min_ is not None and value < min_) or (max_ is not None and value > max_):
|
||||||
msg = "%r not in valid range %r" % (string, (min_, max_))
|
raise argparse.ArgumentTypeError(
|
||||||
raise argparse.ArgumentTypeError(msg)
|
f"{s!r} not in valid range {(min_, max_)!r}"
|
||||||
|
)
|
||||||
return value
|
return value
|
||||||
|
|
||||||
_numeric.__name__ = basetype.__name__
|
_numeric.__name__ = basetype.__name__
|
||||||
return _numeric
|
return _numeric
|
||||||
|
|
||||||
|
|
||||||
|
def str_to_int(mapping: Mapping[str, int]):
|
||||||
|
"""Accept text on command line and convert to integer."""
|
||||||
|
|
||||||
|
def _str_to_int(s: str) -> int:
|
||||||
|
try:
|
||||||
|
return mapping[s]
|
||||||
|
except KeyError:
|
||||||
|
raise argparse.ArgumentTypeError(
|
||||||
|
f"{s!r} must be one of: {', '.join(mapping.keys())}"
|
||||||
|
) from None
|
||||||
|
|
||||||
|
return _str_to_int
|
||||||
|
|
||||||
|
|
||||||
class ArgumentParser(argparse.ArgumentParser):
|
class ArgumentParser(argparse.ArgumentParser):
|
||||||
"""Override parser's default behavior of calling sys.exit()
|
"""Override parser's default behavior of calling sys.exit()
|
||||||
|
|
||||||
https://stackoverflow.com/questions/5943249/python-argparse-and-controlling-overriding-the-exit-status-code
|
https://stackoverflow.com/questions/5943249/python-argparse-and-controlling-overriding-the-exit-status-code
|
||||||
|
|
||||||
|
OCRmyPDF began as a CLI but eventually acquired an API. The API works inside out,
|
||||||
|
by synthesizing a command line argument. So we subclass the standard parser with
|
||||||
|
one that doesn't call sys.exit(). Obviously this is not the ideal way to do things
|
||||||
|
but it works for us.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, *args, **kwargs):
|
def __init__(self, *args, **kwargs):
|
||||||
super().__init__(*args, **kwargs)
|
super().__init__(*args, **kwargs)
|
||||||
self._api_mode = False
|
self._api_mode = False
|
||||||
|
|
||||||
|
def enable_api_mode(self):
|
||||||
|
self._api_mode = True
|
||||||
|
|
||||||
def error(self, message):
|
def error(self, message):
|
||||||
if not self._api_mode:
|
if not self._api_mode:
|
||||||
super().error(message)
|
super().error(message)
|
||||||
@@ -48,6 +74,8 @@ class ArgumentParser(argparse.ArgumentParser):
|
|||||||
|
|
||||||
|
|
||||||
class LanguageSetAction(argparse.Action):
|
class LanguageSetAction(argparse.Action):
|
||||||
|
"""Manages a list of languages."""
|
||||||
|
|
||||||
def __init__(self, option_strings, dest, default=None, **kwargs):
|
def __init__(self, option_strings, dest, default=None, **kwargs):
|
||||||
if default is None:
|
if default is None:
|
||||||
default = set()
|
default = set()
|
||||||
@@ -125,7 +153,7 @@ Online documentation is located at:
|
|||||||
'output_file',
|
'output_file',
|
||||||
metavar="output_pdf",
|
metavar="output_pdf",
|
||||||
help="Output searchable PDF file (or '-' to write to standard output). "
|
help="Output searchable PDF file (or '-' to write to standard output). "
|
||||||
"Existing files will be ovewritten. If same as input file, the "
|
"Existing files will be overwritten. If same as input file, the "
|
||||||
"input file will be updated only if processing is successful.",
|
"input file will be updated only if processing is successful.",
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
@@ -145,7 +173,7 @@ Online documentation is located at:
|
|||||||
)
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
'--output-type',
|
'--output-type',
|
||||||
choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3'],
|
choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3', 'none'],
|
||||||
default='pdfa',
|
default='pdfa',
|
||||||
help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for "
|
help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for "
|
||||||
"long term archiving (default, recommended) but may not suitable "
|
"long term archiving (default, recommended) but may not suitable "
|
||||||
@@ -153,7 +181,8 @@ Online documentation is located at:
|
|||||||
"also has problems with full Unicode text. 'pdf' attempts to "
|
"also has problems with full Unicode text. 'pdf' attempts to "
|
||||||
"preserve file contents as much as possible. 'pdf-a1' creates a "
|
"preserve file contents as much as possible. 'pdf-a1' creates a "
|
||||||
"PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a "
|
"PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a "
|
||||||
"PDF/A3-b file.",
|
"PDF/A3-b file. 'none' will produce no output, which may be helpful if "
|
||||||
|
"only the --sidecar is desired.",
|
||||||
)
|
)
|
||||||
|
|
||||||
# Use null string '\0' as sentinel to indicate the user supplied no argument,
|
# Use null string '\0' as sentinel to indicate the user supplied no argument,
|
||||||
@@ -210,7 +239,13 @@ Online documentation is located at:
|
|||||||
help=argparse.SUPPRESS,
|
help=argparse.SUPPRESS,
|
||||||
)
|
)
|
||||||
jobcontrol.add_argument(
|
jobcontrol.add_argument(
|
||||||
'--use-threads', action='store_true', help=argparse.SUPPRESS
|
'--use-threads', action='store_true', default=True, help=argparse.SUPPRESS
|
||||||
|
)
|
||||||
|
jobcontrol.add_argument(
|
||||||
|
'--no-use-threads',
|
||||||
|
action='store_false',
|
||||||
|
dest='use_threads',
|
||||||
|
help=argparse.SUPPRESS,
|
||||||
)
|
)
|
||||||
|
|
||||||
metadata = parser.add_argument_group(
|
metadata = parser.add_argument_group(
|
||||||
@@ -284,14 +319,6 @@ Online documentation is located at:
|
|||||||
help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they "
|
help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they "
|
||||||
"will not be included in OCR. This can eliminate false characters.",
|
"will not be included in OCR. This can eliminate false characters.",
|
||||||
)
|
)
|
||||||
preprocessing.add_argument(
|
|
||||||
'--threshold',
|
|
||||||
action='store_true',
|
|
||||||
help=(
|
|
||||||
"EXPERIMENTAL. Threshold image to 1bpp before sending it to Tesseract "
|
|
||||||
"for OCR. Can improve OCR quality compared to Tesseract's thresholder."
|
|
||||||
),
|
|
||||||
)
|
|
||||||
|
|
||||||
ocrsettings = parser.add_argument_group("OCR options", "Control how OCR is applied")
|
ocrsettings = parser.add_argument_group("OCR options", "Control how OCR is applied")
|
||||||
ocrsettings.add_argument(
|
ocrsettings.add_argument(
|
||||||
@@ -325,70 +352,6 @@ Online documentation is located at:
|
|||||||
"but include skipped pages in final output",
|
"but include skipped pages in final output",
|
||||||
)
|
)
|
||||||
|
|
||||||
optimizing = parser.add_argument_group(
|
|
||||||
"Optimization options", "Control how the PDF is optimized after OCR"
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'-O',
|
|
||||||
'--optimize',
|
|
||||||
type=int,
|
|
||||||
choices=range(0, 4),
|
|
||||||
default=1,
|
|
||||||
help=(
|
|
||||||
"Control how PDF is optimized after processing:"
|
|
||||||
"0 - do not optimize; "
|
|
||||||
"1 - do safe, lossless optimizations (default); "
|
|
||||||
"2 - do some lossy optimizations; "
|
|
||||||
"3 - do aggressive lossy optimizations (including lossy JBIG2)"
|
|
||||||
),
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--jpeg-quality',
|
|
||||||
type=numeric(int, 0, 100),
|
|
||||||
default=0,
|
|
||||||
metavar='Q',
|
|
||||||
help=(
|
|
||||||
"Adjust JPEG quality level for JPEG optimization. "
|
|
||||||
"100 is best quality and largest output size; "
|
|
||||||
"1 is lowest quality and smallest output; "
|
|
||||||
"0 uses the default."
|
|
||||||
),
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--jpg-quality',
|
|
||||||
type=numeric(int, 0, 100),
|
|
||||||
default=0,
|
|
||||||
metavar='Q',
|
|
||||||
dest='jpeg_quality',
|
|
||||||
help=argparse.SUPPRESS, # Alias for --jpeg-quality
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--png-quality',
|
|
||||||
type=numeric(int, 0, 100),
|
|
||||||
default=0,
|
|
||||||
metavar='Q',
|
|
||||||
help=(
|
|
||||||
"Adjust PNG quality level to use when quantizing PNGs. "
|
|
||||||
"Values have same meaning as with --jpeg-quality"
|
|
||||||
),
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--jbig2-lossy',
|
|
||||||
action='store_true',
|
|
||||||
help=(
|
|
||||||
"Enable JBIG2 lossy mode (better compression, not suitable for some "
|
|
||||||
"use cases - see documentation)."
|
|
||||||
),
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--jbig2-page-group-size',
|
|
||||||
type=numeric(int, 1, 10000),
|
|
||||||
default=0,
|
|
||||||
metavar='N',
|
|
||||||
# Adjust number of pages to consider at once for JBIG2 compression
|
|
||||||
help=argparse.SUPPRESS,
|
|
||||||
)
|
|
||||||
|
|
||||||
advanced = parser.add_argument_group(
|
advanced = parser.add_argument_group(
|
||||||
"Advanced", "Advanced options to control OCRmyPDF"
|
"Advanced", "Advanced options to control OCRmyPDF"
|
||||||
)
|
)
|
||||||
@@ -407,7 +370,7 @@ Online documentation is located at:
|
|||||||
metavar='MPixels',
|
metavar='MPixels',
|
||||||
help="Set maximum number of pixels to unpack before treating an image as a "
|
help="Set maximum number of pixels to unpack before treating an image as a "
|
||||||
"decompression bomb",
|
"decompression bomb",
|
||||||
default=128.0,
|
default=250.0,
|
||||||
)
|
)
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--pdf-renderer',
|
'--pdf-renderer',
|
||||||
|
|||||||
@@ -1,8 +1,10 @@
|
|||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
# © 2021 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
"""Bindings to external libraries"""
|
"""Data files used to generate certain PDFs."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
@@ -4,12 +4,18 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""OCRmyPDF's exceptions."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
from textwrap import dedent
|
from textwrap import dedent
|
||||||
|
|
||||||
|
|
||||||
class ExitCode(IntEnum):
|
class ExitCode(IntEnum):
|
||||||
|
"""OCRmyPDF's exit codes."""
|
||||||
|
|
||||||
|
# pylint: disable=invalid-name
|
||||||
ok = 0
|
ok = 0
|
||||||
bad_args = 1
|
bad_args = 1
|
||||||
input_file = 2
|
input_file = 2
|
||||||
@@ -26,6 +32,8 @@ class ExitCode(IntEnum):
|
|||||||
|
|
||||||
|
|
||||||
class ExitCodeException(Exception):
|
class ExitCodeException(Exception):
|
||||||
|
"""An exception which should return an exit code with sys.exit()."""
|
||||||
|
|
||||||
exit_code = ExitCode.other_error
|
exit_code = ExitCode.other_error
|
||||||
message = ""
|
message = ""
|
||||||
|
|
||||||
@@ -37,17 +45,24 @@ class ExitCodeException(Exception):
|
|||||||
|
|
||||||
|
|
||||||
class BadArgsError(ExitCodeException):
|
class BadArgsError(ExitCodeException):
|
||||||
|
"""Invalid arguments on the command line or API."""
|
||||||
|
|
||||||
exit_code = ExitCode.bad_args
|
exit_code = ExitCode.bad_args
|
||||||
|
|
||||||
|
|
||||||
class PdfMergeFailedError(ExitCodeException):
|
class PdfMergeFailedError(ExitCodeException): # deprecated
|
||||||
|
"""An intermediate PDF can't be merged.
|
||||||
|
|
||||||
|
No longer in use.
|
||||||
|
"""
|
||||||
|
|
||||||
exit_code = ExitCode.input_file
|
exit_code = ExitCode.input_file
|
||||||
message = dedent(
|
message = dedent(
|
||||||
'''\
|
'''\
|
||||||
Failed to merge PDF image layer with OCR layer
|
Failed to merge PDF image layer with OCR layer
|
||||||
|
|
||||||
Usually this happens because the input PDF file is malformed and
|
Usually this happens because the input PDF file is malformed and
|
||||||
ocrmypdf cannot automatically correct the problem on its own.
|
ocrmypdf cannot correct the problem on its own.
|
||||||
|
|
||||||
Try using
|
Try using
|
||||||
ocrmypdf --pdf-renderer sandwich [..other args..]
|
ocrmypdf --pdf-renderer sandwich [..other args..]
|
||||||
@@ -56,34 +71,50 @@ class PdfMergeFailedError(ExitCodeException):
|
|||||||
|
|
||||||
|
|
||||||
class MissingDependencyError(ExitCodeException):
|
class MissingDependencyError(ExitCodeException):
|
||||||
|
"""A third-party dependency is missing."""
|
||||||
|
|
||||||
exit_code = ExitCode.missing_dependency
|
exit_code = ExitCode.missing_dependency
|
||||||
|
|
||||||
|
|
||||||
class UnsupportedImageFormatError(ExitCodeException):
|
class UnsupportedImageFormatError(ExitCodeException):
|
||||||
|
"""The image format is not supported."""
|
||||||
|
|
||||||
exit_code = ExitCode.input_file
|
exit_code = ExitCode.input_file
|
||||||
|
|
||||||
|
|
||||||
class DpiError(ExitCodeException):
|
class DpiError(ExitCodeException):
|
||||||
|
"""Missing information about input image DPI."""
|
||||||
|
|
||||||
exit_code = ExitCode.input_file
|
exit_code = ExitCode.input_file
|
||||||
|
|
||||||
|
|
||||||
class OutputFileAccessError(ExitCodeException):
|
class OutputFileAccessError(ExitCodeException):
|
||||||
|
"""Cannot access the intended output file path."""
|
||||||
|
|
||||||
exit_code = ExitCode.file_access_error
|
exit_code = ExitCode.file_access_error
|
||||||
|
|
||||||
|
|
||||||
class PriorOcrFoundError(ExitCodeException):
|
class PriorOcrFoundError(ExitCodeException):
|
||||||
|
"""This file already has OCR."""
|
||||||
|
|
||||||
exit_code = ExitCode.already_done_ocr
|
exit_code = ExitCode.already_done_ocr
|
||||||
|
|
||||||
|
|
||||||
class InputFileError(ExitCodeException):
|
class InputFileError(ExitCodeException):
|
||||||
|
"""Something is wrong with the input file."""
|
||||||
|
|
||||||
exit_code = ExitCode.input_file
|
exit_code = ExitCode.input_file
|
||||||
|
|
||||||
|
|
||||||
class SubprocessOutputError(ExitCodeException):
|
class SubprocessOutputError(ExitCodeException):
|
||||||
|
"""A subprocess returned an unexpected error."""
|
||||||
|
|
||||||
exit_code = ExitCode.child_process_error
|
exit_code = ExitCode.child_process_error
|
||||||
|
|
||||||
|
|
||||||
class EncryptedPdfError(ExitCodeException):
|
class EncryptedPdfError(ExitCodeException):
|
||||||
|
"""Input PDF is encrypted."""
|
||||||
|
|
||||||
exit_code = ExitCode.encrypted_pdf
|
exit_code = ExitCode.encrypted_pdf
|
||||||
message = dedent(
|
message = dedent(
|
||||||
'''\
|
'''\
|
||||||
@@ -100,5 +131,7 @@ class EncryptedPdfError(ExitCodeException):
|
|||||||
|
|
||||||
|
|
||||||
class TesseractConfigError(ExitCodeException):
|
class TesseractConfigError(ExitCodeException):
|
||||||
|
"""Tesseract config can't be parsed."""
|
||||||
|
|
||||||
exit_code = ExitCode.invalid_config
|
exit_code = ExitCode.invalid_config
|
||||||
message = "Error occurred while parsing a Tesseract configuration file"
|
message = "Error occurred while parsing a Tesseract configuration file"
|
||||||
|
|||||||
@@ -20,6 +20,8 @@ be guaranteed, some workers may end up with too much work while others are idle.
|
|||||||
It is less efficient than the standard implementation, so not th edefault.
|
It is less efficient than the standard implementation, so not th edefault.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import logging.handlers
|
import logging.handlers
|
||||||
import signal
|
import signal
|
||||||
@@ -37,9 +39,11 @@ from ocrmypdf.helpers import remove_all_log_handlers
|
|||||||
|
|
||||||
|
|
||||||
class MessageType(Enum):
|
class MessageType(Enum):
|
||||||
exception = auto()
|
"""Implement basic IPC messaging."""
|
||||||
result = auto()
|
|
||||||
complete = auto()
|
exception = auto() # pylint: disable=invalid-name
|
||||||
|
result = auto() # pylint: disable=invalid-name
|
||||||
|
complete = auto() # pylint: disable=invalid-name
|
||||||
|
|
||||||
|
|
||||||
def split_every(n: int, iterable: Iterable) -> Iterator:
|
def split_every(n: int, iterable: Iterable) -> Iterator:
|
||||||
@@ -59,8 +63,12 @@ def process_sigbus(*args):
|
|||||||
|
|
||||||
|
|
||||||
class ConnectionLogHandler(logging.handlers.QueueHandler):
|
class ConnectionLogHandler(logging.handlers.QueueHandler):
|
||||||
|
"""Handler used by child processes to forward log messages to parent."""
|
||||||
|
|
||||||
def __init__(self, conn: Connection) -> None:
|
def __init__(self, conn: Connection) -> None:
|
||||||
super().__init__(None)
|
# sets the parent's queue to None - parent only touches queue
|
||||||
|
# in enqueue() which we override
|
||||||
|
super().__init__(None) # type: ignore
|
||||||
self.conn = conn
|
self.conn = conn
|
||||||
|
|
||||||
def enqueue(self, record):
|
def enqueue(self, record):
|
||||||
@@ -89,7 +97,7 @@ def process_loop(
|
|||||||
for args in task_args:
|
for args in task_args:
|
||||||
try:
|
try:
|
||||||
result = task(args)
|
result = task(args)
|
||||||
except Exception as e:
|
except Exception as e: # pylint: disable=broad-except
|
||||||
conn.send((MessageType.exception, e))
|
conn.send((MessageType.exception, e))
|
||||||
break
|
break
|
||||||
else:
|
else:
|
||||||
@@ -101,6 +109,8 @@ def process_loop(
|
|||||||
|
|
||||||
|
|
||||||
class LambdaExecutor(Executor):
|
class LambdaExecutor(Executor):
|
||||||
|
"""Executor for AWS Lambda or similar environments that lack semaphores."""
|
||||||
|
|
||||||
def _execute(
|
def _execute(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
@@ -126,8 +136,8 @@ class LambdaExecutor(Executor):
|
|||||||
if not grouped_args:
|
if not grouped_args:
|
||||||
return
|
return
|
||||||
|
|
||||||
processes = []
|
processes: list[Process] = []
|
||||||
connections = []
|
connections: list[Connection] = []
|
||||||
for chunk in grouped_args:
|
for chunk in grouped_args:
|
||||||
parent_conn, child_conn = Pipe()
|
parent_conn, child_conn = Pipe()
|
||||||
|
|
||||||
@@ -151,11 +161,13 @@ class LambdaExecutor(Executor):
|
|||||||
|
|
||||||
with self.pbar_class(**tqdm_kwargs) as pbar:
|
with self.pbar_class(**tqdm_kwargs) as pbar:
|
||||||
while connections:
|
while connections:
|
||||||
for r in wait(connections):
|
for result in wait(connections):
|
||||||
|
if not isinstance(result, Connection):
|
||||||
|
raise NotImplementedError("We only support Connection()")
|
||||||
try:
|
try:
|
||||||
msg_type, msg = r.recv()
|
msg_type, msg = result.recv()
|
||||||
except EOFError:
|
except EOFError:
|
||||||
connections.remove(r)
|
connections.remove(result)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
if msg_type == MessageType.result:
|
if msg_type == MessageType.result:
|
||||||
@@ -166,7 +178,7 @@ class LambdaExecutor(Executor):
|
|||||||
logger = logging.getLogger(record.name)
|
logger = logging.getLogger(record.name)
|
||||||
logger.handle(record)
|
logger.handle(record)
|
||||||
elif msg_type == MessageType.complete:
|
elif msg_type == MessageType.complete:
|
||||||
connections.remove(r)
|
connections.remove(result)
|
||||||
elif msg_type == MessageType.exception:
|
elif msg_type == MessageType.exception:
|
||||||
for process in processes:
|
for process in processes:
|
||||||
process.terminate()
|
process.terminate()
|
||||||
|
|||||||
+41
-27
@@ -4,6 +4,9 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Support functions."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import multiprocessing
|
import multiprocessing
|
||||||
@@ -19,10 +22,21 @@ from math import isclose, isfinite
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Sequence
|
from typing import Any, Sequence
|
||||||
|
|
||||||
|
import img2pdf
|
||||||
import pikepdf
|
import pikepdf
|
||||||
|
from packaging.version import Version
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
if Version(img2pdf.__version__) < Version('0.4.0'):
|
||||||
|
IMG2PDF_KWARGS = dict(without_pdfw=True)
|
||||||
|
elif Version(img2pdf.__version__) < Version('0.4.3'):
|
||||||
|
IMG2PDF_KWARGS = dict(engine=img2pdf.Engine.pikepdf)
|
||||||
|
else:
|
||||||
|
IMG2PDF_KWARGS = dict(
|
||||||
|
engine=img2pdf.Engine.pikepdf, rotation=img2pdf.Rotation.ifvalid
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class Resolution(namedtuple('Resolution', ('x', 'y'))):
|
class Resolution(namedtuple('Resolution', ('x', 'y'))):
|
||||||
"""The number of pixels per inch in each 2D direction.
|
"""The number of pixels per inch in each 2D direction.
|
||||||
@@ -126,11 +140,11 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
|||||||
os.symlink(os.path.abspath(input_file), soft_link_name)
|
os.symlink(os.path.abspath(input_file), soft_link_name)
|
||||||
|
|
||||||
|
|
||||||
def samefile(f1: os.PathLike, f2: os.PathLike):
|
def samefile(file1: os.PathLike, file2: os.PathLike):
|
||||||
if os.name == 'nt':
|
if os.name == 'nt':
|
||||||
return f1 == f2
|
return file1 == file2
|
||||||
else:
|
else:
|
||||||
return os.path.samefile(f1, f2)
|
return os.path.samefile(file1, file2)
|
||||||
|
|
||||||
|
|
||||||
def is_iterable_notstr(thing: Any) -> bool:
|
def is_iterable_notstr(thing: Any) -> bool:
|
||||||
@@ -138,9 +152,9 @@ def is_iterable_notstr(thing: Any) -> bool:
|
|||||||
return isinstance(thing, Iterable) and not isinstance(thing, str)
|
return isinstance(thing, Iterable) and not isinstance(thing, str)
|
||||||
|
|
||||||
|
|
||||||
def monotonic(L: Sequence) -> bool:
|
def monotonic(seq: Sequence) -> bool:
|
||||||
"""Does this sequence increase monotonically?"""
|
"""Does this sequence increase monotonically?"""
|
||||||
return all(b > a for a, b in zip(L, L[1:]))
|
return all(b > a for a, b in zip(seq, seq[1:]))
|
||||||
|
|
||||||
|
|
||||||
def page_number(input_file: os.PathLike) -> int:
|
def page_number(input_file: os.PathLike) -> int:
|
||||||
@@ -155,7 +169,7 @@ def available_cpu_count() -> int:
|
|||||||
except NotImplementedError:
|
except NotImplementedError:
|
||||||
pass
|
pass
|
||||||
warnings.warn(
|
warnings.warn(
|
||||||
"Could not get CPU count. Assuming one (1) CPU." "Use -j N to set manually."
|
"Could not get CPU count. Assuming one (1) CPU. Use -j N to set manually."
|
||||||
)
|
)
|
||||||
return 1
|
return 1
|
||||||
|
|
||||||
@@ -179,17 +193,17 @@ def is_file_writable(test_file: os.PathLike) -> bool:
|
|||||||
os.W_OK,
|
os.W_OK,
|
||||||
effective_ids=(os.access in os.supports_effective_ids),
|
effective_ids=(os.access in os.supports_effective_ids),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
fp = p.open('wb')
|
||||||
|
except OSError:
|
||||||
|
return False
|
||||||
else:
|
else:
|
||||||
try:
|
fp.close()
|
||||||
fp = p.open('wb')
|
with suppress(OSError):
|
||||||
except OSError:
|
p.unlink()
|
||||||
return False
|
return True
|
||||||
else:
|
except (OSError, RuntimeError) as e:
|
||||||
fp.close()
|
|
||||||
with suppress(OSError):
|
|
||||||
p.unlink()
|
|
||||||
return True
|
|
||||||
except (EnvironmentError, RuntimeError) as e:
|
|
||||||
log.debug(e)
|
log.debug(e)
|
||||||
log.error(str(e))
|
log.error(str(e))
|
||||||
return False
|
return False
|
||||||
@@ -209,11 +223,19 @@ def check_pdf(input_file: Path) -> bool:
|
|||||||
else:
|
else:
|
||||||
with pdf:
|
with pdf:
|
||||||
messages = pdf.check()
|
messages = pdf.check()
|
||||||
|
success = True
|
||||||
for msg in messages:
|
for msg in messages:
|
||||||
if 'error' in msg.lower():
|
if 'error' in msg.lower():
|
||||||
log.error(msg)
|
log.error(msg)
|
||||||
|
success = False
|
||||||
|
elif (
|
||||||
|
"/DecodeParms: operation for dictionary attempted on object "
|
||||||
|
"of type null" in msg
|
||||||
|
):
|
||||||
|
pass # Ignore/spurious warning
|
||||||
else:
|
else:
|
||||||
log.warning(msg)
|
log.warning(msg)
|
||||||
|
success = False
|
||||||
|
|
||||||
sio = StringIO()
|
sio = StringIO()
|
||||||
linearize_msgs = ''
|
linearize_msgs = ''
|
||||||
@@ -221,22 +243,14 @@ def check_pdf(input_file: Path) -> bool:
|
|||||||
# If linearization is missing entirely, we do not complain. We do
|
# If linearization is missing entirely, we do not complain. We do
|
||||||
# complain if linearization is present but incorrect.
|
# complain if linearization is present but incorrect.
|
||||||
pdf.check_linearization(sio)
|
pdf.check_linearization(sio)
|
||||||
except RuntimeError:
|
except (RuntimeError, pikepdf.ForeignObjectError):
|
||||||
pass
|
|
||||||
except (
|
|
||||||
# Workaround for a problematic pikepdf version
|
|
||||||
# pragma: no cover
|
|
||||||
getattr(pikepdf, 'ForeignObjectError')
|
|
||||||
if pikepdf.__version__ == '2.1.0'
|
|
||||||
else NeverRaise
|
|
||||||
):
|
|
||||||
pass
|
pass
|
||||||
else:
|
else:
|
||||||
linearize_msgs = sio.getvalue()
|
linearize_msgs = sio.getvalue()
|
||||||
if linearize_msgs:
|
if linearize_msgs:
|
||||||
log.warning(linearize_msgs)
|
log.warning(linearize_msgs)
|
||||||
|
|
||||||
if not messages and not linearize_msgs:
|
if success and not linearize_msgs:
|
||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
@@ -273,7 +287,7 @@ def deprecated(func):
|
|||||||
def new_func(*args, **kwargs):
|
def new_func(*args, **kwargs):
|
||||||
warnings.simplefilter('always', DeprecationWarning) # turn off filter
|
warnings.simplefilter('always', DeprecationWarning) # turn off filter
|
||||||
warnings.warn(
|
warnings.warn(
|
||||||
"Call to deprecated function {}.".format(func.__name__),
|
f"Call to deprecated function {func.__name__}.",
|
||||||
category=DeprecationWarning,
|
category=DeprecationWarning,
|
||||||
stacklevel=2,
|
stacklevel=2,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -28,17 +28,26 @@
|
|||||||
# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||||
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
|
"""Transform .hocr and page image to text PDF."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
|
import warnings
|
||||||
from math import atan, cos, sin
|
from math import atan, cos, sin
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, NamedTuple, Optional, Tuple, Union
|
from typing import Any, NamedTuple, Optional, Tuple, Union
|
||||||
from xml.etree import ElementTree
|
from xml.etree import ElementTree
|
||||||
|
|
||||||
from reportlab.lib.colors import black, cyan, magenta, red
|
with warnings.catch_warnings():
|
||||||
from reportlab.lib.units import inch
|
warnings.filterwarnings(
|
||||||
from reportlab.pdfgen.canvas import Canvas
|
'ignore', category=DeprecationWarning, message=r".*load_module.*"
|
||||||
|
)
|
||||||
|
from reportlab.lib.colors import black, cyan, magenta, red
|
||||||
|
from reportlab.lib.units import inch
|
||||||
|
from reportlab.pdfgen.canvas import Canvas
|
||||||
|
|
||||||
# According to Wikipedia these languages are supported in the ISO-8859-1 character
|
# According to Wikipedia these languages are supported in the ISO-8859-1 character
|
||||||
# set, meaning reportlab can generate them and they are compatible with hocr,
|
# set, meaning reportlab can generate them and they are compatible with hocr,
|
||||||
@@ -99,7 +108,7 @@ HOCR_OK_LANGS = frozenset(
|
|||||||
Element = ElementTree.Element
|
Element = ElementTree.Element
|
||||||
|
|
||||||
|
|
||||||
class Rect(NamedTuple): # pylint: disable=inherit-non-class
|
class Rect(NamedTuple):
|
||||||
"""A rectangle for managing PDF coordinates."""
|
"""A rectangle for managing PDF coordinates."""
|
||||||
|
|
||||||
x1: Any
|
x1: Any
|
||||||
@@ -109,7 +118,7 @@ class Rect(NamedTuple): # pylint: disable=inherit-non-class
|
|||||||
|
|
||||||
|
|
||||||
class HocrTransformError(Exception):
|
class HocrTransformError(Exception):
|
||||||
pass
|
"""Error while applying hOCR transform."""
|
||||||
|
|
||||||
|
|
||||||
class HocrTransform:
|
class HocrTransform:
|
||||||
@@ -132,7 +141,7 @@ class HocrTransform:
|
|||||||
{'ff': 'ff', 'ffi': 'ffi', 'ffl': 'ffl', 'fi': 'fi', 'fl': 'fl'}
|
{'ff': 'ff', 'ffi': 'ffi', 'ffl': 'ffl', 'fi': 'fi', 'fl': 'fl'}
|
||||||
)
|
)
|
||||||
|
|
||||||
def __init__(self, *, hocr_filename: Union[str, Path], dpi: float):
|
def __init__(self, *, hocr_filename: str | Path, dpi: float):
|
||||||
self.dpi = dpi
|
self.dpi = dpi
|
||||||
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
||||||
|
|
||||||
@@ -196,7 +205,7 @@ class HocrTransform:
|
|||||||
return out
|
return out
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def baseline(cls, element: Element) -> Tuple[float, float]:
|
def baseline(cls, element: Element) -> tuple[float, float]:
|
||||||
"""
|
"""
|
||||||
Returns a tuple containing the baseline slope and intercept.
|
Returns a tuple containing the baseline slope and intercept.
|
||||||
"""
|
"""
|
||||||
@@ -212,7 +221,7 @@ class HocrTransform:
|
|||||||
"""
|
"""
|
||||||
return Rect._make((c / self.dpi * inch) for c in pxl)
|
return Rect._make((c / self.dpi * inch) for c in pxl)
|
||||||
|
|
||||||
def _child_xpath(self, html_tag: str, html_class: Optional[str] = None) -> str:
|
def _child_xpath(self, html_tag: str, html_class: str | None = None) -> str:
|
||||||
xpath = f".//{self.xmlns}{html_tag}"
|
xpath = f".//{self.xmlns}{html_tag}"
|
||||||
if html_class:
|
if html_class:
|
||||||
xpath += f"[@class='{html_class}']"
|
xpath += f"[@class='{html_class}']"
|
||||||
@@ -239,7 +248,7 @@ class HocrTransform:
|
|||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
out_filename: Path,
|
out_filename: Path,
|
||||||
image_filename: Optional[Path] = None,
|
image_filename: Path | None = None,
|
||||||
show_bounding_boxes: bool = False,
|
show_bounding_boxes: bool = False,
|
||||||
fontname: str = "Helvetica",
|
fontname: str = "Helvetica",
|
||||||
invisible_text: bool = False,
|
invisible_text: bool = False,
|
||||||
@@ -287,7 +296,7 @@ class HocrTransform:
|
|||||||
continue
|
continue
|
||||||
|
|
||||||
pxl_coords = self.element_coordinates(elem)
|
pxl_coords = self.element_coordinates(elem)
|
||||||
pt = self.pt_from_pixel(pxl_coords)
|
pt = self.pt_from_pixel(pxl_coords) # pylint: disable=invalid-name
|
||||||
|
|
||||||
# draw the bbox border
|
# draw the bbox border
|
||||||
if show_bounding_boxes: # pragma: no cover
|
if show_bounding_boxes: # pragma: no cover
|
||||||
@@ -342,14 +351,14 @@ class HocrTransform:
|
|||||||
def _do_line(
|
def _do_line(
|
||||||
self,
|
self,
|
||||||
pdf: Canvas,
|
pdf: Canvas,
|
||||||
line: Optional[Element],
|
line: Element | None,
|
||||||
elemclass: str,
|
elemclass: str,
|
||||||
fontname: str,
|
fontname: str,
|
||||||
invisible_text: bool,
|
invisible_text: bool,
|
||||||
interword_spaces: bool,
|
interword_spaces: bool,
|
||||||
show_bounding_boxes: bool,
|
show_bounding_boxes: bool,
|
||||||
):
|
):
|
||||||
if not line:
|
if line is None:
|
||||||
return
|
return
|
||||||
pxl_line_coords = self.element_coordinates(line)
|
pxl_line_coords = self.element_coordinates(line)
|
||||||
line_box = self.pt_from_pixel(pxl_line_coords)
|
line_box = self.pt_from_pixel(pxl_line_coords)
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because one or more lines are too long
@@ -1,516 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
|
||||||
#
|
|
||||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
|
||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
|
||||||
|
|
||||||
|
|
||||||
from pathlib import Path
|
|
||||||
|
|
||||||
from cffi import FFI
|
|
||||||
|
|
||||||
ffibuilder = FFI()
|
|
||||||
ffibuilder.cdef(
|
|
||||||
"""
|
|
||||||
typedef signed char l_int8;
|
|
||||||
typedef unsigned char l_uint8;
|
|
||||||
typedef short l_int16;
|
|
||||||
typedef unsigned short l_uint16;
|
|
||||||
typedef int l_int32;
|
|
||||||
typedef unsigned int l_uint32;
|
|
||||||
typedef float l_float32;
|
|
||||||
typedef double l_float64;
|
|
||||||
typedef long long l_int64;
|
|
||||||
typedef unsigned long long l_uint64;
|
|
||||||
|
|
||||||
typedef int l_ok; /*!< return type 0 if OK, 1 on error */
|
|
||||||
|
|
||||||
struct Pix
|
|
||||||
{
|
|
||||||
l_uint32 w; /* width in pixels */
|
|
||||||
l_uint32 h; /* height in pixels */
|
|
||||||
l_uint32 d; /* depth in bits (bpp) */
|
|
||||||
l_uint32 spp; /* number of samples per pixel */
|
|
||||||
l_uint32 wpl; /* 32-bit words/line */
|
|
||||||
l_uint32 refcount; /* reference count (1 if no clones) */
|
|
||||||
l_int32 xres; /* image res (ppi) in x direction */
|
|
||||||
/* (use 0 if unknown) */
|
|
||||||
l_int32 yres; /* image res (ppi) in y direction */
|
|
||||||
/* (use 0 if unknown) */
|
|
||||||
l_int32 informat; /* input file format, IFF_* */
|
|
||||||
l_int32 special; /* special instructions for I/O, etc */
|
|
||||||
char *text; /* text string associated with pix */
|
|
||||||
struct PixColormap *colormap; /* colormap (may be null) */
|
|
||||||
l_uint32 *data; /* the image data */
|
|
||||||
};
|
|
||||||
typedef struct Pix PIX;
|
|
||||||
|
|
||||||
struct PixColormap
|
|
||||||
{
|
|
||||||
void *array; /* colormap table (array of RGBA_QUAD) */
|
|
||||||
l_int32 depth; /* of pix (1, 2, 4 or 8 bpp) */
|
|
||||||
l_int32 nalloc; /* number of color entries allocated */
|
|
||||||
l_int32 n; /* number of color entries used */
|
|
||||||
};
|
|
||||||
typedef struct PixColormap PIXCMAP;
|
|
||||||
|
|
||||||
/*! Array of pix */
|
|
||||||
struct Pixa
|
|
||||||
{
|
|
||||||
l_int32 n; /*!< number of Pix in ptr array */
|
|
||||||
l_int32 nalloc; /*!< number of Pix ptrs allocated */
|
|
||||||
l_uint32 refcount; /*!< reference count (1 if no clones) */
|
|
||||||
struct Pix **pix; /*!< the array of ptrs to pix */
|
|
||||||
struct Boxa *boxa; /*!< array of boxes */
|
|
||||||
};
|
|
||||||
typedef struct Pixa PIXA;
|
|
||||||
|
|
||||||
/*! Array of compressed pix */
|
|
||||||
struct PixaComp
|
|
||||||
{
|
|
||||||
l_int32 n; /*!< number of PixComp in ptr array */
|
|
||||||
l_int32 nalloc; /*!< number of PixComp ptrs allocated */
|
|
||||||
l_int32 offset; /*!< indexing offset into ptr array */
|
|
||||||
struct PixComp **pixc; /*!< the array of ptrs to PixComp */
|
|
||||||
struct Boxa *boxa; /*!< array of boxes */
|
|
||||||
};
|
|
||||||
typedef struct PixaComp PIXAC;
|
|
||||||
|
|
||||||
struct Box
|
|
||||||
{
|
|
||||||
l_int32 x;
|
|
||||||
l_int32 y;
|
|
||||||
l_int32 w;
|
|
||||||
l_int32 h;
|
|
||||||
l_uint32 refcount; /* reference count (1 if no clones) */
|
|
||||||
|
|
||||||
};
|
|
||||||
typedef struct Box BOX;
|
|
||||||
|
|
||||||
/*! Array of Box */
|
|
||||||
struct Boxa
|
|
||||||
{
|
|
||||||
l_int32 n; /*!< number of box in ptr array */
|
|
||||||
l_int32 nalloc; /*!< number of box ptrs allocated */
|
|
||||||
l_uint32 refcount; /*!< reference count (1 if no clones) */
|
|
||||||
struct Box **box; /*!< box ptr array */
|
|
||||||
};
|
|
||||||
typedef struct Boxa BOXA;
|
|
||||||
|
|
||||||
/*! String array: an array of C strings */
|
|
||||||
struct Sarray
|
|
||||||
{
|
|
||||||
l_int32 nalloc; /*!< size of allocated ptr array */
|
|
||||||
l_int32 n; /*!< number of strings allocated */
|
|
||||||
l_int32 refcount; /*!< reference count (1 if no clones) */
|
|
||||||
char **array; /*!< string array */
|
|
||||||
};
|
|
||||||
typedef struct Sarray SARRAY;
|
|
||||||
|
|
||||||
/*! Pdf formatted encoding types */
|
|
||||||
enum {
|
|
||||||
L_DEFAULT_ENCODE = 0, /*!< use default encoding based on image */
|
|
||||||
L_JPEG_ENCODE = 1, /*!< use dct encoding: 8 and 32 bpp, no cmap */
|
|
||||||
L_G4_ENCODE = 2, /*!< use ccitt g4 fax encoding: 1 bpp */
|
|
||||||
L_FLATE_ENCODE = 3, /*!< use flate encoding: any depth, cmap ok */
|
|
||||||
L_JP2K_ENCODE = 4 /*!< use jp2k encoding: 8 and 32 bpp, no cmap */
|
|
||||||
};
|
|
||||||
|
|
||||||
/*! Compressed image data */
|
|
||||||
struct L_Compressed_Data
|
|
||||||
{
|
|
||||||
l_int32 type; /*!< encoding type: L_JPEG_ENCODE, etc */
|
|
||||||
l_uint8 *datacomp; /*!< gzipped raster data */
|
|
||||||
size_t nbytescomp; /*!< number of compressed bytes */
|
|
||||||
char *data85; /*!< ascii85-encoded gzipped raster data */
|
|
||||||
size_t nbytes85; /*!< number of ascii85 encoded bytes */
|
|
||||||
char *cmapdata85; /*!< ascii85-encoded uncompressed cmap */
|
|
||||||
char *cmapdatahex; /*!< hex pdf array for the cmap */
|
|
||||||
l_int32 ncolors; /*!< number of colors in cmap */
|
|
||||||
l_int32 w; /*!< image width */
|
|
||||||
l_int32 h; /*!< image height */
|
|
||||||
l_int32 bps; /*!< bits/sample; typ. 1, 2, 4 or 8 */
|
|
||||||
l_int32 spp; /*!< samples/pixel; typ. 1 or 3 */
|
|
||||||
l_int32 minisblack; /*!< tiff g4 photometry */
|
|
||||||
l_int32 predictor; /*!< flate data has PNG predictors */
|
|
||||||
size_t nbytes; /*!< number of uncompressed raster bytes */
|
|
||||||
l_int32 res; /*!< resolution (ppi) */
|
|
||||||
};
|
|
||||||
typedef struct L_Compressed_Data L_COMP_DATA;
|
|
||||||
|
|
||||||
/*! Selection */
|
|
||||||
struct Sel
|
|
||||||
{
|
|
||||||
l_int32 sy; /*!< sel height */
|
|
||||||
l_int32 sx; /*!< sel width */
|
|
||||||
l_int32 cy; /*!< y location of sel origin */
|
|
||||||
l_int32 cx; /*!< x location of sel origin */
|
|
||||||
l_int32 **data; /*!< {0,1,2}; data[i][j] in [row][col] order */
|
|
||||||
char *name; /*!< used to find sel by name */
|
|
||||||
};
|
|
||||||
typedef struct Sel SEL;
|
|
||||||
|
|
||||||
enum {
|
|
||||||
REMOVE_CMAP_TO_BINARY = 0, /*!< remove colormap for conv to 1 bpp */
|
|
||||||
REMOVE_CMAP_TO_GRAYSCALE = 1, /*!< remove colormap for conv to 8 bpp */
|
|
||||||
REMOVE_CMAP_TO_FULL_COLOR = 2, /*!< remove colormap for conv to 32 bpp */
|
|
||||||
REMOVE_CMAP_WITH_ALPHA = 3, /*!< remove colormap and alpha */
|
|
||||||
REMOVE_CMAP_BASED_ON_SRC = 4 /*!< remove depending on src format */
|
|
||||||
};
|
|
||||||
|
|
||||||
/*! Access and storage flags */
|
|
||||||
enum {
|
|
||||||
L_NOCOPY = 0, /*!< do not copy the object; do not delete the ptr */
|
|
||||||
L_INSERT = L_NOCOPY, /*!< stuff it in; do not copy or clone */
|
|
||||||
L_COPY = 1, /*!< make/use a copy of the object */
|
|
||||||
L_CLONE = 2, /*!< make/use clone (ref count) of the object */
|
|
||||||
L_COPY_CLONE = 3 /*!< make a new array object (e.g., pixa) and fill */
|
|
||||||
/*!< the array with clones (e.g., pix) */
|
|
||||||
};
|
|
||||||
|
|
||||||
/*! Flags for method of extracting barcode widths */
|
|
||||||
enum {
|
|
||||||
L_USE_WIDTHS = 1, /*!< use histogram of barcode widths */
|
|
||||||
L_USE_WINDOWS = 2 /*!< find best window for decoding transitions */
|
|
||||||
};
|
|
||||||
|
|
||||||
/*! Flags for barcode formats */
|
|
||||||
enum {
|
|
||||||
L_BF_UNKNOWN = 0, /*!< unknown format */
|
|
||||||
L_BF_ANY = 1, /*!< try decoding with all known formats */
|
|
||||||
L_BF_CODE128 = 2, /*!< decode with Code128 format */
|
|
||||||
L_BF_EAN8 = 3, /*!< decode with EAN8 format */
|
|
||||||
L_BF_EAN13 = 4, /*!< decode with EAN13 format */
|
|
||||||
L_BF_CODE2OF5 = 5, /*!< decode with Code 2 of 5 format */
|
|
||||||
L_BF_CODEI2OF5 = 6, /*!< decode with Interleaved 2 of 5 format */
|
|
||||||
L_BF_CODE39 = 7, /*!< decode with Code39 format */
|
|
||||||
L_BF_CODE93 = 8, /*!< decode with Code93 format */
|
|
||||||
L_BF_CODABAR = 9, /*!< decode with Code93 format */
|
|
||||||
L_BF_UPCA = 10 /*!< decode with UPC A format */
|
|
||||||
};
|
|
||||||
|
|
||||||
enum {
|
|
||||||
L_SEVERITY_EXTERNAL = 0, /* Get the severity from the environment */
|
|
||||||
L_SEVERITY_ALL = 1, /* Lowest severity: print all messages */
|
|
||||||
L_SEVERITY_DEBUG = 2, /* Print debugging and higher messages */
|
|
||||||
L_SEVERITY_INFO = 3, /* Print informational and higher messages */
|
|
||||||
L_SEVERITY_WARNING = 4, /* Print warning and higher messages */
|
|
||||||
L_SEVERITY_ERROR = 5, /* Print error and higher messages */
|
|
||||||
L_SEVERITY_NONE = 6 /* Highest severity: print no messages */
|
|
||||||
};
|
|
||||||
|
|
||||||
enum {
|
|
||||||
SEL_DONT_CARE = 0,
|
|
||||||
SEL_HIT = 1,
|
|
||||||
SEL_MISS = 2
|
|
||||||
};
|
|
||||||
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
|
|
||||||
ffibuilder.cdef(
|
|
||||||
"""
|
|
||||||
PIX * pixRead ( const char *filename );
|
|
||||||
PIX * pixReadMem ( const l_uint8 *data, size_t size );
|
|
||||||
PIX * pixReadStream ( FILE *fp, l_int32 hint );
|
|
||||||
PIX * pixScale ( PIX *pixs, l_float32 scalex, l_float32 scaley );
|
|
||||||
l_int32 pixFindSkew ( PIX *pixs, l_float32 *pangle, l_float32 *pconf );
|
|
||||||
l_int32 pixWriteImpliedFormat ( const char *filename, PIX *pix, l_int32 quality, l_int32 progressive );
|
|
||||||
l_int32 getImpliedFileFormat ( const char *filename );
|
|
||||||
l_ok pixWriteStream ( FILE *fp, PIX *pix, l_int32 format );
|
|
||||||
l_ok pixWriteStreamJpeg ( FILE *fp, PIX *pixs, l_int32 quality, l_int32 progressive );
|
|
||||||
l_ok pixWriteMem ( l_uint8 **pdata, size_t *psize, PIX *pix, l_int32 format );
|
|
||||||
l_ok pixWriteMemJpeg ( l_uint8 **pdata, size_t *psize, PIX *pix, l_int32 quality, l_int32 progressive );
|
|
||||||
l_int32
|
|
||||||
pixWriteMemPng(l_uint8 **pdata,
|
|
||||||
size_t *psize,
|
|
||||||
PIX *pix,
|
|
||||||
l_float32 gamma);
|
|
||||||
|
|
||||||
void pixDestroy ( PIX **ppix );
|
|
||||||
|
|
||||||
l_ok
|
|
||||||
pixEqual(PIX *pix1,
|
|
||||||
PIX *pix2,
|
|
||||||
l_int32 *psame);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixEndianByteSwapNew(PIX *pixs);
|
|
||||||
|
|
||||||
PIX * pixDeskew ( PIX *pixs, l_int32 redsearch );
|
|
||||||
char * getLeptonicaVersion ( );
|
|
||||||
l_int32 pixCorrelationBinary(PIX *pix1, PIX *pix2, l_float32 *pval);
|
|
||||||
PIX *pixRotate180(PIX *pixd, PIX *pixs);
|
|
||||||
PIX *
|
|
||||||
pixRotateOrth(PIX *pixs,
|
|
||||||
l_int32 quads);
|
|
||||||
|
|
||||||
l_int32 pixCountPixels ( PIX *pix, l_int32 *pcount, l_int32 *tab8 );
|
|
||||||
PIX * pixAnd ( PIX *pixd, PIX *pixs1, PIX *pixs2 );
|
|
||||||
l_int32 * makePixelSumTab8 ( void );
|
|
||||||
|
|
||||||
PIX * pixDeserializeFromMemory ( const l_uint32 *data, size_t nbytes );
|
|
||||||
l_int32 pixSerializeToMemory ( PIX *pixs, l_uint32 **pdata, size_t *pnbytes );
|
|
||||||
|
|
||||||
PIX * pixConvertRGBToLuminance(PIX *pixs);
|
|
||||||
|
|
||||||
PIX * pixConvertTo8(PIX *pixs, l_int32 cmapflag);
|
|
||||||
|
|
||||||
PIX * pixRemoveColormap(PIX *pixs, l_int32 type);
|
|
||||||
|
|
||||||
l_int32
|
|
||||||
pixOtsuAdaptiveThreshold(PIX *pixs,
|
|
||||||
l_int32 sx,
|
|
||||||
l_int32 sy,
|
|
||||||
l_int32 smoothx,
|
|
||||||
l_int32 smoothy,
|
|
||||||
l_float32 scorefract,
|
|
||||||
PIX **ppixth,
|
|
||||||
PIX **ppixd);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixOtsuThreshOnBackgroundNorm(PIX *pixs,
|
|
||||||
PIX *pixim,
|
|
||||||
l_int32 sx,
|
|
||||||
l_int32 sy,
|
|
||||||
l_int32 thresh,
|
|
||||||
l_int32 mincount,
|
|
||||||
l_int32 bgval,
|
|
||||||
l_int32 smoothx,
|
|
||||||
l_int32 smoothy,
|
|
||||||
l_float32 scorefract,
|
|
||||||
l_int32 *pthresh);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixMaskedThreshOnBackgroundNorm(PIX *pixs,
|
|
||||||
PIX *pixim,
|
|
||||||
l_int32 sx,
|
|
||||||
l_int32 sy,
|
|
||||||
l_int32 thresh,
|
|
||||||
l_int32 mincount,
|
|
||||||
l_int32 smoothx,
|
|
||||||
l_int32 smoothy,
|
|
||||||
l_float32 scorefract,
|
|
||||||
l_int32 *pthresh);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixCleanBackgroundToWhite(PIX *pixs,
|
|
||||||
PIX *pixim,
|
|
||||||
PIX *pixg,
|
|
||||||
l_float32 gamma,
|
|
||||||
l_int32 blackval,
|
|
||||||
l_int32 whiteval);
|
|
||||||
|
|
||||||
BOX *
|
|
||||||
pixFindPageForeground ( PIX *pixs,
|
|
||||||
l_int32 threshold,
|
|
||||||
l_int32 mindist,
|
|
||||||
l_int32 erasedist,
|
|
||||||
l_int32 showmorph,
|
|
||||||
PIXAC *pixac );
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixClipRectangle(PIX *pixs,
|
|
||||||
BOX *box,
|
|
||||||
BOX **pboxc);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixBackgroundNorm(PIX *pixs,
|
|
||||||
PIX *pixim,
|
|
||||||
PIX *pixg,
|
|
||||||
l_int32 sx,
|
|
||||||
l_int32 sy,
|
|
||||||
l_int32 thresh,
|
|
||||||
l_int32 mincount,
|
|
||||||
l_int32 bgval,
|
|
||||||
l_int32 smoothx,
|
|
||||||
l_int32 smoothy);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixGammaTRC(PIX *pixd,
|
|
||||||
PIX *pixs,
|
|
||||||
l_float32 gamma,
|
|
||||||
l_int32 minval,
|
|
||||||
l_int32 maxval);
|
|
||||||
|
|
||||||
|
|
||||||
l_int32
|
|
||||||
pixNumSignificantGrayColors(PIX *pixs,
|
|
||||||
l_int32 darkthresh,
|
|
||||||
l_int32 lightthresh,
|
|
||||||
l_float32 minfract,
|
|
||||||
l_int32 factor,
|
|
||||||
l_int32 *pncolors);
|
|
||||||
|
|
||||||
l_int32
|
|
||||||
pixColorFraction(PIX *pixs,
|
|
||||||
l_int32 darkthresh,
|
|
||||||
l_int32 lightthresh,
|
|
||||||
l_int32 diffthresh,
|
|
||||||
l_int32 factor,
|
|
||||||
l_float32 *ppixfract,
|
|
||||||
l_float32 *pcolorfract);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixColorMagnitude(PIX *pixs,
|
|
||||||
l_int32 rwhite,
|
|
||||||
l_int32 gwhite,
|
|
||||||
l_int32 bwhite,
|
|
||||||
l_int32 type);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixMaskOverColorPixels(PIX *pixs,
|
|
||||||
l_int32 threshdiff,
|
|
||||||
l_int32 mindist);
|
|
||||||
|
|
||||||
l_int32
|
|
||||||
pixGetAverageMaskedRGB(PIX *pixs,
|
|
||||||
PIX *pixm,
|
|
||||||
l_int32 x,
|
|
||||||
l_int32 y,
|
|
||||||
l_int32 factor,
|
|
||||||
l_int32 type,
|
|
||||||
l_float32 *prval,
|
|
||||||
l_float32 *pgval,
|
|
||||||
l_float32 *pbval);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixGlobalNormRGB(PIX * pixd,
|
|
||||||
PIX * pixs,
|
|
||||||
l_int32 rval,
|
|
||||||
l_int32 gval,
|
|
||||||
l_int32 bval,
|
|
||||||
l_int32 mapval);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixInvert(PIX * pixd,
|
|
||||||
PIX * pixs);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixRemoveColormapGeneral(PIX *pixs,
|
|
||||||
l_int32 type,
|
|
||||||
l_int32 ifnocmap);
|
|
||||||
|
|
||||||
l_int32
|
|
||||||
pixGenerateCIData(PIX *pixs,
|
|
||||||
l_int32 type,
|
|
||||||
l_int32 quality,
|
|
||||||
l_int32 ascii85,
|
|
||||||
L_COMP_DATA **pcid);
|
|
||||||
|
|
||||||
SARRAY *
|
|
||||||
pixProcessBarcodes(PIX *pixs,
|
|
||||||
l_int32 format,
|
|
||||||
l_int32 method,
|
|
||||||
SARRAY **psaw,
|
|
||||||
l_int32 debugflag);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixaGetPix(PIXA *pixa,
|
|
||||||
l_int32 index,
|
|
||||||
l_int32 accesstype);
|
|
||||||
|
|
||||||
BOX*
|
|
||||||
pixaGetBox (PIXA * pixa,
|
|
||||||
l_int32 index,
|
|
||||||
l_int32 accesstype );
|
|
||||||
|
|
||||||
PIXA *
|
|
||||||
pixExtractBarcodes(PIX *pixs,
|
|
||||||
l_int32 debugflag);
|
|
||||||
|
|
||||||
BOXA *
|
|
||||||
pixLocateBarcodes ( PIX *pixs,
|
|
||||||
l_int32 thresh,
|
|
||||||
PIX **ppixb,
|
|
||||||
PIX **ppixm );
|
|
||||||
|
|
||||||
SARRAY *
|
|
||||||
pixReadBarcodes(PIXA *pixa,
|
|
||||||
l_int32 format,
|
|
||||||
l_int32 method,
|
|
||||||
SARRAY **psaw,
|
|
||||||
l_int32 debugflag);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixGenHalftoneMask(PIX *pixs,
|
|
||||||
PIX **ppixtext,
|
|
||||||
l_int32 *phtfound,
|
|
||||||
PIXA *pixadb);
|
|
||||||
|
|
||||||
l_int32
|
|
||||||
l_generateCIDataForPdf(const char *fname,
|
|
||||||
PIX *pix,
|
|
||||||
l_int32 quality,
|
|
||||||
L_COMP_DATA **pcid);
|
|
||||||
|
|
||||||
|
|
||||||
BOX *
|
|
||||||
boxClone ( BOX *box );
|
|
||||||
|
|
||||||
BOX *
|
|
||||||
boxaGetBox ( BOXA *boxa, l_int32 index, l_int32 accessflag );
|
|
||||||
|
|
||||||
SEL *
|
|
||||||
selCreateFromString ( const char *text, l_int32 h, l_int32 w, const char *name );
|
|
||||||
|
|
||||||
SEL *
|
|
||||||
selCreateBrick ( l_int32 h, l_int32 w, l_int32 cy, l_int32 cx, l_int32 type );
|
|
||||||
|
|
||||||
char *
|
|
||||||
selPrintToString(SEL *sel);
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixDilate ( PIX *pixd, PIX *pixs, SEL *sel );
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixErode ( PIX *pixd, PIX *pixs, SEL *sel );
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixHMT ( PIX *pixd, PIX *pixs, SEL *sel );
|
|
||||||
|
|
||||||
PIX *
|
|
||||||
pixSubtract ( PIX *pixd, PIX *pixs1, PIX *pixs2 );
|
|
||||||
|
|
||||||
void
|
|
||||||
boxDestroy(BOX **pbox);
|
|
||||||
|
|
||||||
void
|
|
||||||
boxaDestroy ( BOXA **pboxa );
|
|
||||||
|
|
||||||
void
|
|
||||||
pixaDestroy(PIXA **ppixa);
|
|
||||||
|
|
||||||
l_ok
|
|
||||||
pixRenderBoxa ( PIX *pix, BOXA *boxa, l_int32 width, l_int32 op );
|
|
||||||
|
|
||||||
void
|
|
||||||
l_CIDataDestroy(L_COMP_DATA **pcid);
|
|
||||||
|
|
||||||
void
|
|
||||||
sarrayDestroy(SARRAY **psa);
|
|
||||||
|
|
||||||
void
|
|
||||||
lept_free(void *ptr);
|
|
||||||
|
|
||||||
void selDestroy ( SEL **psel );
|
|
||||||
|
|
||||||
l_int32
|
|
||||||
setMsgSeverity(l_int32 newsev);
|
|
||||||
|
|
||||||
void
|
|
||||||
leptSetStderrHandler(void (*handler)(const char *));
|
|
||||||
"""
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
ffibuilder.set_source("ocrmypdf.lib._leptonica", None)
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
ffibuilder.compile(verbose=True)
|
|
||||||
if Path('ocrmypdf/lib/_leptonica.py').exists() and Path('src/ocrmypdf').exists():
|
|
||||||
output = Path('ocrmypdf/lib/_leptonica.py')
|
|
||||||
output.rename('src/ocrmypdf/lib/_leptonica.py')
|
|
||||||
Path('ocrmypdf/lib').rmdir()
|
|
||||||
Path('ocrmypdf').rmdir()
|
|
||||||
+178
-88
@@ -4,37 +4,40 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""Post-processing image optimization of OCR PDFs."""
|
||||||
|
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
|
import threading
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import (
|
from typing import Callable, Iterator, MutableSet, NamedTuple, NewType, Sequence
|
||||||
Callable,
|
from zlib import compress
|
||||||
Dict,
|
|
||||||
Iterator,
|
|
||||||
List,
|
|
||||||
MutableSet,
|
|
||||||
NamedTuple,
|
|
||||||
NewType,
|
|
||||||
Optional,
|
|
||||||
Sequence,
|
|
||||||
Tuple,
|
|
||||||
)
|
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
import pikepdf
|
from pikepdf import (
|
||||||
from pikepdf import Dictionary, Name, Object, Pdf, PdfImage
|
Dictionary,
|
||||||
|
Name,
|
||||||
|
Object,
|
||||||
|
ObjectStreamMode,
|
||||||
|
Pdf,
|
||||||
|
PdfError,
|
||||||
|
PdfImage,
|
||||||
|
Stream,
|
||||||
|
UnsupportedImageTypeError,
|
||||||
|
)
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ocrmypdf import leptonica
|
|
||||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||||
from ocrmypdf._exec import jbig2enc, pngquant
|
from ocrmypdf._exec import jbig2enc, pngquant
|
||||||
from ocrmypdf._jobcontext import PdfContext
|
from ocrmypdf._jobcontext import PdfContext
|
||||||
from ocrmypdf.exceptions import OutputFileAccessError
|
from ocrmypdf.exceptions import OutputFileAccessError
|
||||||
from ocrmypdf.helpers import safe_symlink
|
from ocrmypdf.helpers import IMG2PDF_KWARGS, safe_symlink
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
@@ -45,7 +48,9 @@ DEFAULT_PNG_QUALITY = 70
|
|||||||
Xref = NewType('Xref', int)
|
Xref = NewType('Xref', int)
|
||||||
|
|
||||||
|
|
||||||
class XrefExt(NamedTuple): # pylint: disable=inherit-non-class
|
class XrefExt(NamedTuple):
|
||||||
|
"""A PDF xref and image extension pair."""
|
||||||
|
|
||||||
xref: Xref
|
xref: Xref
|
||||||
ext: str
|
ext: str
|
||||||
|
|
||||||
@@ -63,49 +68,65 @@ def jpg_name(root: Path, xref: Xref) -> Path:
|
|||||||
|
|
||||||
|
|
||||||
def extract_image_filter(
|
def extract_image_filter(
|
||||||
pike: Pdf, root: Path, image: Object, xref: Xref
|
pike: Pdf, root: Path, image: Stream, xref: Xref
|
||||||
) -> Optional[Tuple[PdfImage, Tuple[Name, Object]]]:
|
) -> tuple[PdfImage, tuple[Name, Object]] | None:
|
||||||
del pike # unused args
|
del pike # unused args
|
||||||
del root
|
del root
|
||||||
|
|
||||||
if image.Subtype != Name.Image:
|
if image.Subtype != Name.Image:
|
||||||
return None
|
return None
|
||||||
if image.Length < 100:
|
if image.Length < 100:
|
||||||
log.debug(f"Skipping small image, xref {xref}")
|
log.debug(f"xref {xref}: skipping image with small stream size")
|
||||||
return None
|
return None
|
||||||
if image.Width < 8 or image.Height < 8: # Issue 732
|
if image.Width < 8 or image.Height < 8: # Issue 732
|
||||||
log.debug(f"Skipping oddly sized image, xref {xref}")
|
log.debug(f"xref {xref}: skipping image with unusually small dimensions")
|
||||||
return None
|
return None
|
||||||
|
|
||||||
pim = PdfImage(image)
|
pim = PdfImage(image)
|
||||||
|
|
||||||
if len(pim.filter_decodeparms) > 1:
|
if len(pim.filter_decodeparms) > 1:
|
||||||
log.debug(f"Skipping multiply filtered image, xref {xref}")
|
first_filtdp = pim.filter_decodeparms[0]
|
||||||
return None
|
second_filtdp = pim.filter_decodeparms[1]
|
||||||
filtdp = pim.filter_decodeparms[0]
|
if (
|
||||||
|
len(pim.filter_decodeparms) == 2
|
||||||
|
and first_filtdp[0] == Name.FlateDecode
|
||||||
|
and first_filtdp[1].get(Name.Predictor, 1) == 1
|
||||||
|
and second_filtdp[0] == Name.DCTDecode
|
||||||
|
and not second_filtdp[1]
|
||||||
|
):
|
||||||
|
log.debug(
|
||||||
|
f"xref {xref}: found image compressed as /FlateDecode /DCTDecode, "
|
||||||
|
"marked for JPEG optimization"
|
||||||
|
)
|
||||||
|
filtdp = pim.filter_decodeparms[1]
|
||||||
|
else:
|
||||||
|
log.debug(f"xref {xref}: skipping image with multiple compression filters")
|
||||||
|
return None
|
||||||
|
else:
|
||||||
|
filtdp = pim.filter_decodeparms[0]
|
||||||
|
|
||||||
if pim.bits_per_component > 8:
|
if pim.bits_per_component > 8:
|
||||||
log.debug(f"Skipping wide gamut image, xref {xref}")
|
log.debug(f"xref {xref}: skipping wide gamut image")
|
||||||
return None # Don't mess with wide gamut images
|
return None # Don't mess with wide gamut images
|
||||||
|
|
||||||
if filtdp[0] == Name.JPXDecode:
|
if filtdp[0] == Name.JPXDecode:
|
||||||
log.debug(f"Skipping JPEG2000 iamge, xref {xref}")
|
log.debug(f"xref {xref}: skipping JPEG2000 image")
|
||||||
return None # Don't do JPEG2000
|
return None # Don't do JPEG2000
|
||||||
|
|
||||||
if filtdp[0] == Name.CCITTFaxDecode and filtdp[1].get('/K', 0) >= 0:
|
if filtdp[0] == Name.CCITTFaxDecode and filtdp[1].get('/K', 0) >= 0:
|
||||||
log.debug(f"Skipping CCITT Group 3 image, xref {xref}")
|
log.debug(f"xref {xref}: skipping CCITT Group 3 image")
|
||||||
return None # pikepdf doesn't support Group 3 yet
|
return None # pikepdf doesn't support Group 3 yet
|
||||||
|
|
||||||
if Name.Decode in image:
|
if Name.Decode in image:
|
||||||
log.debug(f"Skipping image with Decode table, xref {xref}")
|
log.debug(f"xref {xref}: skipping image with Decode table")
|
||||||
return None # Don't mess with custom Decode tables
|
return None # Don't mess with custom Decode tables
|
||||||
|
|
||||||
return pim, filtdp
|
return pim, filtdp
|
||||||
|
|
||||||
|
|
||||||
def extract_image_jbig2(
|
def extract_image_jbig2(
|
||||||
*, pike: pikepdf.Pdf, root: Path, image: Object, xref: Xref, options
|
*, pike: Pdf, root: Path, image: Stream, xref: Xref, options
|
||||||
) -> Optional[XrefExt]:
|
) -> XrefExt | None:
|
||||||
del options # unused arg
|
del options # unused arg
|
||||||
|
|
||||||
result = extract_image_filter(pike, root, image, xref)
|
result = extract_image_filter(pike, root, image, xref)
|
||||||
@@ -123,16 +144,16 @@ def extract_image_jbig2(
|
|||||||
# Showing the palette or ICC to jbig2enc will cause it to perform
|
# Showing the palette or ICC to jbig2enc will cause it to perform
|
||||||
# colorspace transform to 1bpp, which will conflict the palette or
|
# colorspace transform to 1bpp, which will conflict the palette or
|
||||||
# ICC if it exists.
|
# ICC if it exists.
|
||||||
colorspace = pim.obj.get(pikepdf.Name.ColorSpace, None)
|
colorspace = pim.obj.get(Name.ColorSpace, None)
|
||||||
if colorspace is not None or pim.image_mask:
|
if colorspace is not None or pim.image_mask:
|
||||||
try:
|
try:
|
||||||
# Set to DeviceGray temporarily; we already in 1 bpc.
|
# Set to DeviceGray temporarily; we already in 1 bpc.
|
||||||
pim.obj.ColorSpace = pikepdf.Name.DeviceGray
|
pim.obj.ColorSpace = Name.DeviceGray
|
||||||
imgname = root / f'{xref:08d}'
|
imgname = root / f'{xref:08d}'
|
||||||
with imgname.open('wb') as f:
|
with imgname.open('wb') as f:
|
||||||
ext = pim.extract_to(stream=f)
|
ext = pim.extract_to(stream=f)
|
||||||
imgname.rename(imgname.with_suffix(ext))
|
imgname.rename(imgname.with_suffix(ext))
|
||||||
except pikepdf.UnsupportedImageTypeError:
|
except UnsupportedImageTypeError:
|
||||||
return None
|
return None
|
||||||
finally:
|
finally:
|
||||||
# Restore image colorspace after temporarily setting it to DeviceGray
|
# Restore image colorspace after temporarily setting it to DeviceGray
|
||||||
@@ -145,8 +166,8 @@ def extract_image_jbig2(
|
|||||||
|
|
||||||
|
|
||||||
def extract_image_generic(
|
def extract_image_generic(
|
||||||
*, pike: Pdf, root: Path, image: PdfImage, xref: Xref, options
|
*, pike: Pdf, root: Path, image: Stream, xref: Xref, options
|
||||||
) -> Optional[XrefExt]:
|
) -> XrefExt | None:
|
||||||
result = extract_image_filter(pike, root, image, xref)
|
result = extract_image_filter(pike, root, image, xref)
|
||||||
if result is None:
|
if result is None:
|
||||||
return None
|
return None
|
||||||
@@ -165,20 +186,12 @@ def extract_image_generic(
|
|||||||
# jpeg_quality_estimate = 117.0 * (bytes_per_pixel ** 0.213)
|
# jpeg_quality_estimate = 117.0 * (bytes_per_pixel ** 0.213)
|
||||||
# if jpeg_quality_estimate < 65:
|
# if jpeg_quality_estimate < 65:
|
||||||
# return None
|
# return None
|
||||||
|
|
||||||
# We could get the ICC profile here, but there's no need to look at it
|
|
||||||
# for quality transcoding
|
|
||||||
# if icc:
|
|
||||||
# stream = BytesIO(raw_jpeg.read_raw_bytes())
|
|
||||||
# iccbytes = icc.read_bytes()
|
|
||||||
# with Image.open(stream) as im:
|
|
||||||
# im.save(jpg_name(root, xref), icc_profile=iccbytes)
|
|
||||||
try:
|
try:
|
||||||
imgname = root / f'{xref:08d}'
|
imgname = root / f'{xref:08d}'
|
||||||
with imgname.open('wb') as f:
|
with imgname.open('wb') as f:
|
||||||
ext = pim.extract_to(stream=f)
|
ext = pim.extract_to(stream=f)
|
||||||
imgname.rename(imgname.with_suffix(ext))
|
imgname.rename(imgname.with_suffix(ext))
|
||||||
except pikepdf.UnsupportedImageTypeError:
|
except UnsupportedImageTypeError:
|
||||||
return None
|
return None
|
||||||
return XrefExt(xref, ext)
|
return XrefExt(xref, ext)
|
||||||
elif (
|
elif (
|
||||||
@@ -193,7 +206,11 @@ def extract_image_generic(
|
|||||||
elif not pim.indexed and pim.colorspace in pim.SIMPLE_COLORSPACES:
|
elif not pim.indexed and pim.colorspace in pim.SIMPLE_COLORSPACES:
|
||||||
# An optimization opportunity here, not currently taken, is directly
|
# An optimization opportunity here, not currently taken, is directly
|
||||||
# generating a PNG from compressed data
|
# generating a PNG from compressed data
|
||||||
pim.as_pil_image().save(png_name(root, xref))
|
try:
|
||||||
|
pim.as_pil_image().save(png_name(root, xref))
|
||||||
|
except NotImplementedError:
|
||||||
|
log.warning("PDF contains an atypical image that cannot be optimized.")
|
||||||
|
return None
|
||||||
return XrefExt(xref, '.png')
|
return XrefExt(xref, '.png')
|
||||||
elif (
|
elif (
|
||||||
not pim.indexed
|
not pim.indexed
|
||||||
@@ -214,8 +231,8 @@ def extract_images(
|
|||||||
pike: Pdf,
|
pike: Pdf,
|
||||||
root: Path,
|
root: Path,
|
||||||
options,
|
options,
|
||||||
extract_fn: Callable[..., Optional[XrefExt]],
|
extract_fn: Callable[..., XrefExt | None],
|
||||||
) -> Iterator[Tuple[int, XrefExt]]:
|
) -> Iterator[tuple[int, XrefExt]]:
|
||||||
"""Extract image using extract_fn
|
"""Extract image using extract_fn
|
||||||
|
|
||||||
Enumerate images on each page, lookup their xref/ID number in the PDF.
|
Enumerate images on each page, lookup their xref/ID number in the PDF.
|
||||||
@@ -244,13 +261,13 @@ def extract_images(
|
|||||||
if image.objgen[1] != 0:
|
if image.objgen[1] != 0:
|
||||||
continue # Ignore images in an incremental PDF
|
continue # Ignore images in an incremental PDF
|
||||||
xref = Xref(image.objgen[0])
|
xref = Xref(image.objgen[0])
|
||||||
if hasattr(image, 'SMask'):
|
if Name.SMask in image:
|
||||||
# Ignore soft masks
|
# Ignore soft masks
|
||||||
smask_xref = Xref(image.SMask.objgen[0])
|
smask_xref = Xref(image.SMask.objgen[0])
|
||||||
exclude_xrefs.add(smask_xref)
|
exclude_xrefs.add(smask_xref)
|
||||||
log.debug(f"Skipping image {smask_xref} because it is an SMask")
|
log.debug(f"xref {smask_xref}: skipping image because it is an SMask")
|
||||||
include_xrefs.add(xref)
|
include_xrefs.add(xref)
|
||||||
log.debug(f"Treating {xref} as an optimization candidate")
|
log.debug(f"xref {xref}: treating as an optimization candidate")
|
||||||
if xref not in pageno_for_xref:
|
if xref not in pageno_for_xref:
|
||||||
pageno_for_xref[xref] = pageno
|
pageno_for_xref[xref] = pageno
|
||||||
|
|
||||||
@@ -262,7 +279,9 @@ def extract_images(
|
|||||||
pike=pike, root=root, image=image, xref=xref, options=options
|
pike=pike, root=root, image=image, xref=xref, options=options
|
||||||
)
|
)
|
||||||
except Exception: # pylint: disable=broad-except
|
except Exception: # pylint: disable=broad-except
|
||||||
log.exception(f"While extracting image xref {xref}, an error occurred")
|
log.exception(
|
||||||
|
f"xref {xref}: While extracting this image, an error occurred"
|
||||||
|
)
|
||||||
errors += 1
|
errors += 1
|
||||||
else:
|
else:
|
||||||
if result:
|
if result:
|
||||||
@@ -272,7 +291,7 @@ def extract_images(
|
|||||||
|
|
||||||
def extract_images_generic(
|
def extract_images_generic(
|
||||||
pike: Pdf, root: Path, options
|
pike: Pdf, root: Path, options
|
||||||
) -> Tuple[List[Xref], List[Xref]]:
|
) -> tuple[list[Xref], list[Xref]]:
|
||||||
"""Extract any >=2bpp image we think we can improve"""
|
"""Extract any >=2bpp image we think we can improve"""
|
||||||
|
|
||||||
jpegs = []
|
jpegs = []
|
||||||
@@ -283,11 +302,11 @@ def extract_images_generic(
|
|||||||
pngs.append(xref_ext.xref)
|
pngs.append(xref_ext.xref)
|
||||||
elif xref_ext.ext == '.jpg':
|
elif xref_ext.ext == '.jpg':
|
||||||
jpegs.append(xref_ext.xref)
|
jpegs.append(xref_ext.xref)
|
||||||
log.debug("Optimizable images: JPEGs: %s PNGs: %s", len(jpegs), len(pngs))
|
log.debug(f"Optimizable images: JPEGs: {len(jpegs)} PNGs: {len(pngs)}")
|
||||||
return jpegs, pngs
|
return jpegs, pngs
|
||||||
|
|
||||||
|
|
||||||
def extract_images_jbig2(pike: Pdf, root: Path, options) -> Dict[int, List[XrefExt]]:
|
def extract_images_jbig2(pike: Pdf, root: Path, options) -> dict[int, list[XrefExt]]:
|
||||||
"""Extract any bitonal image that we think we can improve as JBIG2"""
|
"""Extract any bitonal image that we think we can improve as JBIG2"""
|
||||||
|
|
||||||
jbig2_groups = defaultdict(list)
|
jbig2_groups = defaultdict(list)
|
||||||
@@ -295,16 +314,16 @@ def extract_images_jbig2(pike: Pdf, root: Path, options) -> Dict[int, List[XrefE
|
|||||||
group = pageno // options.jbig2_page_group_size
|
group = pageno // options.jbig2_page_group_size
|
||||||
jbig2_groups[group].append(xref_ext)
|
jbig2_groups[group].append(xref_ext)
|
||||||
|
|
||||||
log.debug("Optimizable images: JBIG2 groups: %s", (len(jbig2_groups),))
|
log.debug(f"Optimizable images: JBIG2 groups: {len(jbig2_groups)}")
|
||||||
return jbig2_groups
|
return jbig2_groups
|
||||||
|
|
||||||
|
|
||||||
def _produce_jbig2_images(
|
def _produce_jbig2_images(
|
||||||
jbig2_groups: Dict[int, List[XrefExt]], root: Path, options, executor: Executor
|
jbig2_groups: dict[int, list[XrefExt]], root: Path, options, executor: Executor
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Produce JBIG2 images from their groups"""
|
"""Produce JBIG2 images from their groups"""
|
||||||
|
|
||||||
def jbig2_group_args(root: Path, groups: Dict[int, List[XrefExt]]):
|
def jbig2_group_args(root: Path, groups: dict[int, list[XrefExt]]):
|
||||||
for group, xref_exts in groups.items():
|
for group, xref_exts in groups.items():
|
||||||
prefix = f'group{group:08d}'
|
prefix = f'group{group:08d}'
|
||||||
yield (
|
yield (
|
||||||
@@ -313,7 +332,7 @@ def _produce_jbig2_images(
|
|||||||
prefix, # =out_prefix
|
prefix, # =out_prefix
|
||||||
)
|
)
|
||||||
|
|
||||||
def jbig2_single_args(root, groups: Dict[int, List[XrefExt]]):
|
def jbig2_single_args(root, groups: dict[int, list[XrefExt]]):
|
||||||
for group, xref_exts in groups.items():
|
for group, xref_exts in groups.items():
|
||||||
prefix = f'group{group:08d}'
|
prefix = f'group{group:08d}'
|
||||||
# Second loop is to ensure multiple images per page are unpacked
|
# Second loop is to ensure multiple images per page are unpacked
|
||||||
@@ -348,7 +367,7 @@ def _produce_jbig2_images(
|
|||||||
|
|
||||||
def convert_to_jbig2(
|
def convert_to_jbig2(
|
||||||
pike: Pdf,
|
pike: Pdf,
|
||||||
jbig2_groups: Dict[int, List[XrefExt]],
|
jbig2_groups: dict[int, list[XrefExt]],
|
||||||
root: Path,
|
root: Path,
|
||||||
options,
|
options,
|
||||||
executor: Executor,
|
executor: Executor,
|
||||||
@@ -365,6 +384,7 @@ def convert_to_jbig2(
|
|||||||
When the JBIG2 symbolic coder is not used, each JBIG2 stands on its own
|
When the JBIG2 symbolic coder is not used, each JBIG2 stands on its own
|
||||||
and needs no dictionary. Currently this must be lossless JBIG2.
|
and needs no dictionary. Currently this must be lossless JBIG2.
|
||||||
"""
|
"""
|
||||||
|
jbig2_globals_dict: Dictionary | None
|
||||||
|
|
||||||
_produce_jbig2_images(jbig2_groups, root, options, executor)
|
_produce_jbig2_images(jbig2_groups, root, options, executor)
|
||||||
|
|
||||||
@@ -373,7 +393,7 @@ def convert_to_jbig2(
|
|||||||
jbig2_symfile = root / (prefix + '.sym')
|
jbig2_symfile = root / (prefix + '.sym')
|
||||||
if jbig2_symfile.exists():
|
if jbig2_symfile.exists():
|
||||||
jbig2_globals_data = jbig2_symfile.read_bytes()
|
jbig2_globals_data = jbig2_symfile.read_bytes()
|
||||||
jbig2_globals = pikepdf.Stream(pike, jbig2_globals_data)
|
jbig2_globals = Stream(pike, jbig2_globals_data)
|
||||||
jbig2_globals_dict = Dictionary(JBIG2Globals=jbig2_globals)
|
jbig2_globals_dict = Dictionary(JBIG2Globals=jbig2_globals)
|
||||||
elif options.jbig2_page_group_size == 1:
|
elif options.jbig2_page_group_size == 1:
|
||||||
jbig2_globals_dict = None
|
jbig2_globals_dict = None
|
||||||
@@ -390,45 +410,41 @@ def convert_to_jbig2(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _optimize_jpeg(args):
|
def _optimize_jpeg(args: tuple[Xref, Path, Path, int]) -> tuple[Xref, Path | None]:
|
||||||
xref, in_jpg, opt_jpg, jpeg_quality = args
|
xref, in_jpg, opt_jpg, jpeg_quality = args
|
||||||
|
|
||||||
# This may produce a debug warning from PIL
|
|
||||||
# DEBUG:PIL.Image:Error closing: 'NoneType' object has no attribute
|
|
||||||
# 'close'. Seems to be mostly harmless
|
|
||||||
# https://github.com/python-pillow/Pillow/issues/1144
|
|
||||||
with Image.open(in_jpg) as im:
|
with Image.open(in_jpg) as im:
|
||||||
im.save(opt_jpg, optimize=True, quality=jpeg_quality)
|
im.save(opt_jpg, optimize=True, quality=jpeg_quality)
|
||||||
|
|
||||||
if opt_jpg.stat().st_size > in_jpg.stat().st_size:
|
if opt_jpg.stat().st_size > in_jpg.stat().st_size:
|
||||||
log.debug("xref %s, jpeg, made larger - skip", xref)
|
log.debug(f"xref {xref}, jpeg, made larger - skip")
|
||||||
opt_jpg.unlink()
|
opt_jpg.unlink()
|
||||||
opt_jpg = None
|
return xref, None
|
||||||
return xref, opt_jpg
|
return xref, opt_jpg
|
||||||
|
|
||||||
|
|
||||||
def transcode_jpegs(
|
def transcode_jpegs(
|
||||||
pike: Pdf, jpegs: Sequence[Xref], root: Path, options, executor
|
pike: Pdf, jpegs: Sequence[Xref], root: Path, options, executor: Executor
|
||||||
) -> None:
|
) -> None:
|
||||||
def jpeg_args():
|
def jpeg_args() -> Iterator[tuple[Xref, Path, Path, int]]:
|
||||||
for xref in jpegs:
|
for xref in jpegs:
|
||||||
in_jpg = jpg_name(root, xref)
|
in_jpg = jpg_name(root, xref)
|
||||||
opt_jpg = in_jpg.with_suffix('.opt.jpg')
|
opt_jpg = in_jpg.with_suffix('.opt.jpg')
|
||||||
yield xref, in_jpg, opt_jpg, options.jpeg_quality
|
yield xref, in_jpg, opt_jpg, options.jpeg_quality
|
||||||
|
|
||||||
def finish_jpeg(result, pbar):
|
def finish_jpeg(result: tuple[Xref, Path | None], pbar):
|
||||||
xref, opt_jpg = result
|
xref, opt_jpg = result
|
||||||
if opt_jpg:
|
if opt_jpg:
|
||||||
compdata = leptonica.CompressedData.open(opt_jpg)
|
compdata = opt_jpg.read_bytes() # JPEG can inserted into PDF as is
|
||||||
im_obj = pike.get_object(xref, 0)
|
im_obj = pike.get_object(xref, 0)
|
||||||
im_obj.write(compdata.read(), filter=Name.DCTDecode)
|
im_obj.write(compdata, filter=Name.DCTDecode)
|
||||||
pbar.update()
|
pbar.update()
|
||||||
|
|
||||||
executor(
|
executor(
|
||||||
use_threads=True, # Processes are significantly slower at this task
|
use_threads=True, # Processes are significantly slower at this task
|
||||||
max_workers=options.jobs,
|
max_workers=options.jobs,
|
||||||
tqdm_kwargs=dict(
|
tqdm_kwargs=dict(
|
||||||
desc="JPEGs",
|
desc="Recompressing JPEGs",
|
||||||
total=len(jpegs),
|
total=len(jpegs),
|
||||||
unit='image',
|
unit='image',
|
||||||
disable=not options.progress_bar,
|
disable=not options.progress_bar,
|
||||||
@@ -439,13 +455,80 @@ def transcode_jpegs(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _find_deflatable_jpeg(
|
||||||
|
*, pike: Pdf, root: Path, image: Stream, xref: Xref, options
|
||||||
|
) -> XrefExt | None:
|
||||||
|
result = extract_image_filter(pike, root, image, xref)
|
||||||
|
if result is None:
|
||||||
|
return None
|
||||||
|
_pim, filtdp = result
|
||||||
|
|
||||||
|
if filtdp[0] == Name.DCTDecode and not filtdp[1] and options.optimize >= 1:
|
||||||
|
return XrefExt(xref, '.memory')
|
||||||
|
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _deflate_jpeg(args: tuple[Pdf, threading.Lock, Xref, int]) -> tuple[Xref, bytes]:
|
||||||
|
pike, lock, xref, complevel = args
|
||||||
|
with lock:
|
||||||
|
xobj = pike.get_object(xref, 0)
|
||||||
|
try:
|
||||||
|
data = xobj.read_raw_bytes()
|
||||||
|
except PdfError:
|
||||||
|
return xref, b''
|
||||||
|
compdata = compress(data, complevel)
|
||||||
|
if len(compdata) >= len(data):
|
||||||
|
return xref, b''
|
||||||
|
return xref, compdata
|
||||||
|
|
||||||
|
|
||||||
|
def deflate_jpegs(pike: Pdf, root: Path, options, executor: Executor) -> None:
|
||||||
|
jpegs = []
|
||||||
|
for _pageno, xref_ext in extract_images(pike, root, options, _find_deflatable_jpeg):
|
||||||
|
xref = xref_ext.xref
|
||||||
|
log.debug(f'xref {xref}: marking this JPEG as deflatable')
|
||||||
|
jpegs.append(xref)
|
||||||
|
|
||||||
|
complevel = 9 if options.optimize == 3 else 6
|
||||||
|
|
||||||
|
# Our calls to xobj.write() in finish() need coordination
|
||||||
|
lock = threading.Lock()
|
||||||
|
|
||||||
|
def deflate_args() -> Iterator:
|
||||||
|
for xref in jpegs:
|
||||||
|
yield pike, lock, xref, complevel
|
||||||
|
|
||||||
|
def finish(result, pbar):
|
||||||
|
xref, compdata = result
|
||||||
|
if len(compdata) > 0:
|
||||||
|
with lock:
|
||||||
|
xobj = pike.get_object(xref, 0)
|
||||||
|
xobj.write(compdata, filter=[Name.FlateDecode, Name.DCTDecode])
|
||||||
|
pbar.update()
|
||||||
|
|
||||||
|
executor(
|
||||||
|
use_threads=True, # We're sharing the pdf directly, must use threads
|
||||||
|
max_workers=options.jobs,
|
||||||
|
tqdm_kwargs=dict(
|
||||||
|
desc="Deflating JPEGs",
|
||||||
|
total=len(jpegs),
|
||||||
|
unit='image',
|
||||||
|
disable=not options.progress_bar,
|
||||||
|
),
|
||||||
|
task=_deflate_jpeg,
|
||||||
|
task_arguments=deflate_args(),
|
||||||
|
task_finished=finish,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool:
|
def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool:
|
||||||
output = filename.with_suffix('.png.pdf')
|
output = filename.with_suffix('.png.pdf')
|
||||||
with output.open('wb') as f:
|
with output.open('wb') as f:
|
||||||
img2pdf.convert(fspath(filename), outputstream=f)
|
img2pdf.convert(fspath(filename), outputstream=f, **IMG2PDF_KWARGS)
|
||||||
|
|
||||||
with pikepdf.open(output) as pdf_image:
|
with Pdf.open(output) as pdf_image:
|
||||||
foreign_image = next(pdf_image.pages[0].images.values())
|
foreign_image = next(iter(pdf_image.pages[0].images.values()))
|
||||||
local_image = pike.copy_foreign(foreign_image)
|
local_image = pike.copy_foreign(foreign_image)
|
||||||
|
|
||||||
im_obj = pike.get_object(xref, 0)
|
im_obj = pike.get_object(xref, 0)
|
||||||
@@ -524,17 +607,20 @@ def transcode_pngs(
|
|||||||
_transcode_png(pike, filename, xref)
|
_transcode_png(pike, filename, xref)
|
||||||
|
|
||||||
|
|
||||||
|
DEFAULT_EXECUTOR = SerialExecutor()
|
||||||
|
|
||||||
|
|
||||||
def optimize(
|
def optimize(
|
||||||
input_file: Path,
|
input_file: Path,
|
||||||
output_file: Path,
|
output_file: Path,
|
||||||
context,
|
context,
|
||||||
save_settings,
|
save_settings,
|
||||||
executor: Executor = SerialExecutor(),
|
executor: Executor = DEFAULT_EXECUTOR,
|
||||||
) -> None:
|
) -> Path:
|
||||||
options = context.options
|
options = context.options
|
||||||
if options.optimize == 0:
|
if options.optimize == 0:
|
||||||
safe_symlink(input_file, output_file)
|
safe_symlink(input_file, output_file)
|
||||||
return
|
return output_file
|
||||||
|
|
||||||
if options.jpeg_quality == 0:
|
if options.jpeg_quality == 0:
|
||||||
options.jpeg_quality = DEFAULT_JPEG_QUALITY if options.optimize < 3 else 40
|
options.jpeg_quality = DEFAULT_JPEG_QUALITY if options.optimize < 3 else 40
|
||||||
@@ -543,12 +629,13 @@ def optimize(
|
|||||||
if options.jbig2_page_group_size == 0:
|
if options.jbig2_page_group_size == 0:
|
||||||
options.jbig2_page_group_size = 10 if options.jbig2_lossy else 1
|
options.jbig2_page_group_size = 10 if options.jbig2_lossy else 1
|
||||||
|
|
||||||
with pikepdf.Pdf.open(input_file) as pike:
|
with Pdf.open(input_file) as pike:
|
||||||
root = output_file.parent / 'images'
|
root = output_file.parent / 'images'
|
||||||
root.mkdir(exist_ok=True)
|
root.mkdir(exist_ok=True)
|
||||||
|
|
||||||
jpegs, pngs = extract_images_generic(pike, root, options)
|
jpegs, pngs = extract_images_generic(pike, root, options)
|
||||||
transcode_jpegs(pike, jpegs, root, options, executor)
|
transcode_jpegs(pike, jpegs, root, options, executor)
|
||||||
|
deflate_jpegs(pike, root, options, executor)
|
||||||
# if options.optimize >= 2:
|
# if options.optimize >= 2:
|
||||||
# Try pngifying the jpegs
|
# Try pngifying the jpegs
|
||||||
# transcode_pngs(pike, jpegs, jpg_name, root, options)
|
# transcode_pngs(pike, jpegs, jpg_name, root, options)
|
||||||
@@ -568,19 +655,22 @@ def optimize(
|
|||||||
f"Output file not created after optimizing. We probably ran "
|
f"Output file not created after optimizing. We probably ran "
|
||||||
f"out of disk space in the temporary folder: {tempfile.gettempdir()}."
|
f"out of disk space in the temporary folder: {tempfile.gettempdir()}."
|
||||||
)
|
)
|
||||||
ratio = input_size / output_size
|
|
||||||
savings = 1 - output_size / input_size
|
savings = 1 - output_size / input_size
|
||||||
log.info(f"Optimize ratio: {ratio:.2f} savings: {(savings):.1%}")
|
|
||||||
|
|
||||||
if savings < 0:
|
if savings < 0:
|
||||||
log.info("Image optimization did not improve the file - discarded")
|
log.info(
|
||||||
|
"Image optimization did not improve the file - "
|
||||||
|
"optimizations will not be used"
|
||||||
|
)
|
||||||
# We still need to save the file
|
# We still need to save the file
|
||||||
with pikepdf.open(input_file) as pike:
|
with Pdf.open(input_file) as pike:
|
||||||
pike.remove_unreferenced_resources()
|
pike.remove_unreferenced_resources()
|
||||||
pike.save(output_file, **save_settings)
|
pike.save(output_file, **save_settings)
|
||||||
else:
|
else:
|
||||||
safe_symlink(target_file, output_file)
|
safe_symlink(target_file, output_file)
|
||||||
|
|
||||||
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def main(infile, outfile, level, jobs=1):
|
def main(infile, outfile, level, jobs=1):
|
||||||
from shutil import copy # pylint: disable=import-outside-toplevel
|
from shutil import copy # pylint: disable=import-outside-toplevel
|
||||||
@@ -612,9 +702,9 @@ def main(infile, outfile, level, jobs=1):
|
|||||||
jb2lossy=False,
|
jb2lossy=False,
|
||||||
)
|
)
|
||||||
|
|
||||||
with TemporaryDirectory() as td:
|
with TemporaryDirectory() as tmpdir:
|
||||||
context = PdfContext(options, td, infile, None, None)
|
context = PdfContext(options, tmpdir, infile, None, None)
|
||||||
tmpout = Path(td) / 'out.pdf'
|
tmpout = Path(tmpdir) / 'out.pdf'
|
||||||
optimize(
|
optimize(
|
||||||
infile,
|
infile,
|
||||||
tmpout,
|
tmpout,
|
||||||
@@ -622,7 +712,7 @@ def main(infile, outfile, level, jobs=1):
|
|||||||
dict(
|
dict(
|
||||||
compress_streams=True,
|
compress_streams=True,
|
||||||
preserve_pdfa=True,
|
preserve_pdfa=True,
|
||||||
object_stream_mode=pikepdf.ObjectStreamMode.generate,
|
object_stream_mode=ObjectStreamMode.generate,
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
copy(fspath(tmpout), fspath(outfile))
|
copy(fspath(tmpout), fspath(outfile))
|
||||||
|
|||||||
+19
-14
@@ -9,21 +9,25 @@
|
|||||||
Utilities for PDF/A production and confirmation with Ghostspcript.
|
Utilities for PDF/A production and confirmation with Ghostspcript.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import base64
|
import base64
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Dict, Iterator, Union
|
from typing import Iterator
|
||||||
|
|
||||||
|
try:
|
||||||
|
from importlib.resources import files as package_files
|
||||||
|
except ImportError:
|
||||||
|
from importlib_resources import files as package_files # type: ignore
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
import pkg_resources
|
|
||||||
|
|
||||||
ICC_PROFILE_RELPATH = 'data/sRGB.icc'
|
SRGB_ICC_PROFILE_NAME = 'sRGB.icc'
|
||||||
|
|
||||||
SRGB_ICC_PROFILE = pkg_resources.resource_filename('ocrmypdf', ICC_PROFILE_RELPATH)
|
|
||||||
|
|
||||||
|
|
||||||
def _postscript_objdef(
|
def _postscript_objdef(
|
||||||
alias: str,
|
alias: str,
|
||||||
dictionary: Dict[str, str],
|
dictionary: dict[str, str],
|
||||||
*,
|
*,
|
||||||
stream_name: str = None,
|
stream_name: str = None,
|
||||||
stream_data: bytes = None,
|
stream_data: bytes = None,
|
||||||
@@ -95,19 +99,20 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
|||||||
target_filename: filename to save
|
target_filename: filename to save
|
||||||
icc: ICC identifier such as 'sRGB'
|
icc: ICC identifier such as 'sRGB'
|
||||||
References:
|
References:
|
||||||
Adobe PDFMARK Reference: https://www.adobe.com/content/dam/acom/en/devnet/acrobat/pdfs/pdfmark_reference.pdf
|
Adobe PDFMARK Reference:
|
||||||
|
https://www.adobe.com/content/dam/acom/en/devnet/acrobat/pdfs/pdfmark_reference.pdf
|
||||||
"""
|
"""
|
||||||
if icc == 'sRGB':
|
if icc != 'sRGB':
|
||||||
icc_profile = SRGB_ICC_PROFILE
|
|
||||||
else:
|
|
||||||
raise NotImplementedError("Only supporting sRGB")
|
raise NotImplementedError("Only supporting sRGB")
|
||||||
|
|
||||||
bytes_icc_profile = Path(icc_profile).read_bytes()
|
bytes_icc_profile = (
|
||||||
ps = '\n'.join(_make_postscript(icc, bytes_icc_profile, 3))
|
package_files('ocrmypdf.data') / SRGB_ICC_PROFILE_NAME
|
||||||
|
).read_bytes()
|
||||||
|
postscript = '\n'.join(_make_postscript(icc, bytes_icc_profile, 3))
|
||||||
|
|
||||||
# We should have encoded everything to pure ASCII by this point, and
|
# We should have encoded everything to pure ASCII by this point, and
|
||||||
# to be safe, only allow ASCII in PostScript
|
# to be safe, only allow ASCII in PostScript
|
||||||
Path(target_filename).write_text(ps, encoding='ascii')
|
Path(target_filename).write_text(postscript, encoding='ascii')
|
||||||
return target_filename
|
return target_filename
|
||||||
|
|
||||||
|
|
||||||
@@ -128,7 +133,7 @@ def file_claims_pdfa(filename: Path):
|
|||||||
}
|
}
|
||||||
valid_part_conforms = {'1A', '1B', '2A', '2B', '2U', '3A', '3B', '3U'}
|
valid_part_conforms = {'1A', '1B', '2A', '2B', '2U', '3A', '3B', '3U'}
|
||||||
conformance = f'PDF/A-{pdfmeta.pdfa_status}'
|
conformance = f'PDF/A-{pdfmeta.pdfa_status}'
|
||||||
pdfa_dict: Dict[str, Union[str, bool]] = {}
|
pdfa_dict: dict[str, str | bool] = {}
|
||||||
if pdfmeta.pdfa_status in valid_part_conforms:
|
if pdfmeta.pdfa_status in valid_part_conforms:
|
||||||
pdfa_dict['pass'] = True
|
pdfa_dict['pass'] = True
|
||||||
pdfa_dict['output'] = 'pdfa'
|
pdfa_dict['output'] = 'pdfa'
|
||||||
|
|||||||
@@ -6,4 +6,8 @@
|
|||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
|
"""For extracting information about PDFs prior to OCR."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PdfInfo
|
from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PdfInfo
|
||||||
|
|||||||
+176
-89
@@ -6,22 +6,42 @@
|
|||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
|
"""Extract information about the content of a PDF."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import atexit
|
import atexit
|
||||||
import logging
|
import logging
|
||||||
import re
|
import re
|
||||||
from collections import defaultdict, namedtuple
|
from collections import defaultdict
|
||||||
from contextlib import ExitStack
|
from contextlib import ExitStack
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from enum import Enum
|
from enum import Enum, auto
|
||||||
from functools import partial
|
from functools import partial
|
||||||
from math import hypot, inf, isclose
|
from math import hypot, inf, isclose
|
||||||
from os import PathLike
|
from os import PathLike
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Container, Iterator, Optional, Tuple, Union
|
from typing import (
|
||||||
|
Container,
|
||||||
|
Iterable,
|
||||||
|
Iterator,
|
||||||
|
Mapping,
|
||||||
|
NamedTuple,
|
||||||
|
Optional,
|
||||||
|
Sequence,
|
||||||
|
Tuple,
|
||||||
|
)
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
import pikepdf
|
from pikepdf import (
|
||||||
from pikepdf import Object, Pdf, PdfMatrix
|
Object,
|
||||||
|
Pdf,
|
||||||
|
PdfImage,
|
||||||
|
PdfInlineImage,
|
||||||
|
PdfMatrix,
|
||||||
|
UnsupportedImageTypeError,
|
||||||
|
parse_content_stream,
|
||||||
|
)
|
||||||
|
|
||||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||||
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
||||||
@@ -30,13 +50,41 @@ from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes
|
|||||||
|
|
||||||
logger = logging.getLogger()
|
logger = logging.getLogger()
|
||||||
|
|
||||||
Colorspace = Enum('Colorspace', 'gray rgb cmyk lab icc index sep devn pattern jpeg2000')
|
|
||||||
|
|
||||||
Encoding = Enum(
|
class Colorspace(Enum):
|
||||||
'Encoding', 'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate runlength'
|
"""Description of common image colorspaces in a PDF."""
|
||||||
)
|
|
||||||
|
|
||||||
FRIENDLY_COLORSPACE = {
|
# pylint: disable=invalid-name
|
||||||
|
gray = auto()
|
||||||
|
rgb = auto()
|
||||||
|
cmyk = auto()
|
||||||
|
lab = auto()
|
||||||
|
icc = auto()
|
||||||
|
index = auto()
|
||||||
|
sep = auto()
|
||||||
|
devn = auto()
|
||||||
|
pattern = auto()
|
||||||
|
jpeg2000 = auto()
|
||||||
|
|
||||||
|
|
||||||
|
class Encoding(Enum):
|
||||||
|
"""Description of common image encodings in a PDF."""
|
||||||
|
|
||||||
|
# pylint: disable=invalid-name
|
||||||
|
ccitt = auto()
|
||||||
|
jpeg = auto()
|
||||||
|
jpeg2000 = auto()
|
||||||
|
jbig2 = auto()
|
||||||
|
asciihex = auto()
|
||||||
|
ascii85 = auto()
|
||||||
|
lzw = auto()
|
||||||
|
flate = auto()
|
||||||
|
runlength = auto()
|
||||||
|
|
||||||
|
|
||||||
|
FloatRect = Tuple[float, float, float, float]
|
||||||
|
|
||||||
|
FRIENDLY_COLORSPACE: dict[str, Colorspace] = {
|
||||||
'/DeviceGray': Colorspace.gray,
|
'/DeviceGray': Colorspace.gray,
|
||||||
'/CalGray': Colorspace.gray,
|
'/CalGray': Colorspace.gray,
|
||||||
'/DeviceRGB': Colorspace.rgb,
|
'/DeviceRGB': Colorspace.rgb,
|
||||||
@@ -54,7 +102,7 @@ FRIENDLY_COLORSPACE = {
|
|||||||
'/I': Colorspace.index,
|
'/I': Colorspace.index,
|
||||||
}
|
}
|
||||||
|
|
||||||
FRIENDLY_ENCODING = {
|
FRIENDLY_ENCODING: dict[str, Encoding] = {
|
||||||
'/CCITTFaxDecode': Encoding.ccitt,
|
'/CCITTFaxDecode': Encoding.ccitt,
|
||||||
'/DCTDecode': Encoding.jpeg,
|
'/DCTDecode': Encoding.jpeg,
|
||||||
'/JPXDecode': Encoding.jpeg2000,
|
'/JPXDecode': Encoding.jpeg2000,
|
||||||
@@ -68,7 +116,7 @@ FRIENDLY_ENCODING = {
|
|||||||
'/RL': Encoding.runlength,
|
'/RL': Encoding.runlength,
|
||||||
}
|
}
|
||||||
|
|
||||||
FRIENDLY_COMP = {
|
FRIENDLY_COMP: dict[Colorspace, int] = {
|
||||||
Colorspace.gray: 1,
|
Colorspace.gray: 1,
|
||||||
Colorspace.rgb: 3,
|
Colorspace.rgb: 3,
|
||||||
Colorspace.cmyk: 4,
|
Colorspace.cmyk: 4,
|
||||||
@@ -86,24 +134,46 @@ def _is_unit_square(shorthand):
|
|||||||
return all(isclose(a, b, rel_tol=1e-3) for a, b in pairwise)
|
return all(isclose(a, b, rel_tol=1e-3) for a, b in pairwise)
|
||||||
|
|
||||||
|
|
||||||
XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_depth'])
|
class XobjectSettings(NamedTuple):
|
||||||
|
"""Info about an XObject found in a PDF."""
|
||||||
|
|
||||||
InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth'])
|
name: str
|
||||||
|
shorthand: tuple[float, float, float, float, float, float]
|
||||||
|
stack_depth: int
|
||||||
|
|
||||||
ContentsInfo = namedtuple(
|
|
||||||
'ContentsInfo',
|
|
||||||
['xobject_settings', 'inline_images', 'found_vector', 'found_text', 'name_index'],
|
|
||||||
)
|
|
||||||
|
|
||||||
TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt'])
|
class InlineSettings(NamedTuple):
|
||||||
|
"""Info about an inline image found in a PDF."""
|
||||||
|
|
||||||
|
iimage: PdfInlineImage
|
||||||
|
shorthand: tuple[float, float, float, float, float, float]
|
||||||
|
stack_depth: int
|
||||||
|
|
||||||
|
|
||||||
|
class ContentsInfo(NamedTuple):
|
||||||
|
"""Info about various objects found in a PDF."""
|
||||||
|
|
||||||
|
xobject_settings: list[XobjectSettings]
|
||||||
|
inline_images: list[InlineSettings]
|
||||||
|
found_vector: bool
|
||||||
|
found_text: bool
|
||||||
|
name_index: Mapping[str, list[XobjectSettings]]
|
||||||
|
|
||||||
|
|
||||||
|
class TextboxInfo(NamedTuple):
|
||||||
|
"""Info about a text box found in a PDF."""
|
||||||
|
|
||||||
|
bbox: tuple[float, float, float, float]
|
||||||
|
is_visible: bool
|
||||||
|
is_corrupt: bool
|
||||||
|
|
||||||
|
|
||||||
class VectorMarker:
|
class VectorMarker:
|
||||||
pass
|
"""Sentinel indicating vector drawing operations were found on a page."""
|
||||||
|
|
||||||
|
|
||||||
class TextMarker:
|
class TextMarker:
|
||||||
pass
|
"""Sentinel indicating text drawing operations were found on a page."""
|
||||||
|
|
||||||
|
|
||||||
def _normalize_stack(graphobjs):
|
def _normalize_stack(graphobjs):
|
||||||
@@ -146,8 +216,8 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
|||||||
|
|
||||||
stack = []
|
stack = []
|
||||||
ctm = PdfMatrix(initial_shorthand)
|
ctm = PdfMatrix(initial_shorthand)
|
||||||
xobject_settings = []
|
xobject_settings: list[XobjectSettings] = []
|
||||||
inline_images = []
|
inline_images: list[InlineSettings] = []
|
||||||
name_index = defaultdict(lambda: [])
|
name_index = defaultdict(lambda: [])
|
||||||
found_vector = False
|
found_vector = False
|
||||||
found_text = False
|
found_text = False
|
||||||
@@ -157,9 +227,7 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
|||||||
operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops)
|
operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops)
|
||||||
|
|
||||||
for n, graphobj in enumerate(
|
for n, graphobj in enumerate(
|
||||||
_normalize_stack(
|
_normalize_stack(parse_content_stream(contentstream, operator_whitelist))
|
||||||
pikepdf.parse_content_stream(contentstream, operator_whitelist)
|
|
||||||
)
|
|
||||||
):
|
):
|
||||||
operands, operator = graphobj
|
operands, operator = graphobj
|
||||||
if operator == 'q':
|
if operator == 'q':
|
||||||
@@ -167,7 +235,7 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
|||||||
if len(stack) > 32: # See docstring
|
if len(stack) > 32: # See docstring
|
||||||
if len(stack) > 128:
|
if len(stack) > 128:
|
||||||
raise RuntimeError(
|
raise RuntimeError(
|
||||||
"PDF graphics stack overflowed hard limit, operator %i" % n
|
f"PDF graphics stack overflowed hard limit at operator {n}"
|
||||||
)
|
)
|
||||||
warn("PDF graphics stack overflowed spec limit")
|
warn("PDF graphics stack overflowed spec limit")
|
||||||
elif operator == 'Q':
|
elif operator == 'Q':
|
||||||
@@ -185,7 +253,7 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
|||||||
name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack)
|
name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack)
|
||||||
)
|
)
|
||||||
xobject_settings.append(settings)
|
xobject_settings.append(settings)
|
||||||
name_index[image_name].append(settings)
|
name_index[str(image_name)].append(settings)
|
||||||
elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this
|
elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this
|
||||||
iimage = operands[0]
|
iimage = operands[0]
|
||||||
inline = InlineSettings(
|
inline = InlineSettings(
|
||||||
@@ -253,7 +321,7 @@ def _get_dpi(ctm_shorthand, image_size) -> Resolution:
|
|||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
a, b, c, d, _, _ = ctm_shorthand
|
a, b, c, d, _, _ = ctm_shorthand # pylint: disable=invalid-name
|
||||||
|
|
||||||
# Calculate the width and height of the image in PDF units
|
# Calculate the width and height of the image in PDF units
|
||||||
image_drawn = hypot(a, b), hypot(c, d)
|
image_drawn = hypot(a, b), hypot(c, d)
|
||||||
@@ -269,25 +337,32 @@ def _get_dpi(ctm_shorthand, image_size) -> Resolution:
|
|||||||
|
|
||||||
|
|
||||||
class ImageInfo:
|
class ImageInfo:
|
||||||
|
"""Information about an image found in a PDF."""
|
||||||
|
|
||||||
DPI_PREC = Decimal('1.000')
|
DPI_PREC = Decimal('1.000')
|
||||||
|
|
||||||
|
_comp: int | None
|
||||||
|
_name: str
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
name='',
|
name='',
|
||||||
pdfimage: Optional[Object] = None,
|
pdfimage: Object | None = None,
|
||||||
inline: Optional[Object] = None,
|
inline: PdfInlineImage | None = None,
|
||||||
shorthand=None,
|
shorthand=None,
|
||||||
):
|
):
|
||||||
self._name = str(name)
|
self._name = str(name)
|
||||||
self._shorthand = shorthand
|
self._shorthand = shorthand
|
||||||
|
|
||||||
|
pim: PdfInlineImage | PdfImage
|
||||||
|
|
||||||
if inline is not None:
|
if inline is not None:
|
||||||
self._origin = 'inline'
|
self._origin = 'inline'
|
||||||
pim = inline.iimage
|
pim = inline
|
||||||
elif pdfimage is not None:
|
elif pdfimage is not None:
|
||||||
self._origin = 'xobject'
|
self._origin = 'xobject'
|
||||||
pim = pikepdf.PdfImage(pdfimage)
|
pim = PdfImage(pdfimage)
|
||||||
else:
|
else:
|
||||||
raise ValueError("Either pdfimage or inline must be set")
|
raise ValueError("Either pdfimage or inline must be set")
|
||||||
self._width = pim.width
|
self._width = pim.width
|
||||||
@@ -303,32 +378,42 @@ class ImageInfo:
|
|||||||
|
|
||||||
self._bpc = int(pim.bits_per_component)
|
self._bpc = int(pim.bits_per_component)
|
||||||
try:
|
try:
|
||||||
self._enc = FRIENDLY_ENCODING.get(pim.filters[0], 'image')
|
self._enc = FRIENDLY_ENCODING.get(pim.filters[0])
|
||||||
except IndexError:
|
except IndexError:
|
||||||
self._enc = '?'
|
self._enc = None
|
||||||
|
|
||||||
try:
|
try:
|
||||||
self._color = FRIENDLY_COLORSPACE.get(pim.colorspace, '?')
|
self._color = FRIENDLY_COLORSPACE.get(pim.colorspace or '')
|
||||||
except NotImplementedError:
|
except NotImplementedError:
|
||||||
self._color = '?'
|
self._color = None
|
||||||
if self._enc == Encoding.jpeg2000:
|
if self._enc == Encoding.jpeg2000:
|
||||||
self._color = Colorspace.jpeg2000
|
self._color = Colorspace.jpeg2000
|
||||||
|
|
||||||
if self._color == Colorspace.icc:
|
if self._color == Colorspace.icc:
|
||||||
# Check the ICC profile to determine actual colorspace
|
# Check the ICC profile to determine actual colorspace
|
||||||
pim_icc = pim.icc
|
try:
|
||||||
if pim_icc.profile.xcolor_space == 'GRAY':
|
pim_icc = pim.icc
|
||||||
self._comp = 1
|
if pim_icc.profile.xcolor_space == 'GRAY':
|
||||||
elif pim_icc.profile.xcolor_space == 'CMYK':
|
self._comp = 1
|
||||||
self._comp = 4
|
elif pim_icc.profile.xcolor_space == 'CMYK':
|
||||||
else:
|
self._comp = 4
|
||||||
self._comp = 3
|
else:
|
||||||
|
self._comp = 3
|
||||||
|
except (AttributeError, UnsupportedImageTypeError) as ex:
|
||||||
|
self._comp = None
|
||||||
|
logger.warning(
|
||||||
|
f"An image with a corrupt or unreadable ICC profile was found. "
|
||||||
|
f"The output PDF may not match the input PDF visually: {ex}. {self}"
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
self._comp = FRIENDLY_COMP.get(self._color, '?')
|
if isinstance(self._color, Colorspace):
|
||||||
|
self._comp = FRIENDLY_COMP.get(self._color)
|
||||||
|
else:
|
||||||
|
self._comp = None
|
||||||
|
|
||||||
# Bit of a hack... infer grayscale if component count is uncertain
|
# Bit of a hack... infer grayscale if component count is uncertain
|
||||||
# but encoding only supports monochrome.
|
# but encoding only supports monochrome.
|
||||||
if self._comp == '?' and self._enc in (Encoding.ccitt, Encoding.jbig2):
|
if self._comp is None and self._enc in (Encoding.ccitt, Encoding.jbig2):
|
||||||
self._comp = FRIENDLY_COMP[Colorspace.gray]
|
self._comp = FRIENDLY_COMP[Colorspace.gray]
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -353,15 +438,15 @@ class ImageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def color(self):
|
def color(self):
|
||||||
return self._color
|
return self._color if self._color is not None else '?'
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def comp(self):
|
def comp(self):
|
||||||
return self._comp
|
return self._comp if self._comp is not None else '?'
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def enc(self):
|
def enc(self):
|
||||||
return self._enc
|
return self._enc if self._enc is not None else 'image'
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def renderable(self):
|
def renderable(self):
|
||||||
@@ -372,15 +457,10 @@ class ImageInfo:
|
|||||||
return _get_dpi(self._shorthand, (self._width, self._height))
|
return _get_dpi(self._shorthand, (self._width, self._height))
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
class_locals = {
|
|
||||||
attr: getattr(self, attr, None)
|
|
||||||
for attr in dir(self)
|
|
||||||
if not attr.startswith('_')
|
|
||||||
}
|
|
||||||
return (
|
return (
|
||||||
"<ImageInfo '{name}' {type_} {width}x{height} {color} "
|
f"<ImageInfo '{self.name}' {self.type_} {self.width}x{self.height} "
|
||||||
"{comp} {bpc} {enc} {dpi}>"
|
f"{self.color} {self.comp} {self.bpc} {self.enc} {self.dpi}>"
|
||||||
).format(**class_locals)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
||||||
@@ -388,11 +468,11 @@ def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
|||||||
|
|
||||||
for n, inline in enumerate(contentsinfo.inline_images):
|
for n, inline in enumerate(contentsinfo.inline_images):
|
||||||
yield ImageInfo(
|
yield ImageInfo(
|
||||||
name='inline-%02d' % n, shorthand=inline.shorthand, inline=inline
|
name=f'inline-{n:02d}', shorthand=inline.shorthand, inline=inline.iimage
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _image_xobjects(container) -> Iterator[Tuple[Object, str]]:
|
def _image_xobjects(container) -> Iterator[tuple[Object, str]]:
|
||||||
"""Search for all XObject-based images in the container
|
"""Search for all XObject-based images in the container
|
||||||
|
|
||||||
Usually the container is a page, but it could also be a Form XObject
|
Usually the container is a page, but it could also be a Form XObject
|
||||||
@@ -413,7 +493,7 @@ def _image_xobjects(container) -> Iterator[Tuple[Object, str]]:
|
|||||||
xobjs = resources['/XObject'].as_dict()
|
xobjs = resources['/XObject'].as_dict()
|
||||||
for xobj in xobjs:
|
for xobj in xobjs:
|
||||||
candidate: Object = xobjs[xobj]
|
candidate: Object = xobjs[xobj]
|
||||||
if not '/Subtype' in candidate:
|
if '/Subtype' not in candidate:
|
||||||
continue
|
continue
|
||||||
if candidate['/Subtype'] == '/Image':
|
if candidate['/Subtype'] == '/Image':
|
||||||
pdfimage = candidate
|
pdfimage = candidate
|
||||||
@@ -480,7 +560,7 @@ def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: Content
|
|||||||
|
|
||||||
def _process_content_streams(
|
def _process_content_streams(
|
||||||
*, pdf: Pdf, container: Object, shorthand=None
|
*, pdf: Pdf, container: Object, shorthand=None
|
||||||
) -> Iterator[Union[VectorMarker, TextMarker, ImageInfo]]:
|
) -> Iterator[VectorMarker | TextMarker | ImageInfo]:
|
||||||
"""Find all individual instances of images drawn in the container
|
"""Find all individual instances of images drawn in the container
|
||||||
|
|
||||||
Usually the container is a page, but it may also be a Form XObject.
|
Usually the container is a page, but it may also be a Form XObject.
|
||||||
@@ -529,10 +609,10 @@ def _process_content_streams(
|
|||||||
yield from _find_form_xobject_images(pdf, container, contentsinfo)
|
yield from _find_form_xobject_images(pdf, container, contentsinfo)
|
||||||
|
|
||||||
|
|
||||||
def _page_has_text(text_blocks, page_width, page_height) -> bool:
|
def _page_has_text(text_blocks: Iterable[FloatRect], page_width, page_height) -> bool:
|
||||||
"""Smarter text detection that ignores text in margins"""
|
"""Smarter text detection that ignores text in margins"""
|
||||||
|
|
||||||
pw, ph = float(page_width), float(page_height)
|
pw, ph = float(page_width), float(page_height) # pylint: disable=invalid-name
|
||||||
|
|
||||||
margin_ratio = 0.125
|
margin_ratio = 0.125
|
||||||
interior_bbox = (
|
interior_bbox = (
|
||||||
@@ -542,7 +622,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
|
|||||||
margin_ratio * ph, # bottom (first quadrant: bottom < top)
|
margin_ratio * ph, # bottom (first quadrant: bottom < top)
|
||||||
)
|
)
|
||||||
|
|
||||||
def rects_intersect(a, b) -> bool:
|
def rects_intersect(a: FloatRect, b: FloatRect) -> bool:
|
||||||
"""
|
"""
|
||||||
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
|
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
|
||||||
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
|
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
|
||||||
@@ -564,26 +644,26 @@ def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]:
|
|||||||
We do this to save memory and ensure that our objects are pickleable.
|
We do this to save memory and ensure that our objects are pickleable.
|
||||||
"""
|
"""
|
||||||
for box in textbox_getter(miner):
|
for box in textbox_getter(miner):
|
||||||
first_line = box._objs[0]
|
first_line = box._objs[0] # pylint: disable=protected-access
|
||||||
first_char = first_line._objs[0]
|
first_char = first_line._objs[0] # pylint: disable=protected-access
|
||||||
|
|
||||||
visible = first_char.rendermode != 3
|
visible = first_char.rendermode != 3
|
||||||
corrupt = first_char.get_text() == '\ufffd'
|
corrupt = first_char.get_text() == '\ufffd'
|
||||||
yield TextboxInfo(box.bbox, visible, corrupt)
|
yield TextboxInfo(box.bbox, visible, corrupt)
|
||||||
|
|
||||||
|
|
||||||
worker_pdf = None
|
worker_pdf = None # pylint: disable=invalid-name
|
||||||
|
|
||||||
|
|
||||||
def _pdf_pageinfo_sync_init(pdf: Pdf, infile: Path, pdfminer_loglevel):
|
def _pdf_pageinfo_sync_init(pdf: Pdf, infile: Path, pdfminer_loglevel):
|
||||||
global worker_pdf # pylint: disable=global-statement
|
global worker_pdf # pylint: disable=global-statement,invalid-name
|
||||||
pikepdf_enable_mmap()
|
pikepdf_enable_mmap()
|
||||||
|
|
||||||
logging.getLogger('pdfminer').setLevel(pdfminer_loglevel)
|
logging.getLogger('pdfminer').setLevel(pdfminer_loglevel)
|
||||||
|
|
||||||
# If the pdf is not opened, open a copy for our worker process to use
|
# If the pdf is not opened, open a copy for our worker process to use
|
||||||
if pdf is None:
|
if pdf is None:
|
||||||
worker_pdf = pikepdf.open(infile)
|
worker_pdf = Pdf.open(infile)
|
||||||
|
|
||||||
def on_process_close():
|
def on_process_close():
|
||||||
worker_pdf.close()
|
worker_pdf.close()
|
||||||
@@ -597,7 +677,7 @@ def _pdf_pageinfo_sync(args):
|
|||||||
pdf = thread_pdf if thread_pdf is not None else worker_pdf
|
pdf = thread_pdf if thread_pdf is not None else worker_pdf
|
||||||
with ExitStack() as stack:
|
with ExitStack() as stack:
|
||||||
if not pdf: # When called with SerialExecutor
|
if not pdf: # When called with SerialExecutor
|
||||||
pdf = stack.enter_context(pikepdf.open(infile))
|
pdf = stack.enter_context(Pdf.open(infile))
|
||||||
page = PageInfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
page = PageInfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
||||||
return page
|
return page
|
||||||
|
|
||||||
@@ -610,8 +690,8 @@ def _pdf_pageinfo_concurrent(
|
|||||||
max_workers,
|
max_workers,
|
||||||
check_pages,
|
check_pages,
|
||||||
detailed_analysis=False,
|
detailed_analysis=False,
|
||||||
):
|
) -> Sequence[PageInfo | None]:
|
||||||
pages = [None] * len(pdf.pages)
|
pages: Sequence[PageInfo | None] = [None] * len(pdf.pages)
|
||||||
|
|
||||||
def update_pageinfo(result, pbar):
|
def update_pageinfo(result, pbar):
|
||||||
page = result
|
page = result
|
||||||
@@ -661,6 +741,12 @@ def _pdf_pageinfo_concurrent(
|
|||||||
|
|
||||||
|
|
||||||
class PageInfo:
|
class PageInfo:
|
||||||
|
"""Information about type of contents on each page in a PDF."""
|
||||||
|
|
||||||
|
_has_text: bool | None
|
||||||
|
_has_vector: bool | None
|
||||||
|
_images: list[ImageInfo]
|
||||||
|
|
||||||
def __init__(
|
def __init__(
|
||||||
self,
|
self,
|
||||||
pdf: Pdf,
|
pdf: Pdf,
|
||||||
@@ -718,21 +804,21 @@ class PageInfo:
|
|||||||
self._has_vector = False
|
self._has_vector = False
|
||||||
self._has_text = False
|
self._has_text = False
|
||||||
self._images = []
|
self._images = []
|
||||||
for ci in _process_content_streams(
|
for info in _process_content_streams(
|
||||||
pdf=pdf, container=page, shorthand=userunit_shorthand
|
pdf=pdf, container=page, shorthand=userunit_shorthand
|
||||||
):
|
):
|
||||||
if isinstance(ci, VectorMarker):
|
if isinstance(info, VectorMarker):
|
||||||
self._has_vector = True
|
self._has_vector = True
|
||||||
elif isinstance(ci, TextMarker):
|
elif isinstance(info, TextMarker):
|
||||||
self._has_text = True
|
self._has_text = True
|
||||||
elif isinstance(ci, ImageInfo):
|
elif isinstance(info, ImageInfo):
|
||||||
self._images.append(ci)
|
self._images.append(info)
|
||||||
else:
|
else:
|
||||||
raise NotImplementedError()
|
raise NotImplementedError()
|
||||||
else:
|
else:
|
||||||
self._has_vector = None # i.e. "no information"
|
self._has_vector = None # i.e. "no information"
|
||||||
self._has_text = None
|
self._has_text = None
|
||||||
self._images = None
|
self._images = []
|
||||||
|
|
||||||
self._dpi = None
|
self._dpi = None
|
||||||
if self._images:
|
if self._images:
|
||||||
@@ -749,7 +835,7 @@ class PageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def has_text(self) -> bool:
|
def has_text(self) -> bool:
|
||||||
return self._has_text
|
return bool(self._has_text)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_corrupt_text(self) -> bool:
|
def has_corrupt_text(self) -> bool:
|
||||||
@@ -759,7 +845,7 @@ class PageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def has_vector(self) -> bool:
|
def has_vector(self) -> bool:
|
||||||
return self._has_vector
|
return bool(self._has_vector)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def width_inches(self) -> Decimal:
|
def width_inches(self) -> Decimal:
|
||||||
@@ -792,9 +878,7 @@ class PageInfo:
|
|||||||
def images(self):
|
def images(self):
|
||||||
return self._images
|
return self._images
|
||||||
|
|
||||||
def get_textareas(
|
def get_textareas(self, visible: bool | None = None, corrupt: bool | None = None):
|
||||||
self, visible: Optional[bool] = None, corrupt: Optional[bool] = None
|
|
||||||
):
|
|
||||||
def predicate(obj, want_visible, want_corrupt):
|
def predicate(obj, want_visible, want_corrupt):
|
||||||
result = True
|
result = True
|
||||||
if want_visible is not None:
|
if want_visible is not None:
|
||||||
@@ -837,6 +921,9 @@ class PageInfo:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
DEFAULT_EXECUTOR = SerialExecutor()
|
||||||
|
|
||||||
|
|
||||||
class PdfInfo:
|
class PdfInfo:
|
||||||
"""Get summary information about a PDF"""
|
"""Get summary information about a PDF"""
|
||||||
|
|
||||||
@@ -848,13 +935,13 @@ class PdfInfo:
|
|||||||
progbar: bool = False,
|
progbar: bool = False,
|
||||||
max_workers: int = None,
|
max_workers: int = None,
|
||||||
check_pages=None,
|
check_pages=None,
|
||||||
executor: Executor = SerialExecutor(),
|
executor: Executor = DEFAULT_EXECUTOR,
|
||||||
):
|
):
|
||||||
self._infile = infile
|
self._infile = infile
|
||||||
if check_pages is None:
|
if check_pages is None:
|
||||||
check_pages = range(0, 1_000_000_000)
|
check_pages = range(0, 1_000_000_000)
|
||||||
|
|
||||||
with pikepdf.open(infile) as pdf:
|
with Pdf.open(infile) as pdf:
|
||||||
if pdf.is_encrypted:
|
if pdf.is_encrypted:
|
||||||
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
||||||
self._pages = _pdf_pageinfo_concurrent(
|
self._pages = _pdf_pageinfo_concurrent(
|
||||||
@@ -875,24 +962,24 @@ class PdfInfo:
|
|||||||
self._has_acroform = True
|
self._has_acroform = True
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def pages(self):
|
def pages(self) -> Sequence[PageInfo | None]:
|
||||||
return self._pages
|
return self._pages
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def min_version(self) -> str:
|
def min_version(self) -> str:
|
||||||
# The minimum PDF is the maximum version that any particular page needs
|
# The minimum PDF is the maximum version that any particular page needs
|
||||||
return max(page.min_version for page in self.pages)
|
return max(page.min_version for page in self.pages if page)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_userunit(self) -> bool:
|
def has_userunit(self) -> bool:
|
||||||
return any(page.userunit != 1.0 for page in self.pages)
|
return any(page.userunit != 1.0 for page in self.pages if page)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def has_acroform(self) -> bool:
|
def has_acroform(self) -> bool:
|
||||||
return self._has_acroform
|
return self._has_acroform
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def filename(self) -> Union[str, Path]:
|
def filename(self) -> str | Path:
|
||||||
if not isinstance(self._infile, (str, Path)):
|
if not isinstance(self._infile, (str, Path)):
|
||||||
raise NotImplementedError("can't get filename from stream")
|
raise NotImplementedError("can't get filename from stream")
|
||||||
return self._infile
|
return self._infile
|
||||||
|
|||||||
@@ -5,6 +5,8 @@
|
|||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import re
|
import re
|
||||||
from math import copysign
|
from math import copysign
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -135,7 +137,7 @@ class LTStateAwareChar(LTChar):
|
|||||||
return self._text
|
return self._text
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
return '<%s %s matrix=%s rendermode=%r font=%r adv=%s text=%r>' % (
|
return '<{} {} matrix={} rendermode={!r} font={!r} adv={} text={!r}>'.format(
|
||||||
self.__class__.__name__,
|
self.__class__.__name__,
|
||||||
bbox2str(self.bbox),
|
bbox2str(self.bbox),
|
||||||
matrix2str(self.matrix),
|
matrix2str(self.matrix),
|
||||||
|
|||||||
+132
-31
@@ -4,17 +4,19 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
"""OCRmyPDF pluggy plugin specification."""
|
||||||
|
|
||||||
from abc import ABC, abstractmethod, abstractstaticmethod
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from abc import ABC, abstractmethod
|
||||||
from argparse import ArgumentParser, Namespace
|
from argparse import ArgumentParser, Namespace
|
||||||
from collections import namedtuple
|
|
||||||
from logging import Handler
|
from logging import Handler
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import TYPE_CHECKING, AbstractSet, List, Optional
|
from typing import TYPE_CHECKING, AbstractSet, NamedTuple, Sequence
|
||||||
|
|
||||||
import pluggy
|
import pluggy
|
||||||
|
|
||||||
from ocrmypdf._concurrent import Executor
|
from ocrmypdf import Executor, PdfContext
|
||||||
from ocrmypdf.helpers import Resolution
|
from ocrmypdf.helpers import Resolution
|
||||||
|
|
||||||
if TYPE_CHECKING:
|
if TYPE_CHECKING:
|
||||||
@@ -43,6 +45,33 @@ def get_logging_console() -> Handler:
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec
|
||||||
|
def initialize(plugin_manager: pluggy.PluginManager):
|
||||||
|
"""Called when this plugin is first loaded into OCRmyPDF.
|
||||||
|
|
||||||
|
The primary intended use of this is for plugins to check compatibility with other
|
||||||
|
plugins and possibly block other blocks, a plugin that wishes to block ocrmypdf's
|
||||||
|
built-in optimize plugin could do:
|
||||||
|
|
||||||
|
.. code-block::
|
||||||
|
|
||||||
|
plugin_manager.set_blocked('ocrmypdf.builtin_plugins.optimize')
|
||||||
|
|
||||||
|
It would also be reasonable for an plugin implementation to check if it is unable
|
||||||
|
to proceed, for example, because a required dependency is missing. (If the plugin's
|
||||||
|
ability to proceed depends on options and arguments, use ``validate`` instead.)
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ocrmypdf.exceptions.ExitCodeException: If options are not acceptable
|
||||||
|
and the application should terminate gracefully with an informative
|
||||||
|
message and error code.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This hook will be called from the main process, and may modify global state
|
||||||
|
before child worker processes are forked.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
@hookspec
|
@hookspec
|
||||||
def add_options(parser: ArgumentParser) -> None:
|
def add_options(parser: ArgumentParser) -> None:
|
||||||
"""Allows the plugin to add its own command line and API arguments.
|
"""Allows the plugin to add its own command line and API arguments.
|
||||||
@@ -133,6 +162,7 @@ def get_progressbar_class():
|
|||||||
Here is how OCRmyPDF will use the progress bar:
|
Here is how OCRmyPDF will use the progress bar:
|
||||||
|
|
||||||
Example:
|
Example:
|
||||||
|
|
||||||
pbar_class = pm.hook.get_progressbar_class()
|
pbar_class = pm.hook.get_progressbar_class()
|
||||||
with pbar_class(**tqdm_kwargs) as pbar:
|
with pbar_class(**tqdm_kwargs) as pbar:
|
||||||
...
|
...
|
||||||
@@ -141,7 +171,7 @@ def get_progressbar_class():
|
|||||||
|
|
||||||
|
|
||||||
@hookspec
|
@hookspec
|
||||||
def validate(pdfinfo: 'PdfInfo', options: Namespace) -> None:
|
def validate(pdfinfo: PdfInfo, options: Namespace) -> None:
|
||||||
"""Called to give a plugin an opportunity to review *options* and *pdfinfo*.
|
"""Called to give a plugin an opportunity to review *options* and *pdfinfo*.
|
||||||
|
|
||||||
*options* contains the "work order" to process a particular file. *pdfinfo*
|
*options* contains the "work order" to process a particular file. *pdfinfo*
|
||||||
@@ -167,8 +197,8 @@ def rasterize_pdf_page(
|
|||||||
raster_device: str,
|
raster_device: str,
|
||||||
raster_dpi: Resolution,
|
raster_dpi: Resolution,
|
||||||
pageno: int,
|
pageno: int,
|
||||||
page_dpi: Optional[Resolution],
|
page_dpi: Resolution | None,
|
||||||
rotation: Optional[int],
|
rotation: int | None,
|
||||||
filter_vector: bool,
|
filter_vector: bool,
|
||||||
) -> Path:
|
) -> Path:
|
||||||
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units.
|
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units.
|
||||||
@@ -197,7 +227,7 @@ def rasterize_pdf_page(
|
|||||||
|
|
||||||
|
|
||||||
@hookspec(firstresult=True)
|
@hookspec(firstresult=True)
|
||||||
def filter_ocr_image(page: 'PageContext', image: 'Image') -> 'Image':
|
def filter_ocr_image(page: PageContext, image: Image.Image) -> Image.Image:
|
||||||
"""Called to filter the image before it is sent to OCR.
|
"""Called to filter the image before it is sent to OCR.
|
||||||
|
|
||||||
This is the image that OCR sees, not what the user sees when they view the
|
This is the image that OCR sees, not what the user sees when they view the
|
||||||
@@ -224,7 +254,7 @@ def filter_ocr_image(page: 'PageContext', image: 'Image') -> 'Image':
|
|||||||
|
|
||||||
|
|
||||||
@hookspec(firstresult=True)
|
@hookspec(firstresult=True)
|
||||||
def filter_page_image(page: 'PageContext', image_filename: Path) -> Path:
|
def filter_page_image(page: PageContext, image_filename: Path) -> Path:
|
||||||
"""Called to filter the whole page before it is inserted into the PDF.
|
"""Called to filter the whole page before it is inserted into the PDF.
|
||||||
|
|
||||||
A whole page image is only produced when preprocessing command line arguments
|
A whole page image is only produced when preprocessing command line arguments
|
||||||
@@ -236,9 +266,9 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path:
|
|||||||
``image_filename``. The hook may overwrite ``image_filename`` with a new file.
|
``image_filename``. The hook may overwrite ``image_filename`` with a new file.
|
||||||
|
|
||||||
The output image should preserve the same physical unit dimensions, that is
|
The output image should preserve the same physical unit dimensions, that is
|
||||||
(width * dpi_x, height * dpi_y). That is, if the image is resized, the DPI
|
``(width * dpi_x, height * dpi_y)``. That is, if the image is resized, the DPI
|
||||||
must be adjusted by the reciprocal. If this is not preserved, the PDF page
|
must be adjusted by the reciprocal. If this is not preserved, the PDF page
|
||||||
will be resized and the OCR layer misaligned. OCRmyPDF does not nothing
|
will be resized and the OCR layer misaligned. OCRmyPDF does nothing
|
||||||
to enforce these constraints; it is up to the plugin to do sensible things.
|
to enforce these constraints; it is up to the plugin to do sensible things.
|
||||||
|
|
||||||
OCRmyPDF will create the PDF page based on the image format used (unless the
|
OCRmyPDF will create the PDF page based on the image format used (unless the
|
||||||
@@ -264,9 +294,7 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path:
|
|||||||
|
|
||||||
|
|
||||||
@hookspec(firstresult=True)
|
@hookspec(firstresult=True)
|
||||||
def filter_pdf_page(
|
def filter_pdf_page(page: PageContext, image_filename: Path, output_pdf: Path) -> Path:
|
||||||
page: 'PageContext', image_filename: Path, output_pdf: Path
|
|
||||||
) -> Path:
|
|
||||||
"""Called to convert a filtered whole page image into a PDF.
|
"""Called to convert a filtered whole page image into a PDF.
|
||||||
|
|
||||||
A whole page image is only produced when preprocessing command line arguments
|
A whole page image is only produced when preprocessing command line arguments
|
||||||
@@ -307,15 +335,18 @@ def filter_pdf_page(
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence'))
|
class OrientationConfidence(NamedTuple):
|
||||||
"""Expresses an OCR engine's confidence in page rotation.
|
"""Expresses an OCR engine's confidence in page rotation.
|
||||||
|
|
||||||
Attributes:
|
Attributes:
|
||||||
angle (int): The clockwise angle (0, 90, 180, 270) that the page should be
|
angle: The clockwise angle (0, 90, 180, 270) that the page should be
|
||||||
rotated. 0 means no rotation.
|
rotated. 0 means no rotation.
|
||||||
confidence (float): How confident the OCR engine is that this the correct
|
confidence: How confident the OCR engine is that this the correct
|
||||||
rotation. 0 is not confident, 15 is very confident. Arbitrary units.
|
rotation. 0 is not confident, 15 is very confident. Arbitrary units.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
angle: int
|
||||||
|
confidence: float
|
||||||
|
|
||||||
|
|
||||||
class OcrEngine(ABC):
|
class OcrEngine(ABC):
|
||||||
@@ -325,11 +356,13 @@ class OcrEngine(ABC):
|
|||||||
Tesseract OCR.
|
Tesseract OCR.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@abstractstaticmethod
|
@staticmethod
|
||||||
|
@abstractmethod
|
||||||
def version() -> str:
|
def version() -> str:
|
||||||
"""Returns the version of the OCR engine."""
|
"""Returns the version of the OCR engine."""
|
||||||
|
|
||||||
@abstractstaticmethod
|
@staticmethod
|
||||||
|
@abstractmethod
|
||||||
def creator_tag(options: Namespace) -> str:
|
def creator_tag(options: Namespace) -> str:
|
||||||
"""Returns the creator tag to identify this software's role in creating the PDF.
|
"""Returns the creator tag to identify this software's role in creating the PDF.
|
||||||
|
|
||||||
@@ -349,24 +382,33 @@ class OcrEngine(ABC):
|
|||||||
to the user, usually in an error message.
|
to the user, usually in an error message.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@abstractstaticmethod
|
@staticmethod
|
||||||
|
@abstractmethod
|
||||||
def languages(options: Namespace) -> AbstractSet[str]:
|
def languages(options: Namespace) -> AbstractSet[str]:
|
||||||
"""Returns the set of all languages that are supported by the engine.
|
"""Returns the set of all languages that are supported by the engine.
|
||||||
|
|
||||||
Languages are typically given in 3-letter ISO 3166-1 codes, but actually
|
Languages are typically given in 3-letter ISO 3166-1 codes, but actually
|
||||||
can be any value understood by the OCR engine."""
|
can be any value understood by the OCR engine."""
|
||||||
|
|
||||||
@abstractstaticmethod
|
@staticmethod
|
||||||
|
@abstractmethod
|
||||||
def get_orientation(input_file: Path, options: Namespace) -> OrientationConfidence:
|
def get_orientation(input_file: Path, options: Namespace) -> OrientationConfidence:
|
||||||
"""Returns the orientation of the image."""
|
"""Returns the orientation of the image."""
|
||||||
|
|
||||||
@abstractstaticmethod
|
@staticmethod
|
||||||
|
def get_deskew(input_file: Path, options: Namespace) -> float:
|
||||||
|
"""Returns the deskew angle of the image, in degrees."""
|
||||||
|
return 0.0
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
@abstractmethod
|
||||||
def generate_hocr(
|
def generate_hocr(
|
||||||
input_file: Path, output_hocr: Path, output_text: Path, options: Namespace
|
input_file: Path, output_hocr: Path, output_text: Path, options: Namespace
|
||||||
) -> None:
|
) -> None:
|
||||||
"""Called to produce a hOCR file and sidecar text file."""
|
"""Called to produce a hOCR file and sidecar text file."""
|
||||||
|
|
||||||
@abstractstaticmethod
|
@staticmethod
|
||||||
|
@abstractmethod
|
||||||
def generate_pdf(
|
def generate_pdf(
|
||||||
input_file: Path, output_pdf: Path, output_text: Path, options: Namespace
|
input_file: Path, output_pdf: Path, output_text: Path, options: Namespace
|
||||||
) -> None:
|
) -> None:
|
||||||
@@ -386,8 +428,7 @@ def get_ocr_engine() -> OcrEngine:
|
|||||||
"""Returns an OcrEngine to use for processing this file.
|
"""Returns an OcrEngine to use for processing this file.
|
||||||
|
|
||||||
The OcrEngine may be instantiated multiple times, by both the main process
|
The OcrEngine may be instantiated multiple times, by both the main process
|
||||||
and child process. As such, it must be obtain store any state in ``options``
|
and child process.
|
||||||
or some common location.
|
|
||||||
|
|
||||||
Note:
|
Note:
|
||||||
This is a :ref:`firstresult hook<firstresult>`.
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
@@ -396,7 +437,7 @@ def get_ocr_engine() -> OcrEngine:
|
|||||||
|
|
||||||
@hookspec(firstresult=True)
|
@hookspec(firstresult=True)
|
||||||
def generate_pdfa(
|
def generate_pdfa(
|
||||||
pdf_pages: List[Path],
|
pdf_pages: list[Path],
|
||||||
pdfmark: Path,
|
pdfmark: Path,
|
||||||
output_file: Path,
|
output_file: Path,
|
||||||
compression: str,
|
compression: str,
|
||||||
@@ -443,3 +484,63 @@ def generate_pdfa(
|
|||||||
See also:
|
See also:
|
||||||
https://github.com/tqdm/tqdm
|
https://github.com/tqdm/tqdm
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def optimize_pdf(
|
||||||
|
input_pdf: Path,
|
||||||
|
output_pdf: Path,
|
||||||
|
context: PdfContext,
|
||||||
|
executor: Executor,
|
||||||
|
linearize: bool,
|
||||||
|
) -> tuple[Path, Sequence[str]]:
|
||||||
|
"""Optimize a PDF after image, OCR and metadata processing.
|
||||||
|
|
||||||
|
If the input_pdf is a PDF/A, the plugin should modify input_pdf in a way
|
||||||
|
that preserves the PDF/A status, or report to the user when this is not possible.
|
||||||
|
|
||||||
|
If the implementation fails to produce a smaller file than the input file, it
|
||||||
|
should return input_pdf instead.
|
||||||
|
|
||||||
|
A plugin that implements a new optimizer may need to suppress the built-in
|
||||||
|
optimizer by implementing an ``initialize`` hook.
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
input_pdf: The input PDF, which has OCR added.
|
||||||
|
output_pdf: The requested filename of the output PDF which should be created
|
||||||
|
by this optimization hook.
|
||||||
|
context: The current context.
|
||||||
|
executor: An initialized executor which may be used during optimization,
|
||||||
|
to distribute optimization tasks.
|
||||||
|
linearize: If True, OCRmyPDF requires ``optimize_pdf`` to return a linearized,
|
||||||
|
also known as fast web view PDF.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Path: If optimization is successful, the hook should return ``output_file``.
|
||||||
|
If optimization does not produce a smaller file, the hook should return
|
||||||
|
``input_file``.
|
||||||
|
Sequence[str]: Any comments that the plugin wishes to report to the user,
|
||||||
|
especially reasons it was not able to further optimize the file. For
|
||||||
|
example, the plugin could report that a required third party was not
|
||||||
|
installed, so a specific optimization was not attempted.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def is_optimization_enabled(context: PdfContext) -> bool:
|
||||||
|
"""For a given PdfContext, OCRmyPDF asks the plugin if optimization is enabled.
|
||||||
|
|
||||||
|
An optimization plugin might be installed and active but could be disabled by
|
||||||
|
user settings.
|
||||||
|
|
||||||
|
If this returns False, OCRmyPDF will take certain actions to finalize the PDF.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
True if the plugin's optimization is enabled.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This is a :ref:`firstresult hook<firstresult>`.
|
||||||
|
"""
|
||||||
|
|||||||
@@ -8,6 +8,8 @@
|
|||||||
"""Utilities to measure OCR quality"""
|
"""Utilities to measure OCR quality"""
|
||||||
|
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import re
|
import re
|
||||||
from typing import Iterable
|
from typing import Iterable
|
||||||
|
|
||||||
|
|||||||
@@ -7,18 +7,20 @@
|
|||||||
|
|
||||||
"""Wrappers to manage subprocess calls"""
|
"""Wrappers to manage subprocess calls"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
from collections.abc import Mapping
|
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from distutils.version import LooseVersion, Version
|
|
||||||
from functools import lru_cache
|
from functools import lru_cache
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||||
from subprocess import run as subprocess_run
|
from subprocess import run as subprocess_run
|
||||||
from typing import Callable, Optional, Type, Union
|
from typing import Callable, Mapping, Sequence, Union
|
||||||
|
|
||||||
|
from packaging.version import Version
|
||||||
|
|
||||||
from ocrmypdf.exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
|
|
||||||
@@ -26,8 +28,18 @@ from ocrmypdf.exceptions import MissingDependencyError
|
|||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
Args = Sequence[Union[Path, str]]
|
||||||
|
OsEnviron = os._Environ # pylint: disable=protected-access
|
||||||
|
|
||||||
def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs):
|
|
||||||
|
def run(
|
||||||
|
args: Args,
|
||||||
|
*,
|
||||||
|
env: OsEnviron | None = None,
|
||||||
|
logs_errors_to_stdout: bool = False,
|
||||||
|
check: bool = False,
|
||||||
|
**kwargs,
|
||||||
|
) -> CompletedProcess:
|
||||||
"""Wrapper around :py:func:`subprocess.run`
|
"""Wrapper around :py:func:`subprocess.run`
|
||||||
|
|
||||||
The main purpose of this wrapper is to log subprocess output in an orderly
|
The main purpose of this wrapper is to log subprocess output in an orderly
|
||||||
@@ -48,7 +60,7 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs):
|
|||||||
stderr = None
|
stderr = None
|
||||||
stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout'
|
stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout'
|
||||||
try:
|
try:
|
||||||
proc = subprocess_run(args, env=env, **kwargs)
|
proc = subprocess_run(args, env=env, check=check, **kwargs)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
stderr = getattr(e, stderr_name, None)
|
stderr = getattr(e, stderr_name, None)
|
||||||
raise
|
raise
|
||||||
@@ -65,7 +77,14 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs):
|
|||||||
return proc
|
return proc
|
||||||
|
|
||||||
|
|
||||||
def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs):
|
def run_polling_stderr(
|
||||||
|
args: Args,
|
||||||
|
*,
|
||||||
|
callback: Callable[[str], None],
|
||||||
|
check: bool = False,
|
||||||
|
env: OsEnviron | None = None,
|
||||||
|
**kwargs,
|
||||||
|
) -> CompletedProcess:
|
||||||
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
||||||
|
|
||||||
Every line of produced by stderr will be forwarded to the callback function.
|
Every line of produced by stderr will be forwarded to the callback function.
|
||||||
@@ -83,6 +102,8 @@ def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs):
|
|||||||
with Popen(args, env=env, **kwargs) as proc:
|
with Popen(args, env=env, **kwargs) as proc:
|
||||||
lines = []
|
lines = []
|
||||||
while proc.poll() is None:
|
while proc.poll() is None:
|
||||||
|
if proc.stderr is None:
|
||||||
|
continue
|
||||||
for msg in iter(proc.stderr.readline, ''):
|
for msg in iter(proc.stderr.readline, ''):
|
||||||
if process_log.isEnabledFor(logging.DEBUG):
|
if process_log.isEnabledFor(logging.DEBUG):
|
||||||
process_log.debug(msg.strip())
|
process_log.debug(msg.strip())
|
||||||
@@ -95,39 +116,38 @@ def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs):
|
|||||||
return CompletedProcess(args, proc.returncode, None, stderr=stderr)
|
return CompletedProcess(args, proc.returncode, None, stderr=stderr)
|
||||||
|
|
||||||
|
|
||||||
def _fix_process_args(args, env, kwargs):
|
def _fix_process_args(
|
||||||
|
args: Args, env: OsEnviron | None, kwargs
|
||||||
|
) -> tuple[Args, OsEnviron, logging.Logger, bool]:
|
||||||
assert 'universal_newlines' not in kwargs, "Use text= instead of universal_newlines"
|
assert 'universal_newlines' not in kwargs, "Use text= instead of universal_newlines"
|
||||||
|
|
||||||
if not env:
|
if not env:
|
||||||
env = os.environ
|
env = os.environ
|
||||||
|
|
||||||
# Search in spoof path if necessary
|
# Search in spoof path if necessary
|
||||||
program = args[0]
|
program = str(args[0])
|
||||||
|
|
||||||
if os.name == 'nt':
|
if sys.platform == 'win32':
|
||||||
|
# pylint: disable=import-outside-toplevel
|
||||||
from ocrmypdf.subprocess._windows import fix_windows_args
|
from ocrmypdf.subprocess._windows import fix_windows_args
|
||||||
|
|
||||||
args = fix_windows_args(program, args, env)
|
args = fix_windows_args(program, args, env)
|
||||||
|
|
||||||
log.debug("Running: %s", args)
|
log.debug("Running: %s", args)
|
||||||
process_log = log.getChild(os.path.basename(program))
|
process_log = log.getChild(os.path.basename(program))
|
||||||
text = kwargs.get('text', False)
|
text = bool(kwargs.get('text', False))
|
||||||
if sys.version_info < (3, 7):
|
|
||||||
if os.name == 'nt':
|
|
||||||
# Can't use close_fds=True on Windows with Python 3.6 or older
|
|
||||||
# https://bugs.python.org/issue19575, etc.
|
|
||||||
kwargs['close_fds'] = False
|
|
||||||
if 'text' in kwargs:
|
|
||||||
# Convert run(...text=) to run(...universal_newlines=) for Python 3.6
|
|
||||||
kwargs['universal_newlines'] = kwargs['text']
|
|
||||||
del kwargs['text']
|
|
||||||
return args, env, process_log, text
|
return args, env, process_log, text
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=None)
|
@lru_cache(maxsize=None)
|
||||||
def get_version(
|
def get_version(
|
||||||
program: str, *, version_arg: str = '--version', regex=r'(\d+(\.\d+)*)', env=None
|
program: str,
|
||||||
):
|
*,
|
||||||
|
version_arg: str = '--version',
|
||||||
|
regex=r'(\d+(\.\d+)*)',
|
||||||
|
env: OsEnviron | None = None,
|
||||||
|
) -> str:
|
||||||
"""Get the version of the specified program
|
"""Get the version of the specified program
|
||||||
|
|
||||||
Arguments:
|
Arguments:
|
||||||
@@ -148,7 +168,7 @@ def get_version(
|
|||||||
check=True,
|
check=True,
|
||||||
env=env,
|
env=env,
|
||||||
)
|
)
|
||||||
output = proc.stdout
|
output: str = proc.stdout
|
||||||
except FileNotFoundError as e:
|
except FileNotFoundError as e:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
f"Could not find program '{program}' on the PATH"
|
f"Could not find program '{program}' on the PATH"
|
||||||
@@ -173,42 +193,42 @@ def get_version(
|
|||||||
return version
|
return version
|
||||||
|
|
||||||
|
|
||||||
missing_program = '''
|
MISSING_PROGRAM = '''
|
||||||
The program '{program}' could not be executed or was not found on your
|
The program '{program}' could not be executed or was not found on your
|
||||||
system PATH.
|
system PATH.
|
||||||
'''
|
'''
|
||||||
|
|
||||||
missing_optional_program = '''
|
MISSING_OPTIONAL_PROGRAM = '''
|
||||||
The program '{program}' could not be executed or was not found on your
|
The program '{program}' could not be executed or was not found on your
|
||||||
system PATH. This program is required when you use the
|
system PATH. This program is required when you use the
|
||||||
{required_for} arguments. You could try omitting these arguments, or install
|
{required_for} arguments. You could try omitting these arguments, or install
|
||||||
the package.
|
the package.
|
||||||
'''
|
'''
|
||||||
|
|
||||||
missing_recommend_program = '''
|
MISSING_RECOMMEND_PROGRAM = '''
|
||||||
The program '{program}' could not be executed or was not found on your
|
The program '{program}' could not be executed or was not found on your
|
||||||
system PATH. This program is recommended when using the {required_for} arguments,
|
system PATH. This program is recommended when using the {required_for} arguments,
|
||||||
but not required, so we will proceed. For best results, install the program.
|
but not required, so we will proceed. For best results, install the program.
|
||||||
'''
|
'''
|
||||||
|
|
||||||
old_version = '''
|
OLD_VERSION = '''
|
||||||
OCRmyPDF requires '{program}' {need_version} or higher. Your system appears
|
OCRmyPDF requires '{program}' {need_version} or higher. Your system appears
|
||||||
to have {found_version}. Please update this program.
|
to have {found_version}. Please update this program.
|
||||||
'''
|
'''
|
||||||
|
|
||||||
old_version_required_for = '''
|
OLD_VERSION_REQUIRED_FOR = '''
|
||||||
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
OCRmyPDF requires '{program}' {need_version} or higher when run with the
|
||||||
{required_for} arguments. If you omit these arguments, OCRmyPDF may be able to
|
{required_for} arguments. If you omit these arguments, OCRmyPDF may be able to
|
||||||
proceed. For best results, install the program.
|
proceed. For best results, install the program.
|
||||||
'''
|
'''
|
||||||
|
|
||||||
osx_install_advice = '''
|
OSX_INSTALL_ADVICE = '''
|
||||||
If you have homebrew installed, try these command to install the missing
|
If you have homebrew installed, try these command to install the missing
|
||||||
package:
|
package:
|
||||||
brew install {package}
|
brew install {package}
|
||||||
'''
|
'''
|
||||||
|
|
||||||
linux_install_advice = '''
|
LINUX_INSTALL_ADVICE = '''
|
||||||
On systems with the aptitude package manager (Debian, Ubuntu), try these
|
On systems with the aptitude package manager (Debian, Ubuntu), try these
|
||||||
commands:
|
commands:
|
||||||
sudo apt-get update
|
sudo apt-get update
|
||||||
@@ -218,14 +238,14 @@ On RPM-based systems (Red Hat, Fedora), search for instructions on
|
|||||||
installing the RPM for {program}.
|
installing the RPM for {program}.
|
||||||
'''
|
'''
|
||||||
|
|
||||||
windows_install_advice = '''
|
WINDOWS_INSTALL_ADVICE = '''
|
||||||
If not already installed, install the Chocolatey package manager. Then use
|
If not already installed, install the Chocolatey package manager. Then use
|
||||||
a command prompt to install the missing package:
|
a command prompt to install the missing package:
|
||||||
choco install {package}
|
choco install {package}
|
||||||
'''
|
'''
|
||||||
|
|
||||||
|
|
||||||
def _get_platform():
|
def _get_platform() -> str:
|
||||||
if sys.platform.startswith('freebsd'):
|
if sys.platform.startswith('freebsd'):
|
||||||
return 'freebsd'
|
return 'freebsd'
|
||||||
elif sys.platform.startswith('linux'):
|
elif sys.platform.startswith('linux'):
|
||||||
@@ -235,46 +255,66 @@ def _get_platform():
|
|||||||
return sys.platform
|
return sys.platform
|
||||||
|
|
||||||
|
|
||||||
def _error_trailer(program, package, **kwargs):
|
def _error_trailer(program: str, package: str | Mapping[str, str], **kwargs) -> None:
|
||||||
|
del kwargs
|
||||||
if isinstance(package, Mapping):
|
if isinstance(package, Mapping):
|
||||||
package = package.get(_get_platform(), program)
|
package = package.get(_get_platform(), program)
|
||||||
|
|
||||||
if _get_platform() == 'darwin':
|
if _get_platform() == 'darwin':
|
||||||
log.info(osx_install_advice.format(**locals()))
|
log.info(OSX_INSTALL_ADVICE.format(**locals()))
|
||||||
elif _get_platform() == 'linux':
|
elif _get_platform() == 'linux':
|
||||||
log.info(linux_install_advice.format(**locals()))
|
log.info(LINUX_INSTALL_ADVICE.format(**locals()))
|
||||||
elif _get_platform() == 'windows':
|
elif _get_platform() == 'windows':
|
||||||
log.info(windows_install_advice.format(**locals()))
|
log.info(WINDOWS_INSTALL_ADVICE.format(**locals()))
|
||||||
|
|
||||||
|
|
||||||
def _error_missing_program(program, package, required_for, recommended):
|
def _error_missing_program(
|
||||||
|
program: str, package: str, required_for: str | None, recommended: bool
|
||||||
|
) -> None:
|
||||||
|
# pylint: disable=unused-argument
|
||||||
if recommended:
|
if recommended:
|
||||||
log.warning(missing_recommend_program.format(**locals()))
|
log.warning(MISSING_RECOMMEND_PROGRAM.format(**locals()))
|
||||||
elif required_for:
|
elif required_for:
|
||||||
log.error(missing_optional_program.format(**locals()))
|
log.error(MISSING_OPTIONAL_PROGRAM.format(**locals()))
|
||||||
else:
|
else:
|
||||||
log.error(missing_program.format(**locals()))
|
log.error(MISSING_PROGRAM.format(**locals()))
|
||||||
_error_trailer(**locals())
|
_error_trailer(**locals())
|
||||||
|
|
||||||
|
|
||||||
def _error_old_version(program, package, need_version, found_version, required_for):
|
def _error_old_version(
|
||||||
|
program: str,
|
||||||
|
package: str,
|
||||||
|
need_version: str,
|
||||||
|
found_version: str,
|
||||||
|
required_for: str | None,
|
||||||
|
) -> None:
|
||||||
|
# pylint: disable=unused-argument
|
||||||
if required_for:
|
if required_for:
|
||||||
log.error(old_version_required_for.format(**locals()))
|
log.error(OLD_VERSION_REQUIRED_FOR.format(**locals()))
|
||||||
else:
|
else:
|
||||||
log.error(old_version.format(**locals()))
|
log.error(OLD_VERSION.format(**locals()))
|
||||||
_error_trailer(**locals())
|
_error_trailer(**locals())
|
||||||
|
|
||||||
|
|
||||||
|
def _remove_leading_v(s: str) -> str:
|
||||||
|
if sys.version_info >= (3, 9):
|
||||||
|
return s.removeprefix('v')
|
||||||
|
|
||||||
|
if s.startswith('v'):
|
||||||
|
return s[1:]
|
||||||
|
return s
|
||||||
|
|
||||||
|
|
||||||
def check_external_program(
|
def check_external_program(
|
||||||
*,
|
*,
|
||||||
program: str,
|
program: str,
|
||||||
package: str,
|
package: str,
|
||||||
version_checker: Union[str, Callable],
|
version_checker: Callable[[], str],
|
||||||
need_version: str,
|
need_version: str,
|
||||||
required_for: Optional[str] = None,
|
required_for: str | None = None,
|
||||||
recommended=False,
|
recommended: bool = False,
|
||||||
version_parser: Type[Version] = LooseVersion,
|
version_parser: type[Version] = Version,
|
||||||
):
|
) -> None:
|
||||||
"""Check for required version of external program and raise exception if not.
|
"""Check for required version of external program and raise exception if not.
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
@@ -294,21 +334,21 @@ def check_external_program(
|
|||||||
try:
|
try:
|
||||||
if callable(version_checker):
|
if callable(version_checker):
|
||||||
found_version = version_checker()
|
found_version = version_checker()
|
||||||
else:
|
else: # deprecated
|
||||||
found_version = version_checker
|
found_version = version_checker
|
||||||
except (CalledProcessError, FileNotFoundError, MissingDependencyError):
|
except (CalledProcessError, FileNotFoundError) as e:
|
||||||
_error_missing_program(program, package, required_for, recommended)
|
_error_missing_program(program, package, required_for, recommended)
|
||||||
if not recommended:
|
if not recommended:
|
||||||
raise MissingDependencyError(program)
|
raise MissingDependencyError(program) from e
|
||||||
|
return
|
||||||
|
except MissingDependencyError:
|
||||||
|
_error_missing_program(program, package, required_for, recommended)
|
||||||
|
if not recommended:
|
||||||
|
raise
|
||||||
return
|
return
|
||||||
|
|
||||||
def remove_leading_v(s):
|
found_version = _remove_leading_v(found_version)
|
||||||
if s.startswith('v'):
|
need_version = _remove_leading_v(need_version)
|
||||||
return s[1:]
|
|
||||||
return s
|
|
||||||
|
|
||||||
found_version = remove_leading_v(found_version)
|
|
||||||
need_version = remove_leading_v(need_version)
|
|
||||||
|
|
||||||
if found_version and version_parser(found_version) < version_parser(need_version):
|
if found_version and version_parser(found_version) < version_parser(need_version):
|
||||||
_error_old_version(program, package, need_version, found_version, required_for)
|
_error_old_version(program, package, need_version, found_version, required_for)
|
||||||
|
|||||||
@@ -3,65 +3,95 @@
|
|||||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
"""Find Tesseract and Ghostscript binaries on Windows using the registry."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
from distutils.version import LooseVersion
|
from itertools import chain
|
||||||
from itertools import chain, filterfalse
|
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any, Callable, Iterator, Optional, Tuple, TypeVar, cast
|
from typing import Any, Callable, Iterable, Iterator, TypeVar
|
||||||
|
|
||||||
|
if sys.version_info >= (3, 10):
|
||||||
|
from typing import TypeAlias
|
||||||
|
else:
|
||||||
|
from typing_extensions import TypeAlias # pragma: no cover
|
||||||
|
|
||||||
|
if sys.platform == 'win32':
|
||||||
|
# mypy understands 'if sys.platform' better than try/except ModuleNotFoundError
|
||||||
|
import winreg # pylint: disable=import-error
|
||||||
|
|
||||||
|
HKEYType: TypeAlias = winreg.HKEYType
|
||||||
|
else:
|
||||||
|
from unittest.mock import Mock
|
||||||
|
|
||||||
|
winreg = Mock(
|
||||||
|
spec=['HKEYType', 'EnumKey', 'EnumValue', 'HKEY_LOCAL_MACHINE', 'OpenKey']
|
||||||
|
)
|
||||||
|
# mypy does not understand winreg.HKeyType where winreg is a Mock (fair enough!)
|
||||||
|
HKEYType: TypeAlias = Any
|
||||||
|
|
||||||
try:
|
|
||||||
import winreg
|
|
||||||
except ModuleNotFoundError as e:
|
|
||||||
raise ModuleNotFoundError("This module is for Windows only") from e
|
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
T = TypeVar('T')
|
T = TypeVar('T')
|
||||||
|
|
||||||
|
|
||||||
def registry_enum(
|
def ghostscript_version_key(s: str) -> tuple[int, int, int]:
|
||||||
key: winreg.HKEYType, enum_fn: Callable[[winreg.HKEYType, int], T]
|
"""Compare Ghostscript version numbers."""
|
||||||
) -> Iterator[T]:
|
try:
|
||||||
LIMIT = 999
|
release = [int(elem) for elem in s.split('.', maxsplit=3)]
|
||||||
|
while len(release) < 3:
|
||||||
|
release.append(0)
|
||||||
|
return (release[0], release[1], release[2])
|
||||||
|
except ValueError:
|
||||||
|
return (0, 0, 0)
|
||||||
|
|
||||||
|
|
||||||
|
def registry_enum(key: HKEYType, enum_fn: Callable[[HKEYType, int], T]) -> Iterator[T]:
|
||||||
|
limit = 999
|
||||||
n = 0
|
n = 0
|
||||||
while n < LIMIT:
|
while n < limit:
|
||||||
try:
|
try:
|
||||||
yield enum_fn(key, n)
|
yield enum_fn(key, n)
|
||||||
n += 1
|
n += 1
|
||||||
except OSError:
|
except OSError:
|
||||||
break
|
break
|
||||||
if n == LIMIT:
|
if n == limit:
|
||||||
raise ValueError(f"Too many registry keys under {key}")
|
raise ValueError(f"Too many registry keys under {key}")
|
||||||
|
|
||||||
|
|
||||||
def registry_subkeys(key: winreg.HKEYType) -> Iterator[str]:
|
def registry_subkeys(key: HKEYType) -> Iterator[str]:
|
||||||
return registry_enum(key, winreg.EnumKey)
|
return registry_enum(key, winreg.EnumKey)
|
||||||
|
|
||||||
|
|
||||||
def registry_values(key: winreg.HKEYType) -> Iterator[Tuple[str, Any, int]]:
|
def registry_values(key: HKEYType) -> Iterator[tuple[str, Any, int]]:
|
||||||
return registry_enum(key, winreg.EnumValue)
|
return registry_enum(key, winreg.EnumValue)
|
||||||
|
|
||||||
|
|
||||||
def registry_path_ghostscript(env=None) -> Iterator[Path]:
|
def registry_path_ghostscript(env=None) -> Iterator[Path]:
|
||||||
|
del env # unused (but needed for protocol)
|
||||||
try:
|
try:
|
||||||
with winreg.OpenKey(
|
with winreg.OpenKey(
|
||||||
winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Artifex\GPL Ghostscript"
|
winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Artifex\GPL Ghostscript"
|
||||||
) as k:
|
) as k:
|
||||||
latest_gs = max(registry_subkeys(k), key=LooseVersion)
|
latest_gs = max(
|
||||||
|
registry_subkeys(k), key=ghostscript_version_key, default=(0, 0, 0)
|
||||||
|
)
|
||||||
with winreg.OpenKey(
|
with winreg.OpenKey(
|
||||||
winreg.HKEY_LOCAL_MACHINE, fr"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
|
winreg.HKEY_LOCAL_MACHINE, fr"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
|
||||||
) as k:
|
) as k:
|
||||||
_, gs_path, _ = next(registry_values(k))
|
for _, gs_path, _ in registry_values(k):
|
||||||
yield Path(gs_path) / 'bin'
|
yield Path(gs_path) / 'bin'
|
||||||
except OSError as e:
|
except OSError as e:
|
||||||
log.warning(e)
|
log.warning(e)
|
||||||
|
|
||||||
|
|
||||||
def registry_path_tesseract(env=None) -> Iterator[Path]:
|
def registry_path_tesseract(env=None) -> Iterator[Path]:
|
||||||
|
del env # unused (but needed for protocol)
|
||||||
try:
|
try:
|
||||||
with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Tesseract-OCR") as k:
|
with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Tesseract-OCR") as k:
|
||||||
for subkey, val, _valtype in registry_values(k):
|
for subkey, val, _valtype in registry_values(k):
|
||||||
@@ -113,7 +143,7 @@ SHIMS = [
|
|||||||
]
|
]
|
||||||
|
|
||||||
|
|
||||||
def fix_windows_args(program, args, env):
|
def fix_windows_args(program: str, args, env):
|
||||||
"""Adjust our desired program and command line arguments for use on Windows"""
|
"""Adjust our desired program and command line arguments for use on Windows"""
|
||||||
|
|
||||||
if sys.version_info < (3, 8):
|
if sys.version_info < (3, 8):
|
||||||
@@ -137,14 +167,12 @@ def fix_windows_args(program, args, env):
|
|||||||
return args
|
return args
|
||||||
|
|
||||||
|
|
||||||
def unique_everseen(iterable, key=None):
|
def unique_everseen(iterable: Iterable[T], key: Callable[[T], T]) -> Iterator[T]:
|
||||||
"List unique elements, preserving order. Remember all elements ever seen."
|
"List unique elements, preserving order."
|
||||||
# unique_everseen('AAAABBBCCDAABBB') --> A B C D
|
# unique_everseen('AAAABBBCCDAABBB') --> A B C D
|
||||||
# unique_everseen('ABBCcAD', str.lower) --> A B C D
|
# unique_everseen('ABBCcAD', str.lower) --> A B C D
|
||||||
seen = set()
|
seen: set[T] = set()
|
||||||
seen_add = seen.add
|
seen_add = seen.add
|
||||||
if key is None:
|
|
||||||
key = lambda x: x
|
|
||||||
for element in iterable:
|
for element in iterable:
|
||||||
k = key(element)
|
k = key(element)
|
||||||
if k not in seen:
|
if k not in seen:
|
||||||
|
|||||||
@@ -4,4 +4,6 @@
|
|||||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
# Empty __init__.py file
|
# Empty __init__.py file
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
||||||
|
|||||||
+9
-7
@@ -1,4 +1,6 @@
|
|||||||
a la Waterman
|
i a la Waterman
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
4 ons linzen
|
4 ons linzen
|
||||||
|
|
||||||
@@ -12,14 +14,14 @@ bloem, boter
|
|||||||
|
|
||||||
laurier, kruidnagel, kerrie, zout
|
laurier, kruidnagel, kerrie, zout
|
||||||
|
|
||||||
De linzgen wassen en in -l liter kokend wa-
|
De linzgen wassen en in-l liter kokend wa-
|
||||||
ter 1 dag laten weken, 2 liter water bij
|
ter 1 dag laten weken, 2 liter water bij
|
||||||
de linzen voegen, zonder het water waarin
|
de linzen voegen, zonder het water waarin
|
||||||
ze geweekt zijn af te gieten, De helft van
|
ze geweekt zijn af te gieten., De helft van
|
||||||
de uien bakken met laurier en Kruicdnagel.
|
de uien bakken met laurier en Kruidnagel.
|
||||||
Alle uien, kerrie en gout bij de linzen
|
Alle uien, kerrie en zgout bij de linzen
|
||||||
voegen, Alles aan de kook brengen,. Van de
|
voegen, Alles aan de kook brengen, Van de
|
||||||
bloem met boter en melk een papje maken en
|
bloem met boter en melk een papje maken en
|
||||||
verder afmaken met de soep, Als de linzen
|
verder afmaken met de soep, Als de linzen
|
||||||
gaar Zijn is de soep klaar.
|
gfgaar Zijn is de soep klaar.
|
||||||
|
|
||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
||||||
|
|||||||
+3
-3
@@ -22,9 +22,9 @@ Mugerre
|
|||||||
|
|
||||||
Milafranga Komunikabideak
|
Milafranga Komunikabideak
|
||||||
|
|
||||||
BAIONA i zeettnansise —
|
BAIONA zeiteninsiie —
|
||||||
|
|
||||||
1 Trenbideak -- ~~~
|
7 Trenbideak -----
|
||||||
|
|
||||||
t\ Basusarri — spmsans20141004 se: . a ~
|
t\ Basusarri — spmeans:20141004 ae: . _ ~
|
||||||
|
|
||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
||||||
|
|||||||
+6
-6
@@ -14,14 +14,14 @@ bloem, boter
|
|||||||
|
|
||||||
laurier, kruidnagel, kerrie, zout
|
laurier, kruidnagel, kerrie, zout
|
||||||
|
|
||||||
De linzgen wassen en in -l liter kokend wa-
|
De linzgen wassen en in-l liter kokend wa-
|
||||||
ter 1 dag laten weken, 2 liter water bij
|
ter 1 dag laten weken, 2 liter water bij
|
||||||
de linzen voegen, zonder het water waarin
|
de linzen voegen, zonder het water waarin
|
||||||
ze geweekt zijn af te gieten, De helft van
|
ze geweekt zijn af te gieten., De helft van
|
||||||
de uien bakken met laurier en Kruicdnagel.
|
de uien bakken met laurier en Kruidnagel.
|
||||||
Alle uien, kerrie en gout bij de linzen
|
Alle uien, kerrie en zgout bij de linzen
|
||||||
voegen, Alles aan de kook brengen,. Van de
|
voegen, Alles aan de kook brengen, Van de
|
||||||
bloem met boter en melk een papje maken en
|
bloem met boter en melk een papje maken en
|
||||||
verder afmaken met de soep, Als de linzen
|
verder afmaken met de soep, Als de linzen
|
||||||
gaar Zijn is de soep klaar.
|
gfgaar Zijn is de soep klaar.
|
||||||
|
|
||||||
+6
-6
@@ -5,11 +5,11 @@
|
|||||||
<head>
|
<head>
|
||||||
<title></title>
|
<title></title>
|
||||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||||
<meta name='ocr-system' content='tesseract 4.1.1' />
|
<meta name='ocr-system' content='tesseract 5.0.0-beta-20210916-12-g19cc9' />
|
||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.zi7wmyqr/000001_ocr.png"; bbox 0 0 1000 800; ppageno 0'>
|
<div class='ocr_page' id='page_1' title='image "/tmp/ocrmypdf.io.gzrr1v_b/000001_ocr.png"; bbox 0 0 1000 800; ppageno 0'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 296 96 704 504">
|
<div class='ocr_carea' id='block_1_1' title="bbox 296 96 704 504">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 296 96 704 504">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 296 96 704 504">
|
||||||
<span class='ocr_line' id='line_1_1' title="bbox 296 96 704 504; baseline 0 296; x_size 169.33333; x_descenders 42.333332; x_ascenders 42.333336">
|
<span class='ocr_line' id='line_1_1' title="bbox 296 96 704 504; baseline 0 296; x_size 169.33333; x_descenders 42.333332; x_ascenders 42.333336">
|
||||||
@@ -20,12 +20,12 @@
|
|||||||
<div class='ocr_carea' id='block_1_2' title="bbox 150 592 841 622">
|
<div class='ocr_carea' id='block_1_2' title="bbox 150 592 841 622">
|
||||||
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 150 592 841 622">
|
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 150 592 841 622">
|
||||||
<span class='ocr_line' id='line_1_2' title="bbox 150 592 841 622; baseline 0 -6; x_size 30; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_2' title="bbox 150 592 841 622; baseline 0 -6; x_size 30; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_2' title='bbox 150 592 230 616; x_wconf 96'>This</span>
|
<span class='ocrx_word' id='word_1_2' title='bbox 150 592 230 616; x_wconf 95'>This</span>
|
||||||
<span class='ocrx_word' id='word_1_3' title='bbox 260 592 384 616; x_wconf 95'>should</span>
|
<span class='ocrx_word' id='word_1_3' title='bbox 260 592 384 616; x_wconf 95'>should</span>
|
||||||
<span class='ocrx_word' id='word_1_4' title='bbox 413 592 449 616; x_wconf 95'>be</span>
|
<span class='ocrx_word' id='word_1_4' title='bbox 413 592 449 616; x_wconf 95'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_5' title='bbox 479 600 493 616; x_wconf 95'>a</span>
|
<span class='ocrx_word' id='word_1_5' title='bbox 479 600 493 616; x_wconf 94'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_6' title='bbox 523 592 668 622; x_wconf 95'>perfect</span>
|
<span class='ocrx_word' id='word_1_6' title='bbox 523 592 668 622; x_wconf 94'>perfect</span>
|
||||||
<span class='ocrx_word' id='word_1_7' title='bbox 698 592 841 616; x_wconf 55'>circle:</span>
|
<span class='ocrx_word' id='word_1_7' title='bbox 698 592 841 616; x_wconf 61'>circle:</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
|
|||||||
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
Tesseract Open Source OCR Engine v5.0.0-beta-20210916-12-g19cc9 with Leptonica
|
||||||
|
|||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user