Compare commits
559
Commits
continuous
..
v3.2
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
344fc40cbc | ||
|
|
7e5c37137b | ||
|
|
1aae11714b | ||
|
|
d82f14a7aa | ||
|
|
4b65e0b093 | ||
|
|
43b0faa830 | ||
|
|
8674c9fb20 | ||
|
|
ccfbb54e8c | ||
|
|
9893ebf889 | ||
|
|
303eb3e93a | ||
|
|
ca546d70e5 | ||
|
|
6a5ea2d64a | ||
|
|
bacbcba58a | ||
|
|
52e8aa434f | ||
|
|
37c508f3f8 | ||
|
|
26e36422cc | ||
|
|
f82cb002bc | ||
|
|
c1eb047a4b | ||
|
|
626ca18f5c | ||
|
|
9058dedfbe | ||
|
|
a0952bfca3 | ||
|
|
354e61946e | ||
|
|
fd6d1d748a | ||
|
|
360acd1e2c | ||
|
|
fc0479f110 | ||
|
|
62728205b6 | ||
|
|
dc0fb25e64 | ||
|
|
f3e04cce56 | ||
|
|
7067110308 | ||
|
|
599d889703 | ||
|
|
2fa8366632 | ||
|
|
c368c51bad | ||
|
|
7c558b3713 | ||
|
|
8d323ae510 | ||
|
|
3b53e9adac | ||
|
|
074c1d71b4 | ||
|
|
1fca9a004d | ||
|
|
b485a1ef78 | ||
|
|
326ef7a3ac | ||
|
|
12bc58b5b6 | ||
|
|
6af0815681 | ||
|
|
66c2b9b78e | ||
|
|
d03c056cb1 | ||
|
|
3f94d628fa | ||
|
|
a64c7dbe99 | ||
|
|
61b3ccb57c | ||
|
|
424b4b33b1 | ||
|
|
e510f89792 | ||
|
|
49cd6cc619 | ||
|
|
9aa3d340d4 | ||
|
|
09782242c8 | ||
|
|
9ec4aa039d | ||
|
|
ecebe2f24b | ||
|
|
7313a77c2a | ||
|
|
45113676a3 | ||
|
|
102bd07019 | ||
|
|
9622e31da9 | ||
|
|
1731ce2a44 | ||
|
|
276f421c44 | ||
|
|
133357779a | ||
|
|
5d8167b232 | ||
|
|
e76ae8c46c | ||
|
|
53a7c0e668 | ||
|
|
4ca243e490 | ||
|
|
9f37446155 | ||
|
|
d7c7559b05 | ||
|
|
b2b66d1344 | ||
|
|
5d111a3c04 | ||
|
|
10416f847f | ||
|
|
79b3472b26 | ||
|
|
f1b2f1ae08 | ||
|
|
ee7d97ae8c | ||
|
|
7d9f473bb1 | ||
|
|
e77a5e5e75 | ||
|
|
6ab19af122 | ||
|
|
276fe49867 | ||
|
|
acb31abe86 | ||
|
|
4f964a3c8a | ||
|
|
df1fda7438 | ||
|
|
d6124c1787 | ||
|
|
80d89b5420 | ||
|
|
74059eecf1 | ||
|
|
78697341a2 | ||
|
|
cfb56dd8ff | ||
|
|
b1769cbe18 | ||
|
|
955b801e7f | ||
|
|
3cea3f1afe | ||
|
|
fd4a227ccb | ||
|
|
19c3097483 | ||
|
|
cdd1a6d03c | ||
|
|
5fb8411571 | ||
|
|
334a15b8c7 | ||
|
|
6390736577 | ||
|
|
d55a214516 | ||
|
|
0994164b9a | ||
|
|
54ee0dd147 | ||
|
|
47c7990fb3 | ||
|
|
997e95de4d | ||
|
|
44204be256 | ||
|
|
9b1d9aa88a | ||
|
|
b775762f6a | ||
|
|
df1a28e319 | ||
|
|
c300b2802a | ||
|
|
01040ace4c | ||
|
|
8367172e0b | ||
|
|
09afd8d25d | ||
|
|
7ed60429b3 | ||
|
|
281eafada0 | ||
|
|
c14e10128a | ||
|
|
3270635192 | ||
|
|
3d26257710 | ||
|
|
c4f134d694 | ||
|
|
83f9dfbac4 | ||
|
|
3a445ad5f7 | ||
|
|
c6d106ec33 | ||
|
|
2ce6834be4 | ||
|
|
b376672dbc | ||
|
|
d07db8547f | ||
|
|
aab08bfcc7 | ||
|
|
e0a25494ee | ||
|
|
fd876d5e4e | ||
|
|
ee7f008ff5 | ||
|
|
d9161a6ddb | ||
|
|
f8d66768e3 | ||
|
|
4f3673d14d | ||
|
|
1712fdb74a | ||
|
|
3a5ffc79e0 | ||
|
|
859b063444 | ||
|
|
bd61e7c644 | ||
|
|
c9abf282b5 | ||
|
|
9dad40b5a3 | ||
|
|
8e2d690cb0 | ||
|
|
c132e091e1 | ||
|
|
630e6cbf1e | ||
|
|
83ff5760a8 | ||
|
|
fed0ee638e | ||
|
|
cc161780df | ||
|
|
898b2b000a | ||
|
|
b3ee743ed7 | ||
|
|
ef17b669fe | ||
|
|
2dff3e07ce | ||
|
|
53c88093ad | ||
|
|
0ec13d3a17 | ||
|
|
0d5104049a | ||
|
|
ce8fa69785 | ||
|
|
30072e0c70 | ||
|
|
eb04a890b2 | ||
|
|
0c53adb04f | ||
|
|
ee5a43fd47 | ||
|
|
c43d6c2cbe | ||
|
|
87aeeacb04 | ||
|
|
6b26e9cad6 | ||
|
|
85af0f0d03 | ||
|
|
f6f4705ea3 | ||
|
|
a4702bff22 | ||
|
|
73c5c48f79 | ||
|
|
adf495e8cc | ||
|
|
9247ea00bf | ||
|
|
a1238d7bf9 | ||
|
|
2d63268f0f | ||
|
|
1cb5f6a90d | ||
|
|
8d848284df | ||
|
|
8fe54d1a5c | ||
|
|
11dd9f14c3 | ||
|
|
16d24f1166 | ||
|
|
97015ef775 | ||
|
|
2744dafb74 | ||
|
|
7b268dbe1a | ||
|
|
8fcbbcef94 | ||
|
|
8f93f0a06e | ||
|
|
387142488c | ||
|
|
6887e232fc | ||
|
|
6ac7ffd77b | ||
|
|
b28faa582a | ||
|
|
454ee029c8 | ||
|
|
a036de318e | ||
|
|
9918c4020e | ||
|
|
3d6264e1b8 | ||
|
|
1c25270503 | ||
|
|
47e50f82c4 | ||
|
|
27ecdfbba8 | ||
|
|
6901550065 | ||
|
|
6e6f918630 | ||
|
|
4633812246 | ||
|
|
14bd1555aa | ||
|
|
b9d7687fa0 | ||
|
|
93b36965e2 | ||
|
|
9e0c443c2f | ||
|
|
60832152b1 | ||
|
|
6a160d22fe | ||
|
|
e35526192c | ||
|
|
bea57bdded | ||
|
|
2a9da225e4 | ||
|
|
a3f37de9b5 | ||
|
|
6064160953 | ||
|
|
8508141314 | ||
|
|
1c95597882 | ||
|
|
587fa63c8e | ||
|
|
b40eec4cb0 | ||
|
|
7bcd48c269 | ||
|
|
2e7cd52c0f | ||
|
|
77d4cb367e | ||
|
|
2c45c5abc6 | ||
|
|
a89afabd79 | ||
|
|
03f7c9bf07 | ||
|
|
d5f4862749 | ||
|
|
8aced0b6d3 | ||
|
|
6b9adef684 | ||
|
|
5440d988fc | ||
|
|
30da4fc569 | ||
|
|
2c1b5e100b | ||
|
|
3684f278ed | ||
|
|
6c3cb6acba | ||
|
|
b98ba8d174 | ||
|
|
d3088829af | ||
|
|
9aaaba1714 | ||
|
|
9adb0d696f | ||
|
|
c270f1ba5f | ||
|
|
7b255b575a | ||
|
|
d7a9f3a2ab | ||
|
|
abf2e7e9bb | ||
|
|
72e5fa9ba0 | ||
|
|
32c1078d2c | ||
|
|
133f901a69 | ||
|
|
42cd683ec0 | ||
|
|
151eb05377 | ||
|
|
16177d0a52 | ||
|
|
5ce544289f | ||
|
|
77bd35c3c7 | ||
|
|
0c5c208db0 | ||
|
|
60eb745331 | ||
|
|
9f90b5cb0a | ||
|
|
5adff94545 | ||
|
|
aa2baabfa9 | ||
|
|
75c2b23efc | ||
|
|
6451017962 | ||
|
|
0f857a6a34 | ||
|
|
7638a88a6a | ||
|
|
bed12d2021 | ||
|
|
587569fcb6 | ||
|
|
8c0dc9a06d | ||
|
|
289e4025ad | ||
|
|
5476eafe4c | ||
|
|
df32f283cd | ||
|
|
68ecaac9cc | ||
|
|
cffd4623ca | ||
|
|
6dc2782e80 | ||
|
|
5df187c086 | ||
|
|
7fd172e41e | ||
|
|
619528a1b5 | ||
|
|
596d468c14 | ||
|
|
eddbf1060a | ||
|
|
33731a6864 | ||
|
|
0c36cd2e24 | ||
|
|
5cef1be26d | ||
|
|
e89f482c3d | ||
|
|
fe3e40305d | ||
|
|
a92b5ceb6b | ||
|
|
0e7e7d8437 | ||
|
|
f47fa98f33 | ||
|
|
d3d5879911 | ||
|
|
b2168e11db | ||
|
|
6d5d8be708 | ||
|
|
ce2dbdf372 | ||
|
|
ec8a35a7a6 | ||
|
|
f6577c22c3 | ||
|
|
43d6c03093 | ||
|
|
1870f116bb | ||
|
|
8b87def013 | ||
|
|
de599d97b5 | ||
|
|
5d7e6b45c4 | ||
|
|
c6091bcfe1 | ||
|
|
466a8a1318 | ||
|
|
a99ba3b696 | ||
|
|
9229f7c6cc | ||
|
|
bf114bb188 | ||
|
|
b8eed2f861 | ||
|
|
ccb1e347be | ||
|
|
8698974f11 | ||
|
|
f2c79c4341 | ||
|
|
4966d1346b | ||
|
|
4a9337f757 | ||
|
|
db311fb6a2 | ||
|
|
02c1dcec8e | ||
|
|
52dc74d3ce | ||
|
|
cc2af2bc15 | ||
|
|
638c6db05d | ||
|
|
f7db8d9aff | ||
|
|
564fb7a87e | ||
|
|
4d88e64774 | ||
|
|
26f1163b46 | ||
|
|
40058e99e0 | ||
|
|
bece4c3e02 | ||
|
|
f0f6b57c87 | ||
|
|
dc2a4ab044 | ||
|
|
b16d6f5b81 | ||
|
|
69ce6ff7b5 | ||
|
|
32ba50b8dc | ||
|
|
36aca45f35 | ||
|
|
925290342d | ||
|
|
4dc0370c57 | ||
|
|
b92f8e43f2 | ||
|
|
22b0733a1d | ||
|
|
6021684ab6 | ||
|
|
f4b1d0cdfe | ||
|
|
d0d8048621 | ||
|
|
cfd119325d | ||
|
|
ad30833ffc | ||
|
|
e5c79a6666 | ||
|
|
63dc753c1b | ||
|
|
017bc1f252 | ||
|
|
bcd67c009d | ||
|
|
635358884e | ||
|
|
2f6cfafdfc | ||
|
|
25234fa30b | ||
|
|
5b17341804 | ||
|
|
9bedfa9a72 | ||
|
|
e1f1220970 | ||
|
|
5855bcd1fe | ||
|
|
a14af5b9ee | ||
|
|
f11c03750e | ||
|
|
ea5cfa40c1 | ||
|
|
c562754d81 | ||
|
|
90d892512a | ||
|
|
9c6fedb15b | ||
|
|
3a7175115f | ||
|
|
98c41f3223 | ||
|
|
d101e96e16 | ||
|
|
a446b6c440 | ||
|
|
b1fec0f1b1 | ||
|
|
1dfdc93745 | ||
|
|
6c5ee4095c | ||
|
|
986fbf63a4 | ||
|
|
2612105d32 | ||
|
|
954fe13f54 | ||
|
|
bb5a00685e | ||
|
|
dabbddb04e | ||
|
|
5f173e5acb | ||
|
|
b28ff40aea | ||
|
|
fccfb4589e | ||
|
|
5384c98013 | ||
|
|
2ed2307573 | ||
|
|
3f8a2d8d3e | ||
|
|
1a13b7c85f | ||
|
|
d7130a1e56 | ||
|
|
f69054cb17 | ||
|
|
80dc6eca2c | ||
|
|
d250fbb3d6 | ||
|
|
09bbe92611 | ||
|
|
69d922e096 | ||
|
|
d510e7e4ae | ||
|
|
5893290dd9 | ||
|
|
5c3bbc4031 | ||
|
|
27cd8cf0db | ||
|
|
b403016d5b | ||
|
|
5a81823969 | ||
|
|
17801401cd | ||
|
|
5c7b2a2a36 | ||
|
|
1db06de287 | ||
|
|
3904178d44 | ||
|
|
8bb9c3610c | ||
|
|
7dcc382ccc | ||
|
|
b71fc807d2 | ||
|
|
15d28d970a | ||
|
|
e083a860e9 | ||
|
|
6463b9dd84 | ||
|
|
c873de6ca4 | ||
|
|
b70863b47e | ||
|
|
3546f84c6d | ||
|
|
1d98917db9 | ||
|
|
1c34fd69cf | ||
|
|
4cf38404cc | ||
|
|
112fb5098b | ||
|
|
5ace6906c7 | ||
|
|
8cfbdaf0d0 | ||
|
|
6703434976 | ||
|
|
62edc15cd7 | ||
|
|
be830ddc31 | ||
|
|
18322b424f | ||
|
|
6901c60db4 | ||
|
|
e369ce6766 | ||
|
|
64e4e5d91e | ||
|
|
efce7de9ae | ||
|
|
38c64ac689 | ||
|
|
6d203e3eee | ||
|
|
81f461e557 | ||
|
|
988bde1387 | ||
|
|
aedbabdbe8 | ||
|
|
6ed53e53c7 | ||
|
|
a78630ce99 | ||
|
|
6653066784 | ||
|
|
e40f1fa081 | ||
|
|
a872ce751d | ||
|
|
317846fbdc | ||
|
|
f581a55544 | ||
|
|
447b291e70 | ||
|
|
01d07253e8 | ||
|
|
034a466094 | ||
|
|
c6211e2335 | ||
|
|
1d03a6417d | ||
|
|
1d62ef27a2 | ||
|
|
996048dc08 | ||
|
|
bf02ee3bdc | ||
|
|
a3c7fba02d | ||
|
|
a8cd7febf6 | ||
|
|
20c008b84f | ||
|
|
7cd73566be | ||
|
|
e56fd53d06 | ||
|
|
810b1b3b3e | ||
|
|
cb0b033fe7 | ||
|
|
46f673a3b7 | ||
|
|
455303b3d4 | ||
|
|
24a84d6380 | ||
|
|
9aa2171052 | ||
|
|
3a46ea1f36 | ||
|
|
d33779f301 | ||
|
|
d6ea0793b8 | ||
|
|
4e5e5bb925 | ||
|
|
3232ed8e38 | ||
|
|
29d6748af8 | ||
|
|
828f195071 | ||
|
|
b0b7e32783 | ||
|
|
c1103c0248 | ||
|
|
940a016e95 | ||
|
|
c6cc098e47 | ||
|
|
54f47ab89b | ||
|
|
fc3de64dce | ||
|
|
414c4e3f3c | ||
|
|
6a9f38d31e | ||
|
|
aa4256d35c | ||
|
|
8a1241ba44 | ||
|
|
7eab052e0f | ||
|
|
552d19e36b | ||
|
|
463b04e795 | ||
|
|
c0d8508264 | ||
|
|
6ef4ba31e2 | ||
|
|
10a3d26291 | ||
|
|
ab994b32ee | ||
|
|
9352b71d78 | ||
|
|
71593421ed | ||
|
|
2754970f37 | ||
|
|
5945454597 | ||
|
|
884dbce712 | ||
|
|
8ee1bc6598 | ||
|
|
7d76c46731 | ||
|
|
f8ccf42c06 | ||
|
|
0abe0f1f10 | ||
|
|
ee8a5d80ff | ||
|
|
f08893b5c8 | ||
|
|
41cd88506e | ||
|
|
081223b138 | ||
|
|
4e60c9ba09 | ||
|
|
95fe7cd3bc | ||
|
|
79ec1d994e | ||
|
|
045362425f | ||
|
|
2b2637fbc3 | ||
|
|
bfc4f7a28d | ||
|
|
407670e1f3 | ||
|
|
d0671d81b5 | ||
|
|
7a74ebbcc3 | ||
|
|
9e69800332 | ||
|
|
7542188592 | ||
|
|
b4a23c005d | ||
|
|
5e0f8be4b1 | ||
|
|
50dee55606 | ||
|
|
da5cd01fe4 | ||
|
|
d3fb317d41 | ||
|
|
88ddeb1fb6 | ||
|
|
f9e2e74bf3 | ||
|
|
87e01aff60 | ||
|
|
7e8481186a | ||
|
|
2b0103a4e6 | ||
|
|
064d4be83c | ||
|
|
ab536d5678 | ||
|
|
9db805c4ad | ||
|
|
f7923a9761 | ||
|
|
fd52650255 | ||
|
|
f0fe295175 | ||
|
|
2f89aa3935 | ||
|
|
5aa27343e0 | ||
|
|
5ce2841389 | ||
|
|
e4ffb58269 | ||
|
|
2ce3d9e19d | ||
|
|
9271fe73a8 | ||
|
|
edaa70b97f | ||
|
|
ab07f4deea | ||
|
|
beb1d7ab54 | ||
|
|
5ce3e9bfec | ||
|
|
2bed210a30 | ||
|
|
2441551156 | ||
|
|
15baca5e08 | ||
|
|
062ef0ca3a | ||
|
|
24b4686944 | ||
|
|
d3d1c20ca2 | ||
|
|
7f7b81154f | ||
|
|
6372cec6b8 | ||
|
|
5ec875325e | ||
|
|
4d80709cfd | ||
|
|
2642c1b3d3 | ||
|
|
c4cd7e1982 | ||
|
|
b993c158d0 | ||
|
|
1b727042fe | ||
|
|
ec26736577 | ||
|
|
a766c5f2b7 | ||
|
|
3b2c804f23 | ||
|
|
486ed6f217 | ||
|
|
ae716a91cb | ||
|
|
d4195b4362 | ||
|
|
6ae0452d87 | ||
|
|
815117f653 | ||
|
|
1c0eb03b3b | ||
|
|
3249fba4a2 | ||
|
|
7c173dcc67 | ||
|
|
357f449e07 | ||
|
|
83560cbd1d | ||
|
|
1860f80cae | ||
|
|
4ea97c4fe4 | ||
|
|
ee738be681 | ||
|
|
e21b3155e5 | ||
|
|
c293ffd621 | ||
|
|
2fdaa7595c | ||
|
|
968a66f66b | ||
|
|
5992afb707 | ||
|
|
9aa83215c4 | ||
|
|
939a148812 | ||
|
|
4ce249e6ed | ||
|
|
422aaa80f3 | ||
|
|
b9a346ce7d | ||
|
|
64b92ed180 | ||
|
|
90fc5c9de4 | ||
|
|
7118c2f04b | ||
|
|
d66712ab42 | ||
|
|
c5f2158b85 | ||
|
|
a5c5353fbd | ||
|
|
d5a3f76234 | ||
|
|
7c18203845 | ||
|
|
d7c238723b | ||
|
|
f3e581d162 | ||
|
|
4f65a31eba | ||
|
|
35d8cffad4 | ||
|
|
0c46a723bd | ||
|
|
fcac99bc73 | ||
|
|
42208aa5fe | ||
|
|
7c3abea232 | ||
|
|
2c23bca913 | ||
|
|
4188d702ed | ||
|
|
318c77b934 | ||
|
|
b041c0080b | ||
|
|
ed93878851 | ||
|
|
df56c134e4 | ||
|
|
c51babfd27 | ||
|
|
8fdbfc3c95 | ||
|
|
4d378c3b14 | ||
|
|
81d5b7b5e5 | ||
|
|
accc082b91 | ||
|
|
4e4b5ddc58 | ||
|
|
4202826dfa | ||
|
|
b011ddd2d9 | ||
|
|
7972a156fc |
@@ -0,0 +1,31 @@
|
||||
*.ipynb
|
||||
*.pdf
|
||||
*.pyc
|
||||
*.rst
|
||||
*.sublime*
|
||||
*/*.pyc
|
||||
*/*/*.pyc
|
||||
*/*/*/*.pyc
|
||||
*/*/*/*/*.pyc
|
||||
*/*/*/*/*/*.pyc
|
||||
*/*/*/*/*/*/*.pyc
|
||||
*/*/*/*/*/*/*/*.pyc
|
||||
.cache/
|
||||
.git/
|
||||
.ipynb_checkpoints/
|
||||
.ruffus_history.sqlite
|
||||
bin/
|
||||
build/
|
||||
dist/
|
||||
htmlcov/
|
||||
include/
|
||||
lib/
|
||||
MANIFEST.in
|
||||
ocrmypdf.egg-info/
|
||||
staging/
|
||||
tests/cache/
|
||||
tests/output/
|
||||
tests/resources/private/
|
||||
tmp/
|
||||
venv-3.4/
|
||||
venv-3.5/
|
||||
@@ -0,0 +1,8 @@
|
||||
# Always use Unix convention for new lines
|
||||
* text eol=lf
|
||||
|
||||
# These files are binary and should be left untouched
|
||||
# (binary is a macro for -text -diff)
|
||||
*.jar binary
|
||||
*.pdf binary
|
||||
*.PDF binary
|
||||
+31
@@ -0,0 +1,31 @@
|
||||
# Development environment
|
||||
*.pyc
|
||||
*.sublime-*
|
||||
venv-3.4/
|
||||
venv-3.5/
|
||||
venv/
|
||||
pyvenv.cfg
|
||||
|
||||
# Package building
|
||||
*.egg-info/
|
||||
.cache/
|
||||
.eggs/
|
||||
build/
|
||||
dist/
|
||||
|
||||
# Automatically generated files
|
||||
ocrmypdf/lib/_*.py
|
||||
ocrmypdf/version.py
|
||||
|
||||
# Code coverage
|
||||
.coverage
|
||||
htmlcov/
|
||||
|
||||
# Testing
|
||||
log/
|
||||
/*.pdf
|
||||
.ipynb_checkpoints/
|
||||
tests/cache/
|
||||
tests/output/
|
||||
tests/resources/private
|
||||
tmp/
|
||||
+36
-23
@@ -1,29 +1,42 @@
|
||||
language: python
|
||||
|
||||
language: generic
|
||||
sudo: required
|
||||
dist: trusty
|
||||
cache:
|
||||
directories:
|
||||
- $HOME/.cache/pip
|
||||
- $HOME/.ccache
|
||||
- tarballs
|
||||
- tests/cache
|
||||
|
||||
python:
|
||||
- 3.4
|
||||
|
||||
before_install:
|
||||
- sudo add-apt-repository ppa:heyarje/libav-11 -y
|
||||
- sudo add-apt-repository ppa:alex-p/tesseract-ocr -y
|
||||
- sudo add-apt-repository ppa:evl.ms/evil -y # for a newer version (2.5.0) of pngquant
|
||||
- sudo apt-get update -qq
|
||||
- sudo apt-get install libleptonica-dev -y # required to build jbig2enc
|
||||
- sudo apt-get install zlib1g-dev -y # required to build jbig2enc
|
||||
# - sudo apt-get install imagemagick -y # required to convert logo to desktop icon
|
||||
# Ubuntu packages
|
||||
- sudo add-apt-repository ppa:evl.ms/precise -y # for Ghostscript 9.15
|
||||
- sudo add-apt-repository ppa:lyrasis/precise-backports -y # for Tesseract 3.03
|
||||
- sudo add-apt-repository ppa:b-eltzner/qpdfview-exp -y # for QPDF 5
|
||||
- sudo add-apt-repository ppa:itachi-san/ffmpeg -y # for libav 11.2 (for unpaper)
|
||||
- sudo apt-get update -qq # must go after all add-apt-repo
|
||||
- sudo apt-get install -y ghostscript tesseract-ocr tesseract-ocr-deu tesseract-ocr-eng tesseract-ocr-fra qpdf poppler-utils gcc libavformat-dev libavcodec-dev libavutil-dev automake make pkg-config xsltproc
|
||||
|
||||
# pip
|
||||
- pip install --upgrade pip
|
||||
|
||||
# Download, make and install unpaper (using ccache)
|
||||
- mkdir -p tarballs
|
||||
- "[ -f tarballs/unpaper-6.1.tar.xz ] || wget -q https://www.flameeyes.eu/files/unpaper-6.1.tar.xz -O tarballs/unpaper-6.1.tar.xz"
|
||||
- tar -xvf tarballs/unpaper-6.1.tar.xz
|
||||
- export PATH="/usr/lib/ccache:$PATH"
|
||||
- pushd unpaper-6.1 && ./configure --prefix=/usr && make -j && sudo make install && popd
|
||||
|
||||
install:
|
||||
- pip install -r requirements.txt
|
||||
- pip install -r test_requirements.txt
|
||||
|
||||
script:
|
||||
- export OCRMYPDF_VERSION=8.3.2
|
||||
- bash build-appimage.sh
|
||||
# remove previous installed libraries to ensure that the tests use the libraries of the AppImage
|
||||
- sudo apt-get remove libleptonica-dev zlib1g-dev -y
|
||||
- bash test/test-appimage.sh
|
||||
- wget https://github.com/probonopd/uploadtool/raw/master/upload.sh
|
||||
- python setup.py clean
|
||||
- python setup.py install
|
||||
- py.test
|
||||
|
||||
after_success:
|
||||
- bash upload.sh OCRmyPDF*.AppImage
|
||||
|
||||
branches:
|
||||
except:
|
||||
# Do not build tags that we create when we upload to GitHub Releases
|
||||
- /^(?i:continuous)/
|
||||
os:
|
||||
- linux
|
||||
|
||||
+61
@@ -0,0 +1,61 @@
|
||||
# OCRmyPDF
|
||||
#
|
||||
# VERSION 3.0.2
|
||||
FROM debian:stretch
|
||||
MAINTAINER James R. Barlow <jim@purplerock.ca>
|
||||
|
||||
# Add unprivileged user
|
||||
RUN useradd docker \
|
||||
&& mkdir /home/docker \
|
||||
&& chown docker:docker /home/docker
|
||||
|
||||
# Update system and install our dependencies
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
locales \
|
||||
ghostscript \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-deu tesseract-ocr-spa tesseract-ocr-eng tesseract-ocr-fra \
|
||||
qpdf \
|
||||
poppler-utils \
|
||||
python3 \
|
||||
python3-pip \
|
||||
python3-venv \
|
||||
python3-reportlab \
|
||||
python3-pil \
|
||||
python3-wheel \
|
||||
unpaper
|
||||
|
||||
# Enforce UTF-8
|
||||
# Borrowed from https://index.docker.io/u/crosbymichael/python/
|
||||
RUN dpkg-reconfigure locales && \
|
||||
locale-gen C.UTF-8 && \
|
||||
/usr/sbin/update-locale LANG=C.UTF-8
|
||||
ENV LC_ALL C.UTF-8
|
||||
|
||||
# Remove the junk
|
||||
RUN apt-get autoremove -y && apt-get clean -y
|
||||
RUN rm -rf /var/lib/apt/lists/* /tmp/* /var/tmp/* /root/*
|
||||
|
||||
# Set up a Python virtualenv and take all of the system packages, so we can
|
||||
# rely on the platform packages rather than importing GCC and compiling them
|
||||
RUN pyvenv /appenv \
|
||||
&& pyvenv --system-site-packages /appenv
|
||||
|
||||
COPY . /application/
|
||||
|
||||
# Install application and dependencies
|
||||
# In this arrangement Pillow and reportlab will be provided by the system
|
||||
RUN . /appenv/bin/activate; \
|
||||
pip install --upgrade pip \
|
||||
&& pip install --no-cache-dir /application \
|
||||
&& pip install --no-cache-dir -r /application/test_requirements.txt
|
||||
|
||||
USER docker
|
||||
WORKDIR /home/docker
|
||||
|
||||
ENV OCRMYPDF_TEST_OUTPUT=/tmp/test-output
|
||||
ENV OCRMYPDF_IN_DOCKER=1
|
||||
|
||||
# Must use array form of ENTRYPOINT
|
||||
# Non-array form does not append other arguments, because that is "intuitive"
|
||||
ENTRYPOINT ["/application/docker-wrapper.sh"]
|
||||
@@ -0,0 +1,13 @@
|
||||
# OCRmyPDF polyglot
|
||||
#
|
||||
# VERSION 3.0.2
|
||||
FROM jbarlow83/ocrmypdf:latest
|
||||
MAINTAINER James R. Barlow <jim@purplerock.ca>
|
||||
|
||||
# Update system and install our dependencies
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
tesseract-ocr-all
|
||||
|
||||
# Must use array form of ENTRYPOINT
|
||||
# Non-array form does not append other arguments, because that is "intuitive"
|
||||
ENTRYPOINT ["/application/docker-wrapper.sh"]
|
||||
+20
@@ -0,0 +1,20 @@
|
||||
Copyright (c) 2013-2015, The OCRmyPDF Authors
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a
|
||||
copy of this software and associated documentation files (the
|
||||
"Software"), to deal in the Software without restriction, including
|
||||
without limitation the rights to use, copy, modify, merge, publish,
|
||||
distribute, sublicense, and/or sell copies of the Software, and to
|
||||
permit persons to whom the Software is furnished to do so, subject to
|
||||
the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included
|
||||
in all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||
OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
|
||||
CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
|
||||
TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||
SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
@@ -0,0 +1,3 @@
|
||||
recursive-exclude tests/output *
|
||||
include requirements.txt
|
||||
include test_requirements.txt
|
||||
@@ -0,0 +1,6 @@
|
||||
#!/bin/sh
|
||||
##############################################################################
|
||||
# Copyright (c) 2013-14: fritz-hh from Github (https://github.com/fritz-hh)
|
||||
##############################################################################
|
||||
|
||||
python3 -m ocrmypdf.main "$@"
|
||||
@@ -1,40 +0,0 @@
|
||||
# OCRmyPDF-AppImage [](https://travis-ci.com/FPille/OCRmyPDF-AppImage)
|
||||
[AppImage][APPIMAGE] for [OCRmyPDF][OCRMYPDF]
|
||||
|
||||
## Usage
|
||||
Download OCRmyPDF*.AppImage, make it executable and run it.
|
||||
```
|
||||
wget https://github.com/FPille/OCRmyPDF-AppImage/releases/download/continuous/OCRmyPDF-8.3.2-x86_64.AppImage
|
||||
chmod +x OCRmyPDF*.AppImage
|
||||
./OCRmyPDF*.AppImage --help
|
||||
```
|
||||
|
||||
Beside OCRmyPDF additional command line programs can be run with this AppImage like:
|
||||
* ghostscript
|
||||
* img2pdf
|
||||
* pngquant
|
||||
* python3.6
|
||||
* qpdf
|
||||
* tesseract
|
||||
* unpaper
|
||||
|
||||
Just use the program name as first parameter plus options:
|
||||
```
|
||||
./OCRmyPDF*.AppImage tesseract -v
|
||||
tesseract 4.1.0
|
||||
leptonica-1.76.0
|
||||
libjpeg 8d (libjpeg-turbo 1.3.0) : libpng 1.2.50 : libtiff 4.0.3 : zlib 1.2.11 : libwebp 0.4.0 : libopenjp2 2.3.0
|
||||
Found AVX2
|
||||
Found AVX
|
||||
Found SSE
|
||||
```
|
||||
Or create a symlink for the corresponding program:
|
||||
```
|
||||
ln -s OCRmyPDF*.AppImage tesseract
|
||||
./tesseract --list-langs
|
||||
```
|
||||
|
||||
|
||||
[APPIMAGE]: https://appimage.org
|
||||
[OCRMYPDF]: https://github.com/jbarlow83/OCRmyPDF
|
||||
|
||||
+258
@@ -0,0 +1,258 @@
|
||||
OCRmyPDF
|
||||
========
|
||||
|
||||
OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to
|
||||
be searched.
|
||||
|
||||
Main features
|
||||
-------------
|
||||
|
||||
- Generates a searchable
|
||||
`PDF/A <https://en.wikipedia.org/?title=PDF/A>`__ file from a regular PDF
|
||||
- Places OCR text accurately below the image to ease copy / paste
|
||||
- Keeps the exact resolution of the original embedded images
|
||||
- When possible, inserts OCR information as a "lossless" operation without rendering vector information
|
||||
- Keeps file size about the same
|
||||
- If requested deskews and/or cleans the image before performing OCR
|
||||
- Validates input and output files
|
||||
- Provides debug mode to enable easy verification of the OCR results
|
||||
- Processes pages in parallel when more than one CPU core is
|
||||
available
|
||||
- Uses `Tesseract OCR <https://github.com/tesseract-ocr/tesseract>`__ engine
|
||||
- Supports the `39 languages <https://code.google.com/p/tesseract-ocr/downloads/list>`__ recognized by Tesseract
|
||||
- Battle-tested on thousands of PDFs, a test suite and continuous integration
|
||||
|
||||
For details: please consult the `release notes <RELEASE_NOTES.rst>`__.
|
||||
|
||||
Motivation
|
||||
----------
|
||||
|
||||
I searched the web for a free command line tool to OCR PDF files on
|
||||
Linux/UNIX: I found many, but none of them were really satisfying.
|
||||
|
||||
- Either they produced PDF files with misplaced text under the image (making copy/paste impossible)
|
||||
- Or they did not display correctly some escaped HTML characters located in the hOCR file produced by the OCR engine
|
||||
- Or they changed the resolution of the embedded images
|
||||
- Or they generated PDF files having a ridiculous big size
|
||||
- Or they crashed when trying to OCR some of my PDF files
|
||||
- Or they did not produce valid PDF files (even though they were readable with my current PDF reader)
|
||||
- On top of that none of them produced PDF/A files (format dedicated for long time storage)
|
||||
|
||||
... so I decided to develop my own tool (using various existing scripts
|
||||
as an inspiration)
|
||||
|
||||
Installation
|
||||
------------
|
||||
|
||||
Download OCRmyPDF here: https://github.com/jbarlow83/OCRmyPDF/releases
|
||||
|
||||
You can install it to a Python virtual environment or system-wide.
|
||||
|
||||
Installing the Docker container
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
For many users, installing the Docker container will be easier than installing all of OCRmyPDF's dependencies. For Windows, it is the only option.
|
||||
|
||||
If you have `Docker <https://docs.docker.com/>`__ installed on your system, you can install
|
||||
a Docker container of the latest release.
|
||||
|
||||
Follow the Docker installation instructions for your platform. If you can run this command
|
||||
successfully, your system is ready to download and execute the image::
|
||||
|
||||
docker run hello-world
|
||||
|
||||
OCRmyPDF will use all available CPU cores. By default, the VirtualBox machine instance on Windows and OS X has only a single CPU core enabled. Use the VirtualBox Manager to determine the name of your Docker container host, and then follow these optional steps to enable multiple CPUs::
|
||||
|
||||
# Optional
|
||||
docker-machine stop "yourVM"
|
||||
VBoxManage modifyvm "yourVM" --cpus 2 # or whatever number of core is desired
|
||||
docker-machine start "yourVM"
|
||||
eval $(docker-machine env "yourVM")
|
||||
|
||||
Assuming you have a Docker engine running somewhere, you can run these commands to download
|
||||
the image::
|
||||
|
||||
docker pull jbarlow83/ocrmypdf
|
||||
|
||||
Then tag it to give a more convenient name, just ocrmypdf::
|
||||
|
||||
docker tag jbarlow83/ocrmypdf ocrmypdf
|
||||
|
||||
You can then run using the command::
|
||||
|
||||
docker run ocrmypdf --help
|
||||
|
||||
To execute the OCRmyPDF on a local file, you must `provide a writable volume to the Docker image <https://docs.docker.com/userguide/dockervolumes/>`__, such as this in this template::
|
||||
|
||||
docker run -v "$(pwd):/home/docker" <other docker arguments> ocrmypdf <your arguments to ocrmypdf>
|
||||
|
||||
In this worked example, the current working directory contains an input file called ``test.pdf`` and the output will go to ``output.pdf``::
|
||||
|
||||
docker run -v "$(pwd):/home/docker" ocrmypdf --skip-text test.pdf output.pdf
|
||||
|
||||
Note that ``ocrmypdf`` has its own separate ``-v VERBOSITYLEVEL`` argument to control debug verbosity. All Docker arguments should before the ``ocrmypdf`` container name and all arguments to ``ocrmypdf`` should be listed after.
|
||||
|
||||
Installing on Mac OS X
|
||||
~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
These instructions probably work on all Mac OS X versions later than 10.7 (Lion). OCRmyPDF is known to work on Yosemite and El Capitan, and regularly tested on El Capitan.
|
||||
|
||||
If it's not already present, `install Homebrew <http://brew.sh/>`__.
|
||||
|
||||
Update Homebrew::
|
||||
|
||||
brew update
|
||||
|
||||
Install or upgrade the required Homebrew packages, if any are missing::
|
||||
|
||||
brew install libpng openjpeg jbig2dec # image libraries
|
||||
brew install qpdf
|
||||
brew install ghostscript
|
||||
brew install python3
|
||||
brew install libxml2
|
||||
brew install leptonica
|
||||
brew install tesseract
|
||||
|
||||
It is also recommended that install Pillow and confirm it can read and write JPEG and PNG files::
|
||||
|
||||
pip3 install --upgrade pip
|
||||
pip3 install --upgrade pillow
|
||||
|
||||
Sometimes, the Python imaging library (Pillow) can end up being compiled and installed without support for JPEG and PNG files. (Arguably, this is an unfixed bug in Pillow's installer.) To confirm that Pillow is compiled correctly and can access JPEG and PNG files, try this command::
|
||||
|
||||
python3 -c "from PIL import Image; im = Image.new('1', (1, 1)); im.save('test.png'); im.save('test.jpg')"
|
||||
|
||||
If you have trouble getting Pillow to access JPEG and PNG files, `review the installation instructions <https://pillow.readthedocs.org/installation.html>`__.
|
||||
|
||||
You can then install OCRmyPDF from PyPI::
|
||||
|
||||
pip3 install ocrmypdf
|
||||
|
||||
The command line program should now be available::
|
||||
|
||||
ocrmypdf --help
|
||||
|
||||
Installing on Ubuntu 14.04 LTS
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
Installing on Ubuntu 14.04 LTS (trusty) is more difficult than other options, because of certain bugs in Python package installation.
|
||||
|
||||
Update apt-get::
|
||||
|
||||
sudo apt-get update
|
||||
sudo apt-get upgrade
|
||||
|
||||
Install system dependencies::
|
||||
|
||||
sudo apt-get install \
|
||||
zlib1g-dev \
|
||||
libjpeg-dev \
|
||||
ghostscript \
|
||||
tesseract-ocr \
|
||||
qpdf \
|
||||
unpaper \
|
||||
python3-pip \
|
||||
python3-pil \
|
||||
python3-pytest \
|
||||
python3-reportlab
|
||||
|
||||
If you wish install OCRmyPDF to the system Python, then install as follows (note this installs new packages
|
||||
into your system Python, which could interfere with other programs)::
|
||||
|
||||
sudo pip3 install ocrmypdf
|
||||
|
||||
If you wish to install OCRmyPDF to a virtual environment to isolate system Python from modified, you can
|
||||
follow these steps. This includes a workaround `for a known, unresolved issue in Ubuntu 14.04's ensurepip
|
||||
package <http://www.thefourtheye.in/2014/12/Python-venv-problem-with-ensurepip-in-Ubuntu.html>`__::
|
||||
|
||||
sudo apt-get install python3-venv
|
||||
python3 -m venv venv-ocrmypdf --without-pip
|
||||
source venv-ocrmypdf/bin/activate
|
||||
wget -O - -o /dev/null https://bootstrap.pypa.io/get-pip.py | python
|
||||
deactivate
|
||||
pyvenv --system-site-packages venv-ocrmypdf
|
||||
source venv-ocrmypdf/bin/activate
|
||||
pip install ocrmypdf
|
||||
|
||||
Ubuntu 14.04 only installs ``unpaper`` version 0.4.2, which is not supported by OCRmyPDF because it is produces invalid output. This program is an optional dependency, and provides page deskewing and cleaning. See `Dockerfile <Dockerfile>`__ for an example of how to building unpaper 6.1 from source. If you choose to install unpaper later, OCRmyPDF will use the foremost version on the system PATH.
|
||||
|
||||
Installing on Windows
|
||||
~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
Direct installation on Windows is not possible. Install the Docker container as described above.
|
||||
|
||||
|
||||
Installing HEAD revision from sources
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
If you have ``git`` and ``python3.4`` or ``python3.5`` installed, you can install from source. When the ``pip`` installer runs,
|
||||
it will alert you if dependencies are missing.
|
||||
|
||||
First, clone the HEAD revision::
|
||||
|
||||
git clone -b master https://github.com/jbarlow83/OCRmyPDF.git
|
||||
cd OCRmyPDF
|
||||
|
||||
To install the HEAD revision from sources::
|
||||
|
||||
pip3 install .
|
||||
|
||||
Or, to install in `development mode <https://pythonhosted.org/setuptools/setuptools.html#development-mode>`__,
|
||||
allowing customization of OCRmyPDF, use the ``-e`` flag::
|
||||
|
||||
pip3 install -e .
|
||||
|
||||
On certain Linux distributions such as Ubuntu, you may need to use
|
||||
run the install command as superuser::
|
||||
|
||||
sudo pip3 install [-e] .
|
||||
|
||||
Note that this will alter your system's Python distribution. If you prefer
|
||||
to not install as superuser, you can install the package in a Python virtual environment::
|
||||
|
||||
git clone -b master https://github.com/jbarlow83/OCRmyPDF.git
|
||||
pyvenv venv
|
||||
source venv/bin/activate
|
||||
cd OCRmyPDF
|
||||
pip3 install .
|
||||
|
||||
However, ``ocrmypdf`` will only be accessible on the system PATH after
|
||||
you activate the virtual environment.
|
||||
|
||||
To run the program::
|
||||
|
||||
ocrmypdf --help
|
||||
|
||||
If not yet installed, the script will notify you about dependencies that
|
||||
need to be installed. The script requires specific versions of the
|
||||
dependencies. Older version than the ones mentioned in the release notes
|
||||
are likely not to be compatible to OCRmyPDF.
|
||||
|
||||
Support
|
||||
-------
|
||||
|
||||
In case you detect an issue, please:
|
||||
|
||||
- Check if your issue is already known
|
||||
- If no problem report exists on github, please create one here:
|
||||
https://github.com/jbarlow83/OCRmyPDF/issues
|
||||
- Describe your problem thoroughly
|
||||
- Append the console output of the script when running the debug mode
|
||||
(``-v 1`` option)
|
||||
- If possible provide your input PDF file as well as the content of the
|
||||
temporary folder (using a file sharing service like Dropbox)
|
||||
|
||||
Press & Media
|
||||
-------------
|
||||
|
||||
- `c't 1-2014, page 59 <http://www.heise.de/ct/inhalt/2014/1/58/>`__:
|
||||
Detailed presentation of OCRmyPDF v1.0 in the leading German IT
|
||||
magazine c't
|
||||
- `heise Open Source, 09/2014: Texterkennung mit
|
||||
OCRmyPDF <http://www.heise.de/-2356670>`__
|
||||
|
||||
Disclaimer
|
||||
----------
|
||||
|
||||
The software is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR
|
||||
CONDITIONS OF ANY KIND, either express or implied.
|
||||
@@ -0,0 +1,569 @@
|
||||
RELEASE NOTES
|
||||
=============
|
||||
|
||||
Please always read this file before installing the package
|
||||
|
||||
Download software here: https://github.com/jbarlow83/OCRmyPDF/tags
|
||||
|
||||
|
||||
v3.2:
|
||||
=========
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- Lossless reconstruction: when possible, OCRmyPDF will inject text layers without
|
||||
otherwise manipulating the content and layout of a PDF page. For example, a PDF containing a mix
|
||||
of vector and raster content would see the vector content preserved. Images may still be transcoded
|
||||
during PDF/A conversion. (``--deskew`` and ``--clean-final`` disable this mode, necessarily.)
|
||||
- New argument ``--tesseract-pagesegmode`` allows you to pass page segmentation arguments to Tesseract OCR.
|
||||
This helps for two column text and other situations that confuse Tesseract.
|
||||
- Added a new "polyglot" version of the Docker image, that generates Tesseract with all languages packs installed,
|
||||
for the polyglots among us. It is much larger.
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- JPEG transcoding quality is now 95 instead of the default 75. Bigger file sizes for less degradation.
|
||||
|
||||
|
||||
|
||||
v3.1.1:
|
||||
=======
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- Fixed bug that caused incorrect page size and DPI calculations on documents with mixed page sizes
|
||||
|
||||
v3.1:
|
||||
=====
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- Default output format is now PDF/A-2b instead of PDF/A-1b
|
||||
- Python 3.5 and OS X El Capitan are now supported platforms - no changes were
|
||||
needed to implement support
|
||||
- Improved some error messages related to missing input files
|
||||
- Fixed issue #20 - uppercase .PDF extension not accepted
|
||||
- Fixed an issue where OCRmyPDF failed to text that certain pages contained previously OCR'ed text,
|
||||
such as OCR text produced by Tesseract 3.04
|
||||
- Inserts /Creator tag into PDFs so that errors can be traced back to this project
|
||||
- Added new option ``--pdf-renderer=auto``, to let OCRmyPDF pick the best PDF renderer.
|
||||
Currently it always chooses the 'hocrtransform' renderer but that behavior may change.
|
||||
- Set up Travis CI automatic integration testing
|
||||
|
||||
v3.0:
|
||||
=====
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- Easier installation with a Docker container or Python's ``pip`` package manager
|
||||
- Eliminated many external dependencies, so it's easier to setup
|
||||
- Now installs ``ocrmypdf`` to ``/usr/local/bin`` or equivalent for system-wide
|
||||
access and easier typing
|
||||
- Improved command line syntax and usage help (``--help``)
|
||||
- Tesseract 3.03+ PDF page rendering can be used instead for better positioning
|
||||
of recognized text (``--pdf-renderer tesseract``)
|
||||
- PDF metadata (title, author, keywords) are now transferred to the
|
||||
output PDF
|
||||
- PDF metadata can also be set from the command line (``--title``, etc.)
|
||||
- Automatic repairs malformed input PDFs if possible
|
||||
- Added test cases to confirm everything is working
|
||||
- Added option to skip extremely large pages that take too long to OCR and are
|
||||
often not OCRable (e.g. large scanned maps or diagrams); other pages are still
|
||||
processed (``--skip-big``)
|
||||
- Added option to kill Tesseract OCR process if it seems to be taking too long on
|
||||
a page, while still processing other pages (``--tesseract-timeout``)
|
||||
- Less common colorspaces (CMYK, palette) are now supported by conversion to RGB
|
||||
- Multiple images on the same PDF page are now supported
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- New, robust rewrite in Python 3.4+ with ruffus_ pipelines
|
||||
- Now uses Ghostscript 9.14's improved color conversion model to preserve PDF colors
|
||||
- OCR text is now rendered in the PDF as invisible text. Previous versions of OCRmyPDF
|
||||
incorrectly rendered visible text with an image on top.
|
||||
- All "tasks" in the pipeline can be executed in parallel on any
|
||||
available CPUs, increasing performance
|
||||
- The ``-o DPI`` argument has been phased out, in favor of ``--oversample DPI``, in
|
||||
case we need ``-o OUTPUTFILE`` in the future
|
||||
- Removed several dependencies, so it's easier to install. We no
|
||||
longer use:
|
||||
|
||||
- GNU parallel_
|
||||
- ImageMagick_
|
||||
- Python 2.7
|
||||
- Poppler
|
||||
- MuPDF_ tools
|
||||
- shell scripts
|
||||
- Java and JHOVE_
|
||||
- libxml2
|
||||
|
||||
- Some new external dependencies are required or optional, compared to v2.x:
|
||||
|
||||
- Ghostscript 9.14+
|
||||
- qpdf_ 5.0.0+
|
||||
- Unpaper_ 6.1 (optional)
|
||||
- some automatically managed Python packages
|
||||
|
||||
.. _ruffus: http://www.ruffus.org.uk/index.html
|
||||
.. _parallel: https://www.gnu.org/software/parallel/
|
||||
.. _ImageMagick: http://www.imagemagick.org/script/index.php
|
||||
.. _MuPDF: http://mupdf.com/docs/
|
||||
.. _qpdf: http://qpdf.sourceforge.net/
|
||||
.. _Unpaper: https://github.com/Flameeyes/unpaper
|
||||
.. _JHOVE: http://jhove.sourceforge.net/
|
||||
|
||||
Release candidates
|
||||
------------------
|
||||
|
||||
- rc9:
|
||||
|
||||
- fix issue #118: report error if ghostscript iccprofiles are missing
|
||||
- fixed another issue related to #111: PDF rasterized to palette file
|
||||
- add support image files with a palette
|
||||
- don't try to validate PDF file after an exception occurs
|
||||
|
||||
- rc8:
|
||||
|
||||
- fix issue #111: exception thrown if PDF is missing DocumentInfo dictionary
|
||||
|
||||
- rc7:
|
||||
|
||||
- fix error when installing direct from pip, "no such file 'requirements.txt'"
|
||||
|
||||
- rc6:
|
||||
|
||||
- dropped libxml2 (Python lxml) since Python 3's internal XML parser is sufficient
|
||||
- set up Docker container
|
||||
- fix Unicode errors if recognized text contains Unicode characters and system locale is not UTF-8
|
||||
|
||||
- rc5:
|
||||
|
||||
- dropped Java and JHOVE in favour of qpdf
|
||||
- improved command line error output
|
||||
- additional tests and bug fixes
|
||||
- tested on Ubuntu 14.04 LTS
|
||||
|
||||
- rc4:
|
||||
|
||||
- dropped MuPDF in favour of qpdf
|
||||
- fixed some installer issues and errors in installation instructions
|
||||
- improve performance: run Ghostscript with multithreaded rendering
|
||||
- improve performance: use multiple cores by default
|
||||
- bug fix: checking for wrong exception on process timeout
|
||||
|
||||
- rc3: skipping version number intentionally to avoid confusion with Tesseract
|
||||
- rc2: first release for public testing to test-PyPI, Github
|
||||
- rc1: testing release process
|
||||
|
||||
Compatibility notes
|
||||
-------------------
|
||||
|
||||
- ``./OCRmyPDF.sh`` script is still available for now
|
||||
- Stacking the verbosity option like ``-vvv`` is no longer supported
|
||||
|
||||
- The configuration file ``config.sh`` has been removed. Instead, you can
|
||||
feed a file to the arguments for common settings:
|
||||
|
||||
::
|
||||
|
||||
ocrmypdf input.pdf output.pdf @settings.txt
|
||||
|
||||
where ``settings.txt`` contains *one argument per line*, for example:
|
||||
|
||||
::
|
||||
|
||||
-l
|
||||
deu
|
||||
--author
|
||||
A. Merkel
|
||||
--pdf-renderer
|
||||
tesseract
|
||||
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- Handling of filenames containing spaces: fixed
|
||||
|
||||
Notes and known issues
|
||||
----------------------
|
||||
|
||||
- Some dependencies may work with lower versions than tested, so try
|
||||
overriding dependencies if they are "in the way" to see if they work.
|
||||
|
||||
- ``--pdf-renderer tesseract`` will output files with an incorrect page size in Tesseract 3.03,
|
||||
due to a bug in Tesseract.
|
||||
|
||||
- PDF files containing "inline images" are not supported and won't be for the 3.0 release. Scanned
|
||||
images almost never contain inline images.
|
||||
|
||||
|
||||
v2.2-stable (2014-09-29):
|
||||
=========================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- None
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- Update to jhove v1.11
|
||||
- Request the python library reportlab v3.0 or newer (So that we could remove a patch to the previous version of reportlab leading to issues for some users)
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- Fix bug on Mac OS X (resolution of simlink to OCRmyPDF.sh script) (thanks to jbarlow83)
|
||||
- Check if the input pdf file exists before to continue
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.2
|
||||
- Dependencies:
|
||||
|
||||
- parallel 20140822
|
||||
- poppler-utils 0.24.5
|
||||
- ImageMagick 6.8.9-4 2014-09-17
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.8
|
||||
- ghostcript (gs): 9.06
|
||||
- java: openjdk version "1.7.0_65"
|
||||
|
||||
|
||||
v2.1-stable (2014-09-20):
|
||||
=========================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- None
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- None
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- Allow execution via simlink
|
||||
- Add support for tesseract 3.03
|
||||
- Add support for newer version of reportlab
|
||||
- Lowered minimum version of gnu parallel
|
||||
- Various typo
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.1
|
||||
- Dependencies:
|
||||
- parallel 20130222
|
||||
- poppler-utils 0.22.2
|
||||
- ImageMagick 6.8.0-7 2013-03-30
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.3
|
||||
- ghoscript (gs): 9.06
|
||||
- java: openjdk version "1.7.0\_17"
|
||||
|
||||
v2.0-stable (2014-01-25):
|
||||
=========================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- Check if the language(s) passed using the -l option is supported by
|
||||
tesseract (fixes #60)
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- Allow OCRmyPDF to be used with tesseract 3.02.01, even though OCR
|
||||
might fail for few PDF file (see issue #28). Rationale: For some
|
||||
linux distribution, no newer version than tesseract 3.02.01 is
|
||||
available
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- More robust algorithm for checking the version of the installed
|
||||
tesseract package
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.1
|
||||
- Dependencies:
|
||||
- parallel 20130222
|
||||
- poppler-utils 0.22.2
|
||||
- ImageMagick 6.8.0-7 2013-03-30
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.3
|
||||
- ghoscript (gs): 9.06
|
||||
- java: openjdk version "1.7.0\_17"
|
||||
|
||||
v2.0-rc2 (2014-01-16):
|
||||
======================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- None
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- Size reduction of final PDF file: (fixes #50)
|
||||
- Support for monochrome (Black&White) images (massive size reduction
|
||||
in final PDF: >80%)
|
||||
- Reduced size of grayscale images (by 13% on test PDF file)
|
||||
- Preventing fi, fl ligatures does not require anymore to pass an
|
||||
additional config file to tesseract using the -C option (fixes #58)
|
||||
- Location of temporary folder according to content of environment
|
||||
variable TMPDIR.
|
||||
- Dependency to pdftk removed
|
||||
- Check for compatible versions of dependencies: (fixes #51)
|
||||
- parallel and tesseract
|
||||
- python libraries reportlab and lxml
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- Improved portability with various shells (dash, bash, tcsh) and OS
|
||||
(FreeBSD, MAC OSX, Linux) (fixes #59)
|
||||
- Corrected bug in case the input PDF file contains a space character
|
||||
(fixes #48)
|
||||
- Prevent spurious error message in case there is no image in a PDF
|
||||
page
|
||||
- Prevent collision of temporary folder names (fixes #57)
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.1
|
||||
- Dependencies:
|
||||
- parallel 20130222
|
||||
- poppler-utils 0.22.2
|
||||
- ImageMagick 6.8.0-7 2013-03-30
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.3
|
||||
- ghoscript (gs): 9.06
|
||||
- java: openjdk version "1.7.0\_17"
|
||||
|
||||
v2.0-rc1 (2014-01-07):
|
||||
======================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- Huge performance improvement on machines having multiple CPU/cores
|
||||
(processing of several pages concurrently) (fixes #18)
|
||||
- By default prevent from processing a PDF file already containing
|
||||
fonts (i.e. text)(it can be overridden with the -f flag) (fixes #16)
|
||||
- Warn if the resolution is too low to get reasonable OCR results
|
||||
(fixes #37)
|
||||
- New option (-o) to perform automatic oversampling if the image
|
||||
resolution is too low. This can improve OCR results.
|
||||
- Warn if using a tesseract version older than v3.02.02 (as older
|
||||
versions are known to produce invalid output) (fixes #41)
|
||||
- Echo version of the installed dependencies (e.g. tesseract) in debug
|
||||
mode in order to ease support (fixes #35)
|
||||
- Echo the arguments passed to the script in debug mode to ease support
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- In debug mode: The debug page is now placed after the respective
|
||||
"normal" page
|
||||
- Reduced disk space usage in temporary folder if -d (deskew) or -c
|
||||
(cleanup) options are not selected
|
||||
- New file src/config.sh containing various configuration parameters
|
||||
- Documentation of the tesseract config file "tess-cfg/no\_ligature"
|
||||
improved
|
||||
- Improved consistency of the temporary file names
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- Improved robustness:
|
||||
- in case vertical resolution differs from horizontal resolution (fixes
|
||||
#38)
|
||||
- in case a PDF page contains more than one image (fixes #36)
|
||||
- Fix a problem occurring if python 3 is the standard interpreter
|
||||
(fixes #33)
|
||||
- Fix a problem occurring if the input PDF file contains special
|
||||
characters like "#" (fixes #34)
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.1
|
||||
- Dependencies:
|
||||
- parallel 20130222
|
||||
- poppler-utils 0.22.2
|
||||
- ImageMagick 6.8.0-7 2013-03-30
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.3
|
||||
- pdftk 1.45
|
||||
- ghoscript (gs): 9.06
|
||||
- java: openjdk version "1.7.0\_17"
|
||||
|
||||
v1.1-stable (2014-01-06):
|
||||
=========================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- N/A
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- N/A
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- Fixed syntax error (bashism) leading to an error message on certain
|
||||
systems (fixes #42)
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.1
|
||||
- Dependencies:
|
||||
- poppler-utils 0.22.2
|
||||
- ImageMagick 6.8.0-7 2013-03-30
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.3
|
||||
- pdftk 1.45
|
||||
- ghoscript (gs): 9.06
|
||||
- java: openjdk version "1.7.0\_17"
|
||||
|
||||
v1.0-stable (2013-05-06):
|
||||
=========================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- In debug mode: compute and echo time required for processing (fixes
|
||||
#26)
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- Removed feature to add metadata in final pdf file (because it lead to
|
||||
to final PDF file that does not comply to the PDF/A-1 format)
|
||||
- Removed feature to set same owner & permissions in final PDF file
|
||||
than in input file
|
||||
- Removed many unused jhove files (e.g. documentation, \*.java and
|
||||
\*.class files)
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- Correction to handle correctly path and input PDF files having spaces
|
||||
(fixes #31)
|
||||
- Resolutions (x/y) that are nearly equal are now supported (fixes #25)
|
||||
- Fix compatibility issue with Ubuntu server 12.04 / Ubuntu server
|
||||
10.04 / Linux Mint 13 Maya and probably other Linux distributions
|
||||
(fixes #27)
|
||||
- Commit missing jhove files (\*.jar mainly) due to wrong .gitignore
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.1
|
||||
- Dependencies:
|
||||
- poppler-utils 0.22.2
|
||||
- ImageMagick 6.8.0-7 2013-03-30
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.3
|
||||
- pdftk 1.45
|
||||
- ghoscript (gs): 9.06
|
||||
- java: openjdk version "1.7.0\_17"
|
||||
|
||||
v1.0-rc2 (2013-04-29):
|
||||
======================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- Keep temporary files if debug mode is set (fixes #22)
|
||||
- Set same owner & permissions in final PDF file than in input file
|
||||
(fixes #9)
|
||||
- Added metadata in final pdf file (fixes #4)
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- N/A
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- Fixed wrong image cropping when deskew option is activated
|
||||
- Exit with error message if page size is not found in hocr file (fixes
|
||||
#21)
|
||||
- Various minor fixes in log messages
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.1
|
||||
- Dependencies:
|
||||
- poppler-utils 0.22.2
|
||||
- ImageMagick 6.8.0-7 2013-03-30
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.3
|
||||
- pdftk 1.45
|
||||
- ghoscript (gs): 9.06
|
||||
- java: openjdk version "1.7.0\_17"
|
||||
|
||||
v1.0-rc1 (2013-04-26):
|
||||
======================
|
||||
|
||||
New features
|
||||
------------
|
||||
|
||||
- First release candidate
|
||||
|
||||
Changes
|
||||
-------
|
||||
|
||||
- N/A
|
||||
|
||||
Fixes
|
||||
-----
|
||||
|
||||
- N/A
|
||||
|
||||
Tested with
|
||||
-----------
|
||||
|
||||
- Operating system: FreeBSD 9.1
|
||||
- Dependencies:
|
||||
- poppler-utils 0.22.2
|
||||
- ImageMagick 6.8.0-7 2013-03-30
|
||||
- Unpaper 0.3
|
||||
- tesseract 3.02.02
|
||||
- Python 2.7.3
|
||||
- pdftk 1.45
|
||||
- ghoscript (gs): 9.06
|
||||
- java: openjdk version "1.7.0\_17"
|
||||
@@ -1,105 +0,0 @@
|
||||
#! /bin/bash
|
||||
|
||||
HERE="$(dirname "$(readlink -f "${0}")")"
|
||||
|
||||
export PATH="$HERE/usr/bin:$HERE/usr/local/bin:$HERE/usr/python/bin:$PATH"
|
||||
export LD_PRELOAD="$HERE/usr/lib/liblept.so.5"
|
||||
export LD_LIBRARY_PATH="$HERE/usr/lib:$HERE/usr/lib/x86_64-linux-gnu:$LD_LIBRARY_PATH"
|
||||
export TESSDATA_PREFIX="$HERE/usr/share/tesseract-ocr/4.00/tessdata"
|
||||
export GS_LIB="$HERE/usr/share/ghostscript/9.26/lib:$HERE/usr/share/ghostscript/9.26/Resource:$HERE/usr/share/ghostscript/9.26/Resource/Init"
|
||||
|
||||
# Allow the AppImage to be symlinked to e.g., /usr/bin/commandname
|
||||
# or called with ./Some*.AppImage commandname ...
|
||||
# refer to https://github.com/AppImage/AppImageKit/wiki/Bundling-command-line-tools
|
||||
|
||||
if [ ! -z "$APPIMAGE" ] ; then
|
||||
BINARY_NAME=$(basename "$ARGV0")
|
||||
else
|
||||
BINARY_NAME=$(basename "$0")
|
||||
export APPDIR="$HERE" # required for the wrapper scripts of linuxdeploy-plugin-python
|
||||
fi
|
||||
|
||||
usage() {
|
||||
echo "
|
||||
==============================================================================
|
||||
AppImage for OCRmyPDF
|
||||
==============================================================================
|
||||
OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be
|
||||
searched or copy-pasted.
|
||||
|
||||
usage:
|
||||
$ARGV0 [ocrmypdf] [--help] [--list-programs]
|
||||
[--list-licenses] [--show-license]
|
||||
|
||||
ocrmypdf execute OCRmyPDF
|
||||
|
||||
--help show this help message
|
||||
|
||||
--list-programs list all programs contained in this AppImage
|
||||
|
||||
--list-licenses list all licenses contained in this AppImage
|
||||
|
||||
--show-license [LICENSE] show content of license file
|
||||
"
|
||||
}
|
||||
|
||||
if [ "$1" == "--help" ] ; then
|
||||
usage
|
||||
exit $?
|
||||
fi
|
||||
|
||||
if [ "$1" == "--list-programs" ] ; then
|
||||
pushd "$HERE"
|
||||
echo ""
|
||||
echo "Run \"$ARGV0\" with one of the following arguments to run the respective program."
|
||||
echo ""
|
||||
find . -type f -perm /111 ! -path '*/lib/*' -execdir basename {} ";" | sort -u | column
|
||||
echo ""
|
||||
exit $?
|
||||
fi
|
||||
|
||||
if [ "$1" == "--list-licenses" ] ; then
|
||||
pushd "$HERE"
|
||||
echo ""
|
||||
echo "Run \"$ARGV0\" with one of the following arguments to display the respective license file."
|
||||
echo ""
|
||||
find . -type f \( ! -path '*/tesseract-ocr-*' -o -path '*/tesseract-ocr-eng/*' \) \
|
||||
\( -iname "license*" -o -iname "*copyright*" -o -iname "*copying*" \) -printf "--show-license %P\n" | sort | column
|
||||
echo ""
|
||||
exit $?
|
||||
fi
|
||||
|
||||
if [ "$1" == "--show-license" ] ; then
|
||||
pushd "$HERE"
|
||||
shift
|
||||
if [ -f "$1" ] ; then
|
||||
less -N "$1"
|
||||
exit $?
|
||||
else
|
||||
echo "\"$1\" is not a valid license file path."
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
if [ ! -z "$1" ] && [ -e "$HERE/bin/$1" ] ; then
|
||||
MAIN="$HERE/bin/$1" ; shift
|
||||
elif [ ! -z "$1" ] && [ -e "$HERE/usr/bin/$1" ] ; then
|
||||
MAIN="$HERE/usr/bin/$1" ; shift
|
||||
elif [ ! -z "$1" ] && [ -e "$HERE/usr/python/bin/$1" ] ; then
|
||||
MAIN="$HERE/usr/python/bin/$1" ; shift
|
||||
elif [ ! -z "$1" ] && [ -e "$HERE/usr/local/bin/$1" ] ; then
|
||||
MAIN="$HERE/usr/local/bin/$1" ; shift
|
||||
elif [ -e "$HERE/bin/$BINARY_NAME" ] ; then
|
||||
MAIN="$HERE/bin/$BINARY_NAME"
|
||||
elif [ -e "$HERE/usr/bin/$BINARY_NAME" ] ; then
|
||||
MAIN="$HERE/usr/bin/$BINARY_NAME"
|
||||
elif [ -e "$HERE/usr/python/bin/$BINARY_NAME" ] ; then
|
||||
MAIN="$HERE/usr/python/bin/$BINARY_NAME"
|
||||
elif [ -e "$HERE/usr/local/bin/$BINARY_NAME" ] ; then
|
||||
MAIN="$HERE/usr/local/bin/$BINARY_NAME"
|
||||
else
|
||||
usage
|
||||
exit $?
|
||||
fi
|
||||
|
||||
exec "${MAIN}" "$@"
|
||||
@@ -1,120 +0,0 @@
|
||||
#! /bin/bash
|
||||
|
||||
set -x
|
||||
set -e
|
||||
|
||||
# use RAM disk if possible
|
||||
if [ "$CI" == "" ] && [ -d /dev/shm ]; then
|
||||
TEMP_BASE=/dev/shm
|
||||
else
|
||||
TEMP_BASE=/tmp
|
||||
fi
|
||||
|
||||
BUILD_DIR=$(mktemp -d -p "$TEMP_BASE" OCRmyPDF-AppImage-build-XXXXXX)
|
||||
|
||||
cleanup () {
|
||||
if [ -d "$BUILD_DIR" ]; then
|
||||
rm -rf "$BUILD_DIR"
|
||||
fi
|
||||
}
|
||||
|
||||
trap cleanup EXIT
|
||||
|
||||
# store repo root as variable
|
||||
REPO_ROOT=$(readlink -f "$(dirname "$(dirname "$0")")")
|
||||
OLD_CWD=$(readlink -f .)
|
||||
|
||||
pushd "$BUILD_DIR"
|
||||
|
||||
mkdir -p AppDir
|
||||
mkdir -p PackageDir
|
||||
mkdir -p jbig2
|
||||
|
||||
# download linuxdeploy AppImage and linuxdeploy-plugin-python AppImage
|
||||
wget https://github.com/TheAssassin/linuxdeploy/releases/download/continuous/linuxdeploy-x86_64.AppImage
|
||||
# wget https://github.com/niess/linuxdeploy-plugin-python/releases/download/continuous/linuxdeploy-plugin-python-x86_64.AppImage
|
||||
|
||||
# use adapted linuxdeploy-plugin-python instead of the original one (otherwise OCRmyPDF breaks)
|
||||
wget https://github.com/FPille/linuxdeploy-plugin-python/releases/download/continuous/linuxdeploy-plugin-python-x86_64.AppImage
|
||||
|
||||
chmod +x linuxdeploy*.AppImage
|
||||
|
||||
|
||||
ARCH=$(uname -i)
|
||||
export ARCH
|
||||
|
||||
|
||||
# .desktop file
|
||||
cat > ocrmypdf.desktop <<\EOF
|
||||
[Desktop Entry]
|
||||
Name=ocrmypdf
|
||||
Type=Application
|
||||
Exec=ocrmypdf
|
||||
Icon=ocrmypdf
|
||||
Terminal=true
|
||||
Comment=OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched
|
||||
Categories=Graphics;Scanning;OCR;
|
||||
EOF
|
||||
|
||||
|
||||
# download logo and convert it to desktop icon
|
||||
# requires Imagemagick (convert)
|
||||
wget https://raw.githubusercontent.com/jbarlow83/OCRmyPDF/master/docs/images/logo-social.png
|
||||
convert logo-social.png -resize 512x512\> -size 512x512 xc:white +swap -gravity center -composite ocrmypdf.png
|
||||
|
||||
|
||||
# download and intsall packages required by OCRmyPDF
|
||||
pushd PackageDir
|
||||
packages=(tesseract-ocr tesseract-ocr-all libavformat56 ghostscript qpdf pngquant)
|
||||
|
||||
for i in "${packages[@]}"
|
||||
do
|
||||
apt-get -d -o dir::cache="$PWD" -o Debug::NoLocking=1 --reinstall install "$i" -y
|
||||
done
|
||||
|
||||
wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O unpaper_6.1-1.deb
|
||||
|
||||
find . -type f -name \*.deb -exec dpkg-deb -X {} "$BUILD_DIR"/AppDir \;
|
||||
popd
|
||||
|
||||
|
||||
# compile and install jbig2
|
||||
# requires libleptonica-dev, zlib1g-dev
|
||||
wget -q https://github.com/agl/jbig2enc/archive/0.29.tar.gz -O - | \
|
||||
tar xz -C jbig2 --strip-components=1
|
||||
pushd jbig2
|
||||
./autogen.sh
|
||||
./configure --prefix="$BUILD_DIR"/AppDir/usr
|
||||
make && make install
|
||||
popd
|
||||
|
||||
pushd "$BUILD_DIR"/AppDir
|
||||
# add some tools to AppDir
|
||||
#cp -f /usr/bin/column ./usr/bin/
|
||||
#cp -f /bin/less ./usr/bin/
|
||||
|
||||
# remove unnecessary data from AppDir
|
||||
[ -d bin ] && rm -rf ./bin
|
||||
[ -d etc ] && rm -rf ./etc
|
||||
[ -d var ] && rm -rf ./var
|
||||
popd
|
||||
|
||||
|
||||
# export LD_LIBRARY_PATH so that dependencies of shared libraries can be deployed by linuxdeploy-x86_64.AppImage
|
||||
export LD_LIBRARY_PATH="$BUILD_DIR/AppDir/usr/lib:$BUILD_DIR/AppDir/usr/lib/x86_64-linux-gnu:$LD_LIBRARY_PATH"
|
||||
|
||||
#OCRMYPDF_VERSION=8.3.2 # exported in .travis.yml file
|
||||
export PIP_REQUIREMENTS="ocrmypdf==$OCRMYPDF_VERSION"
|
||||
export VERSION="$OCRMYPDF_VERSION"
|
||||
export OUTPUT=OCRmyPDF-"$VERSION"-"$ARCH".AppImage
|
||||
export PYTHON_SOURCE=https://www.python.org/ftp/python/3.6.8/Python-3.6.8.tgz
|
||||
|
||||
./linuxdeploy-x86_64.AppImage --appdir AppDir --plugin python \
|
||||
-d ocrmypdf.desktop -i ocrmypdf.png \
|
||||
--custom-apprun "$REPO_ROOT"/appimage/AppRun.sh --output appimage
|
||||
|
||||
|
||||
# move AppImage back to old CWD
|
||||
mv "$OUTPUT" "$OLD_CWD"/
|
||||
|
||||
popd
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/bin/bash
|
||||
|
||||
. /appenv/bin/activate
|
||||
cd /home/docker
|
||||
exec ocrmypdf "$@"
|
||||
@@ -0,0 +1,18 @@
|
||||
from enum import IntEnum
|
||||
import os
|
||||
|
||||
|
||||
class ExitCode(IntEnum):
|
||||
ok = 0
|
||||
bad_args = 1
|
||||
input_file = 2
|
||||
missing_dependency = 3
|
||||
invalid_output_pdfa = 4
|
||||
file_access_error = 5
|
||||
already_done_ocr = 6
|
||||
other_error = 15
|
||||
|
||||
|
||||
def get_program(name):
|
||||
envvar = 'OCRMYPDF_' + name.upper()
|
||||
return os.environ.get(envvar, name)
|
||||
@@ -0,0 +1,56 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
from tempfile import NamedTemporaryFile
|
||||
from subprocess import Popen, PIPE, check_call
|
||||
from shutil import copy
|
||||
from . import get_program
|
||||
|
||||
|
||||
def rasterize_pdf(input_file, output_file, xres, yres, raster_device, log):
|
||||
with NamedTemporaryFile(delete=True) as tmp:
|
||||
args_gs = [
|
||||
get_program('gs'),
|
||||
'-dQUIET',
|
||||
'-dBATCH',
|
||||
'-dNOPAUSE',
|
||||
'-sDEVICE=%s' % raster_device,
|
||||
'-o', tmp.name,
|
||||
'-r{0}x{1}'.format(str(xres), str(yres)),
|
||||
input_file
|
||||
]
|
||||
|
||||
p = Popen(args_gs, close_fds=True, stdout=PIPE, stderr=PIPE,
|
||||
universal_newlines=True)
|
||||
stdout, stderr = p.communicate()
|
||||
if stdout:
|
||||
log.debug(stdout)
|
||||
if stderr:
|
||||
log.error(stderr)
|
||||
|
||||
if p.returncode == 0:
|
||||
copy(tmp.name, output_file)
|
||||
else:
|
||||
log.error('Ghostscript rendering failed')
|
||||
|
||||
|
||||
def generate_pdfa(pdf_pages, output_file, threads=1):
|
||||
with NamedTemporaryFile(delete=True) as gs_pdf:
|
||||
args_gs = [
|
||||
get_program("gs"),
|
||||
"-dQUIET",
|
||||
"-dBATCH",
|
||||
"-dNOPAUSE",
|
||||
'-dNumRenderingThreads=' + str(threads),
|
||||
"-sDEVICE=pdfwrite",
|
||||
"-sColorConversionStrategy=/RGB",
|
||||
"-sProcessColorModel=DeviceRGB",
|
||||
"-dJPEGQ=95",
|
||||
"-dPDFA=2",
|
||||
"-sPDFACompatibilityPolicy=2",
|
||||
"-sOutputICCProfile=srgb.icc",
|
||||
"-sOutputFile=" + gs_pdf.name,
|
||||
]
|
||||
args_gs.extend(pdf_pages)
|
||||
check_call(args_gs)
|
||||
copy(gs_pdf.name, output_file)
|
||||
Executable
+230
@@ -0,0 +1,230 @@
|
||||
#!/usr/bin/env python3
|
||||
##############################################################################
|
||||
# Copyright (c) 2013-14: fritz-hh from Github
|
||||
# (https://github.com/fritz-hh)
|
||||
#
|
||||
# Copyright (c) 2010: Jonathan Brinley from Github
|
||||
# (https://github.com/jbrinley/HocrConverter)
|
||||
# Initial version by Jonathan Brinley, jonathanbrinley@gmail.com
|
||||
##############################################################################
|
||||
from reportlab.pdfgen.canvas import Canvas
|
||||
from reportlab.lib.units import inch
|
||||
from xml.etree import ElementTree
|
||||
from PIL import Image
|
||||
from collections import namedtuple
|
||||
import re
|
||||
import argparse
|
||||
|
||||
|
||||
Rect = namedtuple('Rect', ['x1', 'y1', 'x2', 'y2'])
|
||||
|
||||
|
||||
class HocrTransformError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class HocrTransform():
|
||||
|
||||
"""
|
||||
A class for converting documents from the hOCR format.
|
||||
For details of the hOCR format, see:
|
||||
http://docs.google.com/View?docid=dfxcv4vc_67g844kf
|
||||
"""
|
||||
|
||||
def __init__(self, hocrFileName, dpi):
|
||||
self.dpi = dpi
|
||||
self.boxPattern = re.compile(r'bbox((\s+\d+){4})')
|
||||
|
||||
self.hocr = ElementTree.parse(hocrFileName)
|
||||
|
||||
# if the hOCR file has a namespace, ElementTree requires its use to
|
||||
# find elements
|
||||
matches = re.match(r'({.*})html', self.hocr.getroot().tag)
|
||||
self.xmlns = ''
|
||||
if matches:
|
||||
self.xmlns = matches.group(1)
|
||||
|
||||
# get dimension in pt (not pixel!!!!) of the OCRed image
|
||||
self.width, self.height = None, None
|
||||
for div in self.hocr.findall(
|
||||
".//%sdiv[@class='ocr_page']" % (self.xmlns)):
|
||||
coords = self.element_coordinates(div)
|
||||
pt_coords = self.pt_from_pixel(coords)
|
||||
self.width = pt_coords.x2 - pt_coords.x1
|
||||
self.height = pt_coords.y2 - pt_coords.y1
|
||||
# there shouldn't be more than one, and if there is, we don't want
|
||||
# it
|
||||
break
|
||||
if self.width is None or self.height is None:
|
||||
raise HocrTransformError("hocr file is missing page dimensions")
|
||||
|
||||
def __str__(self):
|
||||
"""
|
||||
Return the textual content of the HTML body
|
||||
"""
|
||||
if self.hocr is None:
|
||||
return ''
|
||||
body = self.hocr.find(".//%sbody" % (self.xmlns))
|
||||
if body:
|
||||
return self._get_element_text(body)
|
||||
else:
|
||||
return ''
|
||||
|
||||
def _get_element_text(self, element):
|
||||
"""
|
||||
Return the textual content of the element and its children
|
||||
"""
|
||||
text = ''
|
||||
if element.text is not None:
|
||||
text += element.text
|
||||
for child in element.getchildren():
|
||||
text += self._get_element_text(child)
|
||||
if element.tail is not None:
|
||||
text += element.tail
|
||||
return text
|
||||
|
||||
def element_coordinates(self, element):
|
||||
"""
|
||||
Returns a tuple containing the coordinates of the bounding box around
|
||||
an element
|
||||
"""
|
||||
out = (0, 0, 0, 0)
|
||||
if 'title' in element.attrib:
|
||||
matches = self.boxPattern.search(element.attrib['title'])
|
||||
if matches:
|
||||
coords = matches.group(1).split()
|
||||
out = Rect._make(int(coords[n]) for n in range(4))
|
||||
return out
|
||||
|
||||
def pt_from_pixel(self, pxl):
|
||||
"""
|
||||
Returns the quantity in PDF units (pt) given quantity in pixels
|
||||
"""
|
||||
return Rect._make(
|
||||
(c / self.dpi * inch) for c in pxl)
|
||||
|
||||
def replace_unsupported_chars(self, s):
|
||||
"""
|
||||
Given an input string, returns the corresponding string that:
|
||||
- is available in the helvetica facetype
|
||||
- does not contain any ligature (to allow easy search in the PDF file)
|
||||
"""
|
||||
# The 'u' before the character to replace indicates that it is a
|
||||
# unicode character
|
||||
s = s.replace(u"fl", "fl")
|
||||
s = s.replace(u"fi", "fi")
|
||||
return s
|
||||
|
||||
def to_pdf(self, outFileName, imageFileName=None, showBoundingboxes=False,
|
||||
fontname="Helvetica", invisibleText=False):
|
||||
"""
|
||||
Creates a PDF file with an image superimposed on top of the text.
|
||||
Text is positioned according to the bounding box of the lines in
|
||||
the hOCR file.
|
||||
The image need not be identical to the image used to create the hOCR
|
||||
file.
|
||||
It can have a lower resolution, different color mode, etc.
|
||||
"""
|
||||
# create the PDF file
|
||||
# page size in points (1/72 in.)
|
||||
pdf = Canvas(
|
||||
outFileName, pagesize=(self.width, self.height), pageCompression=1)
|
||||
|
||||
# draw bounding box for each paragraph
|
||||
# light blue for bounding box of paragraph
|
||||
pdf.setStrokeColorRGB(0, 1, 1)
|
||||
# light blue for bounding box of paragraph
|
||||
pdf.setFillColorRGB(0, 1, 1)
|
||||
pdf.setLineWidth(0) # no line for bounding box
|
||||
for elem in self.hocr.findall(
|
||||
".//%sp[@class='%s']" % (self.xmlns, "ocr_par")):
|
||||
|
||||
elemtxt = self._get_element_text(elem).rstrip()
|
||||
if len(elemtxt) == 0:
|
||||
continue
|
||||
|
||||
pxl_coords = self.element_coordinates(elem)
|
||||
pt = self.pt_from_pixel(pxl_coords)
|
||||
|
||||
# draw the bbox border
|
||||
if showBoundingboxes:
|
||||
pdf.rect(
|
||||
pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1,
|
||||
fill=1)
|
||||
|
||||
# check if element with class 'ocrx_word' are available
|
||||
# otherwise use 'ocr_line' as fallback
|
||||
elemclass = "ocr_line"
|
||||
if self.hocr.find(
|
||||
".//%sspan[@class='ocrx_word']" % (self.xmlns)) is not None:
|
||||
elemclass = "ocrx_word"
|
||||
|
||||
# itterate all text elements
|
||||
# light green for bounding box of word/line
|
||||
pdf.setStrokeColorRGB(1, 0, 0)
|
||||
pdf.setLineWidth(0.5) # bounding box line width
|
||||
pdf.setDash(6, 3) # bounding box is dashed
|
||||
pdf.setFillColorRGB(0, 0, 0) # text in black
|
||||
for elem in self.hocr.findall(
|
||||
".//%sspan[@class='%s']" % (self.xmlns, elemclass)):
|
||||
|
||||
elemtxt = self._get_element_text(elem).rstrip()
|
||||
|
||||
elemtxt = self.replace_unsupported_chars(elemtxt)
|
||||
|
||||
if len(elemtxt) == 0:
|
||||
continue
|
||||
|
||||
pxl_coords = self.element_coordinates(elem)
|
||||
pt = self.pt_from_pixel(pxl_coords)
|
||||
|
||||
# draw the bbox border
|
||||
if showBoundingboxes:
|
||||
pdf.rect(
|
||||
pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1,
|
||||
fill=0)
|
||||
|
||||
text = pdf.beginText()
|
||||
fontsize = pt.y2 - pt.y1
|
||||
text.setFont(fontname, fontsize)
|
||||
if invisibleText:
|
||||
text.setTextRenderMode(3) # Invisible (indicates OCR text)
|
||||
|
||||
# set cursor to bottom left corner of bbox (adjust for dpi)
|
||||
text.setTextOrigin(pt.x1, self.height - pt.y2)
|
||||
|
||||
# scale the width of the text to fill the width of the bbox
|
||||
text.setHorizScale(
|
||||
100 * (pt.x2 - pt.x1) / pdf.stringWidth(
|
||||
elemtxt, fontname, fontsize))
|
||||
|
||||
# write the text to the page
|
||||
text.textLine(elemtxt)
|
||||
pdf.drawText(text)
|
||||
|
||||
# put the image on the page, scaled to fill the page
|
||||
if imageFileName is not None:
|
||||
pdf.drawImage(imageFileName, 0, 0,
|
||||
width=self.width, height=self.height)
|
||||
|
||||
# finish up the page and save it
|
||||
pdf.showPage()
|
||||
pdf.save()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser(description='Convert hocr file to PDF')
|
||||
parser.add_argument('-b', '--boundingboxes', action="store_true",
|
||||
default=False, help='Show bounding boxes borders')
|
||||
parser.add_argument('-r', '--resolution', type=int,
|
||||
default=300,
|
||||
help='Resolution of the image that was OCRed')
|
||||
parser.add_argument('-i', '--image', default=None,
|
||||
help='Path to the image to be placed above the text')
|
||||
parser.add_argument('hocrfile', help='Path to the hocr file to be parsed')
|
||||
parser.add_argument(
|
||||
'outputfile', help='Path to the PDF file to be generated')
|
||||
args = parser.parse_args()
|
||||
|
||||
hocr = HocrTransform(args.hocrfile, args.resolution)
|
||||
hocr.to_pdf(args.outputfile, args.image, args.boundingboxes)
|
||||
@@ -0,0 +1,331 @@
|
||||
#!/usr/bin/env python2
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# © 2013-15: jbarlow83 from Github (https://github.com/jbarlow83)
|
||||
#
|
||||
#
|
||||
# Use Leptonica to detect find and remove page skew. Leptonica uses the method
|
||||
# of differential square sums, which its author claim is faster and more robust
|
||||
# than the Hough transform used by ImageMagick.
|
||||
|
||||
from __future__ import print_function, absolute_import, division
|
||||
import argparse
|
||||
import ctypes as C
|
||||
import sys
|
||||
import os
|
||||
import logging
|
||||
from tempfile import TemporaryFile
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def stderr(*objs):
|
||||
"""Python 2/3 compatible print to stderr.
|
||||
"""
|
||||
print("leptonica.py:", *objs, file=sys.stderr)
|
||||
|
||||
|
||||
from ctypes.util import find_library
|
||||
lept_lib = find_library('lept')
|
||||
if not lept_lib:
|
||||
stderr("Could not find the Leptonica library")
|
||||
sys.exit(3)
|
||||
try:
|
||||
lept = C.cdll.LoadLibrary(lept_lib)
|
||||
except Exception:
|
||||
stderr("Could not load the Leptonica library from %s", lept_lib)
|
||||
sys.exit(3)
|
||||
|
||||
|
||||
class _PIXCOLORMAP(C.Structure):
|
||||
"""struct PixColormap from Leptonica src/pix.h
|
||||
"""
|
||||
|
||||
_fields_ = [
|
||||
("array", C.c_void_p),
|
||||
("depth", C.c_int32),
|
||||
("nalloc", C.c_int32),
|
||||
("n", C.c_int32)
|
||||
]
|
||||
|
||||
|
||||
class _PIX(C.Structure):
|
||||
"""struct Pix from Leptonica src/pix.h
|
||||
"""
|
||||
|
||||
_fields_ = [
|
||||
("w", C.c_uint32),
|
||||
("h", C.c_uint32),
|
||||
("d", C.c_uint32),
|
||||
("wpl", C.c_uint32),
|
||||
("refcount", C.c_uint32),
|
||||
("xres", C.c_int32),
|
||||
("yres", C.c_int32),
|
||||
("informat", C.c_int32),
|
||||
("text", C.POINTER(C.c_char)),
|
||||
("colormap", C.POINTER(_PIXCOLORMAP)),
|
||||
("data", C.POINTER(C.c_uint32))
|
||||
]
|
||||
|
||||
|
||||
PIX = C.POINTER(_PIX)
|
||||
|
||||
lept.pixRead.argtypes = [C.c_char_p]
|
||||
lept.pixRead.restype = PIX
|
||||
lept.pixScale.argtypes = [PIX, C.c_float, C.c_float]
|
||||
lept.pixScale.restype = PIX
|
||||
lept.pixDeskew.argtypes = [PIX, C.c_int32]
|
||||
lept.pixDeskew.restype = PIX
|
||||
lept.pixFindSkew.argtypes = [PIX, C.POINTER(C.c_float), C.POINTER(C.c_float)]
|
||||
lept.pixFindSkew.restype = C.c_int32
|
||||
lept.pixWriteImpliedFormat.argtypes = [C.c_char_p, PIX, C.c_int32, C.c_int32]
|
||||
lept.pixWriteImpliedFormat.restype = C.c_int32
|
||||
lept.pixDestroy.argtypes = [C.POINTER(PIX)]
|
||||
lept.pixDestroy.restype = None
|
||||
lept.getLeptonicaVersion.argtypes = []
|
||||
lept.getLeptonicaVersion.restype = C.c_char_p
|
||||
|
||||
|
||||
class LeptonicaErrorTrap(object):
|
||||
"""Context manager to trap errors reported by Leptonica.
|
||||
|
||||
Leptonica's error return codes are unreliable to the point of being
|
||||
almost useless. It does, however, write errors to stderr provided that is
|
||||
not disabled at its compile time. Fortunately this is done using error
|
||||
macros so it is very self-consistent.
|
||||
|
||||
This context manager redirects stderr to a temporary file which is then
|
||||
read and parsed for error messages. As a side benefit, debug messages
|
||||
from Leptonica are also suppressed.
|
||||
|
||||
"""
|
||||
def __enter__(self):
|
||||
self.tmpfile = TemporaryFile()
|
||||
|
||||
# Save the old stderr, and redirect stderr to temporary file
|
||||
self.old_stderr_fileno = os.dup(sys.stderr.fileno())
|
||||
os.dup2(self.tmpfile.fileno(), sys.stderr.fileno())
|
||||
return
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
# Restore old stderr
|
||||
os.dup2(self.old_stderr_fileno, sys.stderr.fileno())
|
||||
|
||||
# Get data from tmpfile (in with block to ensure it is closed)
|
||||
with self.tmpfile as tmpfile:
|
||||
tmpfile.seek(0) # Cursor will be at end, so move back to beginning
|
||||
leptonica_output = tmpfile.read().decode(errors='replace')
|
||||
|
||||
# If there are Python errors, let them bubble up
|
||||
if exc_type:
|
||||
logger.warning(leptonica_output)
|
||||
return False
|
||||
|
||||
# If there are Leptonica errors, wrap them in Python excpetions
|
||||
if 'Error' in leptonica_output:
|
||||
if 'image file not found' in leptonica_output:
|
||||
raise FileNotFoundError()
|
||||
if 'pixWrite: stream not opened' in leptonica_output:
|
||||
raise LeptonicaIOError()
|
||||
raise LeptonicaError(leptonica_output)
|
||||
|
||||
return False
|
||||
|
||||
|
||||
class LeptonicaError(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class LeptonicaIOError(LeptonicaError):
|
||||
pass
|
||||
|
||||
|
||||
def pixRead(filename):
|
||||
"""Load an image file into a PIX object.
|
||||
|
||||
Leptonica can load TIFF, PNM (PBM, PGM, PPM), PNG, and JPEG. If loading
|
||||
fails then the object will wrap a C null pointer.
|
||||
|
||||
"""
|
||||
with LeptonicaErrorTrap():
|
||||
return lept.pixRead(filename.encode(sys.getfilesystemencoding()))
|
||||
|
||||
|
||||
def pixScale(pix, scalex, scaley):
|
||||
"""Returns the pix object rescaled according to the proportions given."""
|
||||
with LeptonicaErrorTrap():
|
||||
return lept.pixScale(pix, scalex, scaley)
|
||||
|
||||
|
||||
def pixDeskew(pix, reduction_factor=0):
|
||||
"""Returns the deskewed pix object.
|
||||
|
||||
A clone of the original is returned when the algorithm cannot find a skew
|
||||
angle with sufficient confidence.
|
||||
|
||||
reduction_factor -- amount to downsample (0 for default) when searching
|
||||
for skew angle
|
||||
|
||||
"""
|
||||
with LeptonicaErrorTrap():
|
||||
return lept.pixDeskew(pix, reduction_factor)
|
||||
|
||||
|
||||
def pixFindSkew(pix):
|
||||
"""Returns a tuple (deskew angle in degrees, confidence value).
|
||||
|
||||
Returns (None, None) if no angle is available.
|
||||
|
||||
"""
|
||||
with LeptonicaErrorTrap():
|
||||
angle = C.c_float(0.0)
|
||||
confidence = C.c_float(0.0)
|
||||
result = lept.pixFindSkew(pix, C.byref(angle), C.byref(confidence))
|
||||
if result == 0:
|
||||
return (angle.value, confidence.value)
|
||||
else:
|
||||
return (None, None)
|
||||
|
||||
|
||||
def pixWriteImpliedFormat(filename, pix, jpeg_quality=0, jpeg_progressive=0):
|
||||
"""Write pix to the filename, with the extension indicating format.
|
||||
|
||||
jpeg_quality -- quality (iff JPEG; 1 - 100, 0 for default)
|
||||
jpeg_progressive -- (iff JPEG; 0 for baseline seq., 1 for progressive)
|
||||
|
||||
"""
|
||||
fileroot, extension = os.path.splitext(filename)
|
||||
fix_pnm = False
|
||||
if extension.lower() in ('.pbm', '.pgm', '.ppm'):
|
||||
# Leptonica does not process handle these extensions correctly, but
|
||||
# does handle .pnm correctly. Add another .pnm suffix.
|
||||
filename += '.pnm'
|
||||
fix_pnm = True
|
||||
|
||||
with LeptonicaErrorTrap():
|
||||
lept.pixWriteImpliedFormat(
|
||||
filename.encode(sys.getfilesystemencoding()),
|
||||
pix, jpeg_quality, jpeg_progressive)
|
||||
|
||||
if fix_pnm:
|
||||
from shutil import move
|
||||
move(filename, filename[:-4]) # Remove .pnm suffix
|
||||
|
||||
|
||||
def pixDestroy(pix):
|
||||
"""Destroy the pix object.
|
||||
|
||||
Function signature is pixDestroy(struct Pix **), hence C.byref() to pass
|
||||
the address of the pointer.
|
||||
|
||||
"""
|
||||
with LeptonicaErrorTrap():
|
||||
lept.pixDestroy(C.byref(pix))
|
||||
|
||||
|
||||
def getLeptonicaVersion():
|
||||
"""Get Leptonica version string.
|
||||
|
||||
Caveat: Leptonica expects the caller to free this memory. We don't,
|
||||
since that would involve binding to libc to access libc.free(),
|
||||
a pointless effort to reclaim 100 bytes of memory.
|
||||
|
||||
"""
|
||||
return lept.getLeptonicaVersion().decode()
|
||||
|
||||
|
||||
def deskew(infile, outfile, dpi):
|
||||
try:
|
||||
pix_source = pixRead(infile)
|
||||
except LeptonicaIOError:
|
||||
raise LeptonicaIOError("Failed to open file: %s" % infile)
|
||||
|
||||
if dpi < 150:
|
||||
reduction_factor = 1 # Don't downsample too much if DPI is already low
|
||||
else:
|
||||
reduction_factor = 0 # Use default
|
||||
pix_deskewed = pixDeskew(pix_source, reduction_factor)
|
||||
|
||||
try:
|
||||
pixWriteImpliedFormat(outfile, pix_deskewed)
|
||||
except LeptonicaIOError:
|
||||
raise LeptonicaIOError("Failed to open destination file: %s" % outfile)
|
||||
pixDestroy(pix_source)
|
||||
pixDestroy(pix_deskewed)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Python wrapper to access Leptonica")
|
||||
|
||||
subparsers = parser.add_subparsers(title='commands',
|
||||
description='supported operations')
|
||||
|
||||
parser_deskew = subparsers.add_parser('deskew')
|
||||
parser_deskew.add_argument('-r', '--dpi', dest='dpi', action='store',
|
||||
type=int, default=300, help='input resolution')
|
||||
parser_deskew.add_argument('infile', help='image to deskew')
|
||||
parser_deskew.add_argument('outfile', help='deskewed output image')
|
||||
parser_deskew.set_defaults(func=deskew)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
if getLeptonicaVersion() != u'leptonica-1.69':
|
||||
print("Unexpected leptonica version: %s" % getLeptonicaVersion())
|
||||
|
||||
args.func(args)
|
||||
|
||||
|
||||
def _test_output(mode, extension, im_format):
|
||||
from PIL import Image
|
||||
from tempfile import NamedTemporaryFile
|
||||
|
||||
with NamedTemporaryFile(prefix='test-lept-pnm', suffix=extension, delete=True) as tmpfile:
|
||||
im = Image.new(mode=mode, size=(100, 100))
|
||||
im.save(tmpfile)
|
||||
|
||||
pix = pixRead(tmpfile.name)
|
||||
pixWriteImpliedFormat(tmpfile.name, pix)
|
||||
pixDestroy(pix)
|
||||
|
||||
im_roundtrip = Image.open(tmpfile.name)
|
||||
assert im_roundtrip.mode == im.mode, "leptonica mode differs"
|
||||
assert im_roundtrip.format == im_format, \
|
||||
"{0}: leptonica produced a {1}".format(
|
||||
extension,
|
||||
im_roundtrip.format)
|
||||
|
||||
|
||||
def test_pnm_output():
|
||||
params = [['1', '.pbm', 'PPM'], ['L', '.pgm', 'PPM'],
|
||||
['RGB', '.ppm', 'PPM']]
|
||||
for param in params:
|
||||
_test_output(*param)
|
||||
|
||||
|
||||
def test_skew_angle():
|
||||
from PIL import Image, ImageDraw
|
||||
from tempfile import NamedTemporaryFile
|
||||
|
||||
im = Image.new(mode='1', size=(1000, 1000), color=1)
|
||||
|
||||
draw = ImageDraw.Draw(im)
|
||||
for n in range(20):
|
||||
draw.line([(50, 25 + 50*n), (950, 25 + 50*n)], width=1)
|
||||
del draw
|
||||
|
||||
test_angles = [0.1 * ang for ang in range(1, 10)] + \
|
||||
[float(ang) for ang in range(1, 7)]
|
||||
test_angles += [-ang for ang in test_angles]
|
||||
test_angles = sorted(test_angles)
|
||||
|
||||
for rotate_angle in test_angles:
|
||||
rotated_im = im.rotate(rotate_angle)
|
||||
with NamedTemporaryFile(prefix='lept-skew', suffix='.png', delete=True) as tmpfile:
|
||||
rotated_im.save(tmpfile)
|
||||
pix = pixRead(tmpfile.name)
|
||||
angle, confidence = pixFindSkew(pix)
|
||||
pixDestroy(pix)
|
||||
print('{0} {1} {2}'.format(rotate_angle, angle, confidence), file=sys.stderr)
|
||||
|
||||
|
||||
Executable
+878
@@ -0,0 +1,878 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
from contextlib import suppress
|
||||
from tempfile import mkdtemp
|
||||
import sys
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import warnings
|
||||
import multiprocessing
|
||||
import atexit
|
||||
import textwrap
|
||||
import img2pdf
|
||||
|
||||
import PyPDF2 as pypdf
|
||||
from PIL import Image
|
||||
|
||||
from functools import partial
|
||||
|
||||
from ruffus import transform, suffix, merge, active_if, regex, jobs_limit, \
|
||||
formatter, follows, split, collate, check_if_uptodate
|
||||
import ruffus.ruffus_exceptions as ruffus_exceptions
|
||||
import ruffus.cmdline as cmdline
|
||||
|
||||
from .hocrtransform import HocrTransform
|
||||
from .pageinfo import pdf_get_all_pageinfo
|
||||
from .pdfa import generate_pdfa_def
|
||||
from . import ghostscript
|
||||
from . import tesseract
|
||||
from . import qpdf
|
||||
from . import ExitCode
|
||||
|
||||
import pkg_resources
|
||||
|
||||
VERSION = pkg_resources.get_distribution('ocrmypdf').version
|
||||
|
||||
warnings.simplefilter('ignore', pypdf.utils.PdfReadWarning)
|
||||
|
||||
|
||||
BASEDIR = os.path.dirname(os.path.realpath(__file__))
|
||||
|
||||
|
||||
# -------------
|
||||
# External dependencies
|
||||
|
||||
MINIMUM_TESS_VERSION = '3.02.02'
|
||||
|
||||
|
||||
def complain(message):
|
||||
print(*textwrap.wrap(message), file=sys.stderr)
|
||||
|
||||
|
||||
if tesseract.version() < MINIMUM_TESS_VERSION:
|
||||
complain(
|
||||
"Please install tesseract {0} or newer "
|
||||
"(currently installed version is {1})".format(
|
||||
MINIMUM_TESS_VERSION, tesseract.version()))
|
||||
sys.exit(ExitCode.missing_dependency)
|
||||
|
||||
|
||||
try:
|
||||
import PIL.features
|
||||
check_codec = PIL.features.check_codec
|
||||
except (ImportError, AttributeError):
|
||||
def check_codec(codec_name):
|
||||
if codec_name == 'jpg':
|
||||
return 'jpeg_encoder' in dir(Image.core)
|
||||
elif codec_name == 'zlib':
|
||||
return 'zip_encoder' in dir(Image.core)
|
||||
raise NotImplementedError(codec_name)
|
||||
|
||||
|
||||
def check_pil_encoder(codec_name, friendly_name):
|
||||
try:
|
||||
if check_codec(codec_name):
|
||||
return
|
||||
except Exception:
|
||||
pass
|
||||
complain(
|
||||
"ERROR: Your version of the Python imaging library (Pillow) was "
|
||||
"compiled without support for " + friendly_name + " encoding/decoding."
|
||||
"\n"
|
||||
"You will need to uninstall Pillow and reinstall it with PNG and JPEG "
|
||||
"support (libjpeg and zlib)."
|
||||
"\n"
|
||||
"See installation instructions for your platform here:\n"
|
||||
" https://pillow.readthedocs.org/installation.html"
|
||||
)
|
||||
sys.exit(ExitCode.missing_dependency)
|
||||
|
||||
|
||||
check_pil_encoder('jpg', 'JPEG')
|
||||
check_pil_encoder('zlib', 'PNG')
|
||||
|
||||
|
||||
# -------------
|
||||
# Parser
|
||||
|
||||
parser = cmdline.get_argparse(
|
||||
prog="ocrmypdf",
|
||||
description="Generate searchable PDF file from an image-only PDF file.",
|
||||
version=VERSION,
|
||||
fromfile_prefix_chars='@',
|
||||
ignored_args=[
|
||||
'touch_files_only', 'recreate_database', 'checksum_file_name',
|
||||
'key_legend_in_graph', 'draw_graph_horizontally', 'flowchart_format',
|
||||
'forced_tasks', 'target_tasks', 'use_threads', 'jobs'])
|
||||
|
||||
parser.add_argument(
|
||||
'input_file',
|
||||
help="PDF file containing the images to be OCRed")
|
||||
parser.add_argument(
|
||||
'output_file',
|
||||
help="output searchable PDF file")
|
||||
parser.add_argument(
|
||||
'-l', '--language', action='append',
|
||||
help="languages of the file to be OCRed")
|
||||
parser.add_argument(
|
||||
'-j', '--jobs', metavar='N', type=int,
|
||||
help="Use up to N CPU cores simultaneously (default: use all)")
|
||||
|
||||
metadata = parser.add_argument_group(
|
||||
"Metadata options",
|
||||
"Set output PDF/A metadata (default: use input document's title)")
|
||||
metadata.add_argument(
|
||||
'--title', type=str,
|
||||
help="set document title (place multiple words in quotes)")
|
||||
metadata.add_argument(
|
||||
'--author', type=str,
|
||||
help="set document author")
|
||||
metadata.add_argument(
|
||||
'--subject', type=str,
|
||||
help="set document")
|
||||
metadata.add_argument(
|
||||
'--keywords', type=str,
|
||||
help="set document keywords")
|
||||
|
||||
|
||||
preprocessing = parser.add_argument_group(
|
||||
"Preprocessing options",
|
||||
"Improve OCR quality and final image")
|
||||
preprocessing.add_argument(
|
||||
'-d', '--deskew', action='store_true',
|
||||
help="deskew each page before performing OCR")
|
||||
preprocessing.add_argument(
|
||||
'-c', '--clean', action='store_true',
|
||||
help="clean pages from scanning artifacts before performing OCR")
|
||||
preprocessing.add_argument(
|
||||
'-i', '--clean-final', action='store_true',
|
||||
help="incorporate the cleaned image in the final PDF file")
|
||||
preprocessing.add_argument(
|
||||
'--oversample', metavar='DPI', type=int, default=0,
|
||||
help="oversample images to at least the specified DPI, to improve OCR "
|
||||
"results slightly")
|
||||
|
||||
parser.add_argument(
|
||||
'-f', '--force-ocr', action='store_true',
|
||||
help="rasterize any fonts or vector images on each page and apply OCR")
|
||||
parser.add_argument(
|
||||
'-s', '--skip-text', action='store_true',
|
||||
help="skip OCR on any pages that already contain text, but include the"
|
||||
" page in final output")
|
||||
parser.add_argument(
|
||||
'--skip-big', type=float, metavar='MPixels',
|
||||
help="skip OCR on pages larger than the specified amount of megapixels, "
|
||||
"but include skipped pages in final output")
|
||||
# parser.add_argument(
|
||||
# '--exact-image', action='store_true',
|
||||
# help="Use original page from PDF without re-rendering")
|
||||
|
||||
advanced = parser.add_argument_group(
|
||||
"Advanced",
|
||||
"Advanced options for power users")
|
||||
advanced.add_argument(
|
||||
'--tesseract-config', action='append', metavar='CFG', default=[],
|
||||
help="additional Tesseract configuration files")
|
||||
advanced.add_argument(
|
||||
'--tesseract-pagesegmode', action='store', type=int, metavar='PSM',
|
||||
help="set Tesseract page segmentation mode (see tesseract --help)")
|
||||
advanced.add_argument(
|
||||
'--pdf-renderer', choices=['auto', 'tesseract', 'hocr'], default='auto',
|
||||
help='choose OCR PDF renderer')
|
||||
advanced.add_argument(
|
||||
'--tesseract-timeout', default=180.0, type=float, metavar='SECONDS',
|
||||
help='give up on OCR after the timeout, but copy the preprocessed page '
|
||||
'into the final output')
|
||||
|
||||
debugging = parser.add_argument_group(
|
||||
"Debugging",
|
||||
"Arguments to help with troubleshooting and debugging")
|
||||
debugging.add_argument(
|
||||
'-k', '--keep-temporary-files', action='store_true',
|
||||
help="keep temporary files (helpful for debugging)")
|
||||
debugging.add_argument(
|
||||
'-g', '--debug-rendering', action='store_true',
|
||||
help="render each page twice with debug information on second page")
|
||||
|
||||
options = parser.parse_args()
|
||||
|
||||
|
||||
# ----------
|
||||
# Languages
|
||||
|
||||
if not options.language:
|
||||
options.language = ['eng'] # Enforce English hegemony
|
||||
|
||||
# Support v2.x "eng+deu" language syntax
|
||||
if '+' in options.language[0]:
|
||||
options.language = options.language[0].split('+')
|
||||
|
||||
if not set(options.language).issubset(tesseract.languages()):
|
||||
complain(
|
||||
"The installed version of tesseract does not have language "
|
||||
"data for the following requested languages: ")
|
||||
for lang in (set(options.language) - tesseract.languages()):
|
||||
complain(lang)
|
||||
sys.exit(ExitCode.bad_args)
|
||||
|
||||
|
||||
# ----------
|
||||
# Arguments
|
||||
|
||||
if options.pdf_renderer == 'auto':
|
||||
options.pdf_renderer = 'hocr'
|
||||
|
||||
if any((options.deskew, options.clean, options.clean_final)):
|
||||
try:
|
||||
from . import unpaper
|
||||
except ImportError:
|
||||
complain(
|
||||
"Install the 'unpaper' program to use --deskew or --clean.")
|
||||
sys.exit(ExitCode.bad_args)
|
||||
else:
|
||||
unpaper = None
|
||||
|
||||
if options.debug_rendering and options.pdf_renderer == 'tesseract':
|
||||
complain(
|
||||
"Ignoring --debug-rendering because it is not supported with"
|
||||
"--pdf-renderer=tesseract.")
|
||||
|
||||
if options.force_ocr and options.skip_text:
|
||||
complain(
|
||||
"Error: --force-ocr and --skip-text are mutually incompatible.")
|
||||
sys.exit(ExitCode.bad_args)
|
||||
|
||||
if options.clean and not options.clean_final \
|
||||
and options.pdf_renderer == 'tesseract':
|
||||
complain(
|
||||
"Tesseract PDF renderer cannot render --clean pages without "
|
||||
"also performing --clean-final, so --clean-final is assumed.")
|
||||
|
||||
lossless_reconstruction = False
|
||||
if options.pdf_renderer == 'hocr':
|
||||
if not options.deskew and not options.clean_final and not options.force_ocr:
|
||||
lossless_reconstruction = True
|
||||
|
||||
# ----------
|
||||
# Logging
|
||||
|
||||
|
||||
_logger, _logger_mutex = cmdline.setup_logging(__name__, options.log_file,
|
||||
options.verbose)
|
||||
|
||||
|
||||
class WrappedLogger:
|
||||
|
||||
def __init__(self, my_logger, my_mutex):
|
||||
self.logger = my_logger
|
||||
self.mutex = my_mutex
|
||||
|
||||
def log(self, *args, **kwargs):
|
||||
with self.mutex:
|
||||
self.logger.log(*args, **kwargs)
|
||||
|
||||
def debug(self, *args, **kwargs):
|
||||
with self.mutex:
|
||||
self.logger.debug(*args, **kwargs)
|
||||
|
||||
def info(self, *args, **kwargs):
|
||||
with self.mutex:
|
||||
self.logger.info(*args, **kwargs)
|
||||
|
||||
def warning(self, *args, **kwargs):
|
||||
with self.mutex:
|
||||
self.logger.warning(*args, **kwargs)
|
||||
|
||||
def error(self, *args, **kwargs):
|
||||
with self.mutex:
|
||||
self.logger.error(*args, **kwargs)
|
||||
|
||||
def critical(self, *args, **kwargs):
|
||||
with self.mutex:
|
||||
self.logger.critical(*args, **kwargs)
|
||||
|
||||
_log = WrappedLogger(_logger, _logger_mutex)
|
||||
|
||||
|
||||
def re_symlink(input_file, soft_link_name, log=_log):
|
||||
"""
|
||||
Helper function: relinks soft symbolic link if necessary
|
||||
"""
|
||||
# Guard against soft linking to oneself
|
||||
if input_file == soft_link_name:
|
||||
log.debug("Warning: No symbolic link made. You are using " +
|
||||
"the original data directory as the working directory.")
|
||||
return
|
||||
|
||||
# Soft link already exists: delete for relink?
|
||||
if os.path.lexists(soft_link_name):
|
||||
# do not delete or overwrite real (non-soft link) file
|
||||
if not os.path.islink(soft_link_name):
|
||||
raise Exception("%s exists and is not a link" % soft_link_name)
|
||||
try:
|
||||
os.unlink(soft_link_name)
|
||||
except:
|
||||
log.debug("Can't unlink %s" % (soft_link_name))
|
||||
|
||||
if not os.path.exists(input_file):
|
||||
raise Exception("trying to create a broken symlink to %s" % input_file)
|
||||
|
||||
log.debug("os.symlink(%s, %s)" % (input_file, soft_link_name))
|
||||
|
||||
# Create symbolic link using absolute path
|
||||
os.symlink(
|
||||
os.path.abspath(input_file),
|
||||
soft_link_name
|
||||
)
|
||||
|
||||
|
||||
# -------------
|
||||
# The Pipeline
|
||||
|
||||
manager = multiprocessing.Manager()
|
||||
_pdfinfo = manager.list()
|
||||
_pdfinfo_lock = manager.Lock()
|
||||
|
||||
work_folder = mkdtemp(prefix="com.github.ocrmypdf.")
|
||||
|
||||
|
||||
@atexit.register
|
||||
def cleanup_working_files(*args):
|
||||
if options.keep_temporary_files:
|
||||
print("Temporary working files saved at:")
|
||||
print(work_folder)
|
||||
else:
|
||||
with suppress(FileNotFoundError):
|
||||
shutil.rmtree(work_folder)
|
||||
|
||||
|
||||
@transform(
|
||||
input=options.input_file,
|
||||
filter=formatter('(?i)\.pdf'),
|
||||
output=work_folder + '{basename[0]}.repaired.pdf',
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def repair_pdf(
|
||||
input_file,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
|
||||
qpdf.repair(input_file, output_file, log)
|
||||
with pdfinfo_lock:
|
||||
pdfinfo.extend(pdf_get_all_pageinfo(output_file))
|
||||
log.info(pdfinfo)
|
||||
|
||||
|
||||
def get_pageinfo(input_file, pdfinfo, pdfinfo_lock):
|
||||
pageno = int(os.path.basename(input_file)[0:6]) - 1
|
||||
with pdfinfo_lock:
|
||||
pageinfo = pdfinfo[pageno].copy()
|
||||
return pageinfo
|
||||
|
||||
|
||||
def is_ocr_required(pageinfo, log):
|
||||
page = pageinfo['pageno'] + 1
|
||||
ocr_required = True
|
||||
if not pageinfo['images']:
|
||||
# If the page has no images, then it contains vector content or text
|
||||
# or both. It seems quite unlikely that one would find meaningful text
|
||||
# from rasterizing vector content. So skip the page.
|
||||
log.info(
|
||||
"Page {0} has no images - skipping OCR".format(page)
|
||||
)
|
||||
ocr_required = False
|
||||
elif pageinfo['has_text']:
|
||||
s = "Page {0} already has text! – {1}"
|
||||
|
||||
if not options.force_ocr and not options.skip_text:
|
||||
log.error(s.format(page,
|
||||
"aborting (use --force-ocr to force OCR)"))
|
||||
sys.exit(ExitCode.already_done_ocr)
|
||||
elif options.force_ocr:
|
||||
log.info(s.format(page,
|
||||
"rasterizing text and running OCR anyway"))
|
||||
ocr_required = True
|
||||
elif options.skip_text:
|
||||
log.info(s.format(page,
|
||||
"skipping all processing on this page"))
|
||||
ocr_required = False
|
||||
|
||||
if ocr_required and options.skip_big:
|
||||
pixel_count = pageinfo['width_pixels'] * pageinfo['height_pixels']
|
||||
if pixel_count > (options.skip_big * 1000000):
|
||||
ocr_required = False
|
||||
log.info(
|
||||
"Page {0} is very large; skipping due to -b".format(page))
|
||||
|
||||
return ocr_required
|
||||
|
||||
|
||||
@split(
|
||||
repair_pdf,
|
||||
os.path.join(work_folder, '*.page.pdf'),
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def split_pages(
|
||||
input_file,
|
||||
output_files,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
|
||||
for oo in output_files:
|
||||
with suppress(FileNotFoundError):
|
||||
os.unlink(oo)
|
||||
|
||||
npages = qpdf.get_npages(input_file)
|
||||
qpdf.split_pages(input_file, work_folder, npages)
|
||||
|
||||
from glob import glob
|
||||
for filename in glob(os.path.join(work_folder, '*.page.pdf')):
|
||||
pageinfo = get_pageinfo(filename, pdfinfo, pdfinfo_lock)
|
||||
|
||||
alt_suffix = '.ocr.page.pdf' if is_ocr_required(pageinfo, log) \
|
||||
else '.skip.page.pdf'
|
||||
re_symlink(
|
||||
filename,
|
||||
os.path.join(
|
||||
work_folder,
|
||||
os.path.basename(filename)[0:6] + alt_suffix))
|
||||
|
||||
|
||||
@transform(
|
||||
input=split_pages,
|
||||
filter=suffix('.ocr.page.pdf'),
|
||||
output='.page.png',
|
||||
output_dir=work_folder,
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def rasterize_with_ghostscript(
|
||||
input_file,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
pageinfo = get_pageinfo(input_file, pdfinfo, pdfinfo_lock)
|
||||
|
||||
device = 'png16m' # 24-bit
|
||||
if all(image['comp'] == 1 for image in pageinfo['images']):
|
||||
if all(image['bpc'] == 1 for image in pageinfo['images']):
|
||||
device = 'pngmono'
|
||||
elif all(image['bpc'] > 1 and image['color'] == 'index'
|
||||
for image in pageinfo['images']):
|
||||
device = 'png256'
|
||||
elif all(image['bpc'] > 1 and image['color'] == 'gray'
|
||||
for image in pageinfo['images']):
|
||||
device = 'pnggray'
|
||||
|
||||
log.debug("Rendering {0} with {1}".format(
|
||||
os.path.basename(input_file), device))
|
||||
xres = max(pageinfo['xres'], options.oversample or 0)
|
||||
yres = max(pageinfo['yres'], options.oversample or 0)
|
||||
|
||||
ghostscript.rasterize_pdf(input_file, output_file, xres, yres, device, log)
|
||||
|
||||
|
||||
@transform(
|
||||
input=rasterize_with_ghostscript,
|
||||
filter=suffix(".page.png"),
|
||||
output=".pp-deskew.png",
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def preprocess_deskew(
|
||||
input_file,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
|
||||
if not options.deskew:
|
||||
re_symlink(input_file, output_file, log)
|
||||
return
|
||||
|
||||
pageinfo = get_pageinfo(input_file, pdfinfo, pdfinfo_lock)
|
||||
dpi = int(pageinfo['xres'])
|
||||
|
||||
unpaper.deskew(input_file, output_file, dpi, log)
|
||||
|
||||
|
||||
@transform(
|
||||
input=preprocess_deskew,
|
||||
filter=suffix(".pp-deskew.png"),
|
||||
output=".pp-clean.png",
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def preprocess_clean(
|
||||
input_file,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
|
||||
if not options.clean:
|
||||
re_symlink(input_file, output_file, log)
|
||||
return
|
||||
|
||||
pageinfo = get_pageinfo(input_file, pdfinfo, pdfinfo_lock)
|
||||
dpi = int(pageinfo['xres'])
|
||||
|
||||
unpaper.clean(input_file, output_file, dpi, log)
|
||||
|
||||
|
||||
@active_if(options.pdf_renderer == 'hocr')
|
||||
@transform(
|
||||
input=preprocess_clean,
|
||||
filter=suffix(".pp-clean.png"),
|
||||
output=".hocr",
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def ocr_tesseract_hocr(
|
||||
input_file,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
|
||||
tesseract.generate_hocr(
|
||||
input_file=input_file,
|
||||
output_hocr=output_file,
|
||||
language=options.language,
|
||||
tessconfig=options.tesseract_config,
|
||||
timeout=options.tesseract_timeout,
|
||||
pageinfo_getter=partial(get_pageinfo, input_file, pdfinfo,
|
||||
pdfinfo_lock),
|
||||
pagesegmode=options.tesseract_pagesegmode,
|
||||
log=log
|
||||
)
|
||||
|
||||
|
||||
@active_if(options.pdf_renderer == 'hocr')
|
||||
@collate(
|
||||
input=[rasterize_with_ghostscript, preprocess_deskew, preprocess_clean],
|
||||
filter=regex(r".*/(\d{6})(?:\.page|\.pp-deskew|\.pp-clean)\.png"),
|
||||
output=os.path.join(work_folder, r'\1.image'),
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def select_image_for_pdf(
|
||||
infiles,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
if options.clean_final:
|
||||
image_suffix = '.pp-clean.png'
|
||||
elif options.deskew:
|
||||
image_suffix = '.pp-deskew.png'
|
||||
else:
|
||||
image_suffix = '.page.png'
|
||||
image = next(ii for ii in infiles if ii.endswith(image_suffix))
|
||||
|
||||
pageinfo = get_pageinfo(image, pdfinfo, pdfinfo_lock)
|
||||
if all(image['enc'] == 'jpeg' for image in pageinfo['images']):
|
||||
# If all images were JPEGs originally, produce a JPEG as output
|
||||
Image.open(image).save(output_file, format='JPEG')
|
||||
else:
|
||||
re_symlink(image, output_file)
|
||||
|
||||
|
||||
@active_if(options.pdf_renderer == 'hocr')
|
||||
@collate(
|
||||
input=[select_image_for_pdf, split_pages],
|
||||
filter=regex(r".*/(\d{6})(?:\.image|\.ocr\.page\.pdf)"),
|
||||
output=os.path.join(work_folder, r'\1.image-layer.pdf'),
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def select_image_layer(
|
||||
infiles,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
|
||||
page_pdf = next(ii for ii in infiles if ii.endswith('.page.pdf'))
|
||||
image = next(ii for ii in infiles if ii.endswith('.image'))
|
||||
|
||||
if lossless_reconstruction:
|
||||
re_symlink(page_pdf, output_file)
|
||||
else:
|
||||
pageinfo = get_pageinfo(image, pdfinfo, pdfinfo_lock)
|
||||
dpi = round(max(pageinfo['xres'], pageinfo['yres'], options.oversample))
|
||||
with open(output_file, 'wb') as pdf:
|
||||
img2pdf.convert([image], dpi=dpi, outputstream=pdf)
|
||||
|
||||
|
||||
@active_if(options.pdf_renderer == 'hocr')
|
||||
@transform(
|
||||
input=ocr_tesseract_hocr,
|
||||
filter=suffix('.hocr'),
|
||||
output='.hocr.pdf',
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def render_hocr_page(
|
||||
input_file,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
hocr = input_file
|
||||
pageinfo = get_pageinfo(hocr, pdfinfo, pdfinfo_lock)
|
||||
dpi = round(max(pageinfo['xres'], pageinfo['yres'], options.oversample))
|
||||
|
||||
hocrtransform = HocrTransform(hocr, dpi)
|
||||
hocrtransform.to_pdf(output_file, imageFileName=None,
|
||||
showBoundingboxes=False, invisibleText=True)
|
||||
|
||||
|
||||
@active_if(options.pdf_renderer == 'hocr')
|
||||
@active_if(options.debug_rendering)
|
||||
@collate(
|
||||
input=[select_image_for_pdf, ocr_tesseract_hocr],
|
||||
filter=regex(r".*/(\d{6})(?:\.image|\.hocr)"),
|
||||
output=os.path.join(work_folder, r'\1.debug.pdf'),
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def render_hocr_debug_page(
|
||||
infiles,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
hocr = next(ii for ii in infiles if ii.endswith('.hocr'))
|
||||
image = next(ii for ii in infiles if ii.endswith('.image'))
|
||||
|
||||
pageinfo = get_pageinfo(image, pdfinfo, pdfinfo_lock)
|
||||
dpi = round(max(pageinfo['xres'], pageinfo['yres'], options.oversample))
|
||||
|
||||
hocrtransform = HocrTransform(hocr, dpi)
|
||||
hocrtransform.to_pdf(output_file, imageFileName=None,
|
||||
showBoundingboxes=True, invisibleText=False)
|
||||
|
||||
|
||||
@active_if(options.pdf_renderer == 'hocr')
|
||||
@collate(
|
||||
input=[render_hocr_page, select_image_layer],
|
||||
filter=regex(r".*/(\d{6})(?:\.hocr\.pdf|\.image-layer\.pdf)"),
|
||||
output=os.path.join(work_folder, r'\1.rendered.pdf'),
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def add_text_layer(
|
||||
infiles,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
text = next(ii for ii in infiles if ii.endswith('.hocr.pdf'))
|
||||
image = next(ii for ii in infiles if ii.endswith('.image-layer.pdf'))
|
||||
|
||||
pdf_output = pypdf.PdfFileWriter()
|
||||
|
||||
pdf_text = pypdf.PdfFileReader(open(text, "rb"))
|
||||
pdf_image = pypdf.PdfFileReader(open(image, "rb"))
|
||||
|
||||
page = pdf_text.getPage(0)
|
||||
page.mergePage(pdf_image.getPage(0))
|
||||
|
||||
pdf_output.addPage(page)
|
||||
|
||||
with open(output_file, "wb") as out:
|
||||
pdf_output.write(out)
|
||||
|
||||
|
||||
@active_if(options.pdf_renderer == 'tesseract')
|
||||
@collate(
|
||||
input=[preprocess_clean, split_pages],
|
||||
filter=regex(r".*/(\d{6})(?:\.pp-clean\.png|\.page\.pdf)"),
|
||||
output=os.path.join(work_folder, r'\1.rendered.pdf'),
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def tesseract_ocr_and_render_pdf(
|
||||
input_files,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
|
||||
input_image = next((ii for ii in input_files if ii.endswith('.png')), '')
|
||||
input_pdf = next((ii for ii in input_files if ii.endswith('.pdf')))
|
||||
if not input_image:
|
||||
# Skipping this page
|
||||
re_symlink(input_pdf, output_file)
|
||||
return
|
||||
|
||||
tesseract.generate_pdf(
|
||||
input_image=input_image,
|
||||
skip_pdf=input_pdf,
|
||||
output_pdf=output_file,
|
||||
language=options.language,
|
||||
tessconfig=options.tesseract_config,
|
||||
timeout=options.tesseract_timeout,
|
||||
pagesegmode=options.tesseract_pagesegmode,
|
||||
log=log)
|
||||
|
||||
|
||||
@transform(
|
||||
input=repair_pdf,
|
||||
filter=formatter(r'\.repaired\.pdf'),
|
||||
output=os.path.join(work_folder, 'pdfa_def.ps'),
|
||||
extras=[_log])
|
||||
def generate_postscript_stub(
|
||||
input_file,
|
||||
output_file,
|
||||
log):
|
||||
|
||||
pdf = pypdf.PdfFileReader(input_file)
|
||||
|
||||
def from_document_info(key):
|
||||
# pdf.documentInfo.get() DOES NOT behave as expected for a dict-like
|
||||
# object, so call with precautions. TypeError may occur if the PDF
|
||||
# is missing the optional document info section.
|
||||
try:
|
||||
s = pdf.documentInfo[key]
|
||||
return str(s)
|
||||
except (KeyError, TypeError):
|
||||
return ''
|
||||
|
||||
pdfmark = {
|
||||
'title': from_document_info('/Title'),
|
||||
'author': from_document_info('/Author'),
|
||||
'keywords': from_document_info('/Keywords'),
|
||||
'subject': from_document_info('/Subject'),
|
||||
}
|
||||
if options.title:
|
||||
pdfmark['title'] = options.title
|
||||
if options.author:
|
||||
pdfmark['author'] = options.author
|
||||
if options.keywords:
|
||||
pdfmark['keywords'] = options.keywords
|
||||
if options.subject:
|
||||
pdfmark['subject'] = options.subject
|
||||
|
||||
pdfmark['creator'] = '{0} {1} / Tesseract OCR{2} {3}'.format(
|
||||
parser.prog, VERSION,
|
||||
'+PDF' if options.pdf_renderer == 'tesseract' else '',
|
||||
tesseract.version())
|
||||
|
||||
generate_pdfa_def(output_file, pdfmark)
|
||||
|
||||
|
||||
@transform(
|
||||
input=split_pages,
|
||||
filter=suffix('.skip.page.pdf'),
|
||||
output='.done.pdf',
|
||||
output_dir=work_folder,
|
||||
extras=[_log])
|
||||
def skip_page(
|
||||
input_file,
|
||||
output_file,
|
||||
log):
|
||||
re_symlink(input_file, output_file, log)
|
||||
|
||||
|
||||
@merge(
|
||||
input=[add_text_layer, render_hocr_debug_page, skip_page,
|
||||
tesseract_ocr_and_render_pdf, generate_postscript_stub],
|
||||
output=os.path.join(work_folder, 'merged.pdf'),
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def merge_pages(
|
||||
input_files,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
|
||||
def input_file_order(s):
|
||||
'''Sort order: All rendered pages followed
|
||||
by their debug page, if any, followed by Postscript stub.
|
||||
Ghostscript documentation has the Postscript stub at the
|
||||
beginning, but it works at the end and also gets document info
|
||||
right that way.'''
|
||||
if s.endswith('.ps'):
|
||||
return 99999999
|
||||
key = int(os.path.basename(s)[0:6]) * 10
|
||||
if 'debug' in os.path.basename(s):
|
||||
key += 1
|
||||
return key
|
||||
|
||||
pdf_pages = sorted(input_files, key=input_file_order)
|
||||
log.info(pdf_pages)
|
||||
ghostscript.generate_pdfa(pdf_pages, output_file, options.jobs or 1)
|
||||
|
||||
|
||||
@transform(
|
||||
input=merge_pages,
|
||||
filter=formatter(),
|
||||
output=options.output_file,
|
||||
extras=[_log, _pdfinfo, _pdfinfo_lock])
|
||||
def copy_final(
|
||||
input_file,
|
||||
output_file,
|
||||
log,
|
||||
pdfinfo,
|
||||
pdfinfo_lock):
|
||||
shutil.copy(input_file, output_file)
|
||||
|
||||
|
||||
def validate_pdfa(
|
||||
input_file,
|
||||
log):
|
||||
return qpdf.check(input_file, log)
|
||||
|
||||
|
||||
def available_cpu_count():
|
||||
try:
|
||||
return multiprocessing.cpu_count()
|
||||
except NotImplementedError:
|
||||
pass
|
||||
|
||||
try:
|
||||
import psutil
|
||||
return psutil.cpu_count()
|
||||
except (ImportError, AttributeError):
|
||||
pass
|
||||
|
||||
complain(
|
||||
"Could not get CPU count. Assuming one (1) CPU."
|
||||
"Use -j N to set manually.")
|
||||
return 1
|
||||
|
||||
|
||||
def cleanup_ruffus_error_message(msg):
|
||||
msg = re.sub(r'\s+', r' ', msg, re.MULTILINE)
|
||||
msg = re.sub(r"\((.+?)\)", r'\1', msg)
|
||||
msg = msg.strip()
|
||||
return msg
|
||||
|
||||
|
||||
def run_pipeline():
|
||||
if not options.jobs:
|
||||
options.jobs = available_cpu_count()
|
||||
try:
|
||||
options.history_file = os.path.join(work_folder, 'ruffus_history.sqlite')
|
||||
cmdline.run(options)
|
||||
except ruffus_exceptions.RethrownJobError as e:
|
||||
if options.verbose:
|
||||
print(e)
|
||||
|
||||
# Yuck. Hunt through the ruffus exception to find out what the
|
||||
# return code is supposed to be.
|
||||
for exc in e.args:
|
||||
task_name, job_name, exc_name, exc_value, exc_stack = exc
|
||||
if exc_name == 'builtins.SystemExit':
|
||||
match = re.search(r"\.(.+?)\)", exc_value)
|
||||
exit_code_name = match.groups()[0]
|
||||
exit_code = getattr(ExitCode, exit_code_name, 'other_error')
|
||||
return exit_code
|
||||
elif exc_name == 'ruffus.ruffus_exceptions.MissingInputFileError':
|
||||
print(cleanup_ruffus_error_message(exc_value))
|
||||
return ExitCode.input_file
|
||||
elif exc_name == 'builtins.TypeError':
|
||||
# Even though repair_pdf will fail, ruffus will still try
|
||||
# to call split_pages with no input files, likely due to a bug
|
||||
if task_name == 'split_pages':
|
||||
print("Input file '{0}' is not a valid PDF".format(
|
||||
options.input_file))
|
||||
return ExitCode.input_file
|
||||
|
||||
return ExitCode.other_error
|
||||
|
||||
if not validate_pdfa(options.output_file, _log):
|
||||
_log.warning('Output file: The generated PDF/A file is INVALID')
|
||||
return ExitCode.invalid_output_pdfa
|
||||
|
||||
return ExitCode.ok
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(run_pipeline())
|
||||
@@ -0,0 +1,166 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
from subprocess import Popen, PIPE
|
||||
from decimal import Decimal, getcontext
|
||||
import re
|
||||
import sys
|
||||
import PyPDF2 as pypdf
|
||||
|
||||
|
||||
FRIENDLY_COLORSPACE = {
|
||||
'/DeviceGray': 'gray',
|
||||
'/CalGray': 'gray',
|
||||
'/DeviceRGB': 'rgb',
|
||||
'/CalRGB': 'rgb',
|
||||
'/DeviceCMYK': 'cmyk',
|
||||
'/Lab': 'lab',
|
||||
'/ICCBased': 'icc',
|
||||
'/Indexed': 'index',
|
||||
'/Separation': 'sep',
|
||||
'/DeviceN': 'devn',
|
||||
'/Pattern': '-'
|
||||
}
|
||||
|
||||
FRIENDLY_ENCODING = {
|
||||
'/CCITTFaxDecode': 'ccitt',
|
||||
'/DCTDecode': 'jpeg',
|
||||
'/JPXDecode': 'jpx',
|
||||
'/JBIG2Decode': 'jbig2',
|
||||
}
|
||||
|
||||
FRIENDLY_COMP = {
|
||||
'gray': 1,
|
||||
'rgb': 3,
|
||||
'cmyk': 4,
|
||||
'lab': 3,
|
||||
'index': 1
|
||||
}
|
||||
|
||||
|
||||
def _page_has_inline_images(page):
|
||||
# PDF always uses \r\n for separator regardless of platform
|
||||
# Really basic heuristic that might trigger the odd false positive
|
||||
# This is only finds the first image and is not quite spec compliant
|
||||
try:
|
||||
contents = page.getContents()
|
||||
data = contents.getData()
|
||||
except AttributeError:
|
||||
# If we can't access the contents or data (empty page?) then there
|
||||
# are no inline images
|
||||
return False
|
||||
|
||||
begin_image, image_data, end_image = False, False, False
|
||||
for data in re.split(b'\s+', data):
|
||||
if data == b'BI':
|
||||
begin_image = True
|
||||
elif data == b'ID':
|
||||
image_data = True
|
||||
elif data == b'EI':
|
||||
end_image = True
|
||||
if all((begin_image, image_data, end_image)):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _find_page_images(page, pageinfo):
|
||||
try:
|
||||
page['/Resources']['/XObject']
|
||||
except KeyError:
|
||||
return
|
||||
|
||||
# Look for XObject (out of line images)
|
||||
for xobj in page['/Resources']['/XObject']:
|
||||
# PyPDF2 returns the keys as an iterator
|
||||
pdfimage = page['/Resources']['/XObject'][xobj]
|
||||
if pdfimage['/Subtype'] != '/Image':
|
||||
continue
|
||||
if '/ImageMask' in pdfimage:
|
||||
if pdfimage['/ImageMask']:
|
||||
continue
|
||||
image = {}
|
||||
image['width'] = pdfimage['/Width']
|
||||
image['height'] = pdfimage['/Height']
|
||||
image['bpc'] = pdfimage['/BitsPerComponent']
|
||||
if '/Filter' in pdfimage:
|
||||
filter_ = pdfimage['/Filter']
|
||||
if isinstance(filter_, pypdf.generic.ArrayObject):
|
||||
filter_ = filter_[0]
|
||||
image['enc'] = FRIENDLY_ENCODING.get(filter_, 'image')
|
||||
else:
|
||||
image['enc'] = 'image'
|
||||
if '/ColorSpace' in pdfimage:
|
||||
cs = pdfimage['/ColorSpace']
|
||||
if isinstance(cs, pypdf.generic.ArrayObject):
|
||||
cs = cs[0]
|
||||
image['color'] = FRIENDLY_COLORSPACE.get(cs, '-')
|
||||
else:
|
||||
image['color'] = 'jpx' if image['enc'] == 'jpx' else '?'
|
||||
|
||||
image['comp'] = FRIENDLY_COMP.get(image['color'], '?')
|
||||
image['dpi_w'] = image['width'] / pageinfo['width_inches']
|
||||
image['dpi_h'] = image['height'] / pageinfo['height_inches']
|
||||
image['dpi'] = (image['dpi_w'] * image['dpi_h']) ** Decimal(0.5)
|
||||
yield image
|
||||
|
||||
|
||||
def _page_has_text(pdf, page):
|
||||
# Simple test
|
||||
text = page.extractText()
|
||||
if text.strip() != '':
|
||||
return True
|
||||
|
||||
# More nuanced test to deal with quirks of Tesseract PDF generation
|
||||
# Check if there's a Glyphless font
|
||||
try:
|
||||
font = page['/Resources']['/Font']
|
||||
except KeyError:
|
||||
pass
|
||||
else:
|
||||
font_objects = list(font.keys())
|
||||
for font_object in font_objects:
|
||||
basefont = font[font_object]['/BaseFont']
|
||||
if basefont.endswith('GlyphLessFont'):
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def _pdf_get_pageinfo(infile, pageno: int):
|
||||
pageinfo = {}
|
||||
pageinfo['pageno'] = pageno
|
||||
pageinfo['images'] = []
|
||||
|
||||
pdf = pypdf.PdfFileReader(infile)
|
||||
page = pdf.pages[pageno]
|
||||
|
||||
pageinfo['has_text'] = _page_has_text(pdf, page)
|
||||
|
||||
width_pt = page['/MediaBox'][2] - page['/MediaBox'][0]
|
||||
height_pt = page['/MediaBox'][3] - page['/MediaBox'][1]
|
||||
pageinfo['width_inches'] = width_pt / Decimal(72.0)
|
||||
pageinfo['height_inches'] = height_pt / Decimal(72.0)
|
||||
|
||||
pageinfo['images'] = [im for im in _find_page_images(page, pageinfo)]
|
||||
|
||||
# Look for inline images
|
||||
if _page_has_inline_images(page):
|
||||
raise NotImplementedError(
|
||||
"Warning: input PDF contains inline images - not supported")
|
||||
|
||||
if pageinfo['images']:
|
||||
xres = max(image['dpi_w'] for image in pageinfo['images'])
|
||||
yres = max(image['dpi_h'] for image in pageinfo['images'])
|
||||
pageinfo['xres'], pageinfo['yres'] = xres, yres
|
||||
pageinfo['width_pixels'] = \
|
||||
int(round(xres * pageinfo['width_inches']))
|
||||
pageinfo['height_pixels'] = \
|
||||
int(round(yres * pageinfo['height_inches']))
|
||||
|
||||
return pageinfo
|
||||
|
||||
|
||||
def pdf_get_all_pageinfo(infile):
|
||||
pdf = pypdf.PdfFileReader(infile)
|
||||
getcontext().prec = 6
|
||||
return [_pdf_get_pageinfo(infile, n) for n in range(pdf.numPages)]
|
||||
@@ -0,0 +1,136 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# Generate a PDFA_def.ps file for Ghostscript >= 9.14
|
||||
|
||||
from __future__ import print_function, absolute_import, division
|
||||
from string import Template
|
||||
from subprocess import Popen, PIPE
|
||||
import os
|
||||
import codecs
|
||||
from . import get_program
|
||||
|
||||
|
||||
# This is a template written in PostScript which is needed to create PDF/A
|
||||
# files, from the Ghostscript documentation. Lines beginning with % are
|
||||
# comments. Python substitution variables have a '$' prefix.
|
||||
pdfa_def_template = u"""%!
|
||||
% This is a sample prefix file for creating a PDF/A document.
|
||||
% Feel free to modify entries marked with "Customize".
|
||||
% This assumes an ICC profile to reside in the file (ISO Coated sb.icc),
|
||||
% unless the user modifies the corresponding line below.
|
||||
|
||||
% Define entries in the document Info dictionary :
|
||||
/ICCProfile ($icc_profile)
|
||||
def
|
||||
|
||||
[ /Title <$title>
|
||||
/Author <$author>
|
||||
/Subject <$subject>
|
||||
/Keywords <$keywords>
|
||||
/Creator <$creator>
|
||||
/DOCINFO pdfmark
|
||||
|
||||
% Define an ICC profile :
|
||||
|
||||
[/_objdef {icc_PDFA} /type /stream /OBJ pdfmark
|
||||
[{icc_PDFA}
|
||||
<<
|
||||
/N currentpagedevice /ProcessColorModel known {
|
||||
currentpagedevice /ProcessColorModel get dup /DeviceGray eq
|
||||
{pop 1} {
|
||||
/DeviceRGB eq
|
||||
{3}{4} ifelse
|
||||
} ifelse
|
||||
} {
|
||||
(ERROR, unable to determine ProcessColorModel) == flush
|
||||
} ifelse
|
||||
>> /PUT pdfmark
|
||||
[{icc_PDFA} ICCProfile (r) file /PUT pdfmark
|
||||
|
||||
% Define the output intent dictionary :
|
||||
|
||||
[/_objdef {OutputIntent_PDFA} /type /dict /OBJ pdfmark
|
||||
[{OutputIntent_PDFA} <<
|
||||
/Type /OutputIntent % Must be so (the standard requires).
|
||||
/S /GTS_PDFA1 % Must be so (the standard requires).
|
||||
/DestOutputProfile {icc_PDFA} % Must be so (see above).
|
||||
/OutputConditionIdentifier ($icc_identifier)
|
||||
>> /PUT pdfmark
|
||||
[{Catalog} <</OutputIntents [ {OutputIntent_PDFA} ]>> /PUT pdfmark
|
||||
"""
|
||||
|
||||
|
||||
def encode_text_string(s: str) -> str:
|
||||
'''Encode text string to hex string for use in a PDF
|
||||
|
||||
From PDF 32000-1:2008 a string object may be included in hexademical form
|
||||
if it is enclosed in angle brackets. For general Unicode the string should
|
||||
be UTF-16 (big endian) with byte order marks. A non-hexademical
|
||||
representation is doable but this is preferable since it allows the output
|
||||
Postscript file to be completely ASCII and no escaping of Postscript
|
||||
characters is necessary.
|
||||
'''
|
||||
if s == '':
|
||||
return ''
|
||||
utf16_bytes = s.encode('utf-16be')
|
||||
ascii_hex_bytes = codecs.encode(b'\xfe\xff' + utf16_bytes, 'hex')
|
||||
ascii_hex_str = ascii_hex_bytes.decode('ascii').lower()
|
||||
return ascii_hex_str
|
||||
|
||||
|
||||
def _get_pdfa_def(icc_profile, icc_identifier, pdfmark):
|
||||
pdfmark_utf16 = {k: encode_text_string(v) for k, v in pdfmark.items()}
|
||||
|
||||
t = Template(pdfa_def_template)
|
||||
result = t.substitute(icc_profile=icc_profile,
|
||||
icc_identifier=icc_identifier,
|
||||
title=pdfmark_utf16.get('title', ''),
|
||||
author=pdfmark_utf16.get('author', ''),
|
||||
subject=pdfmark_utf16.get('subject', ''),
|
||||
creator=pdfmark_utf16.get('creator', ''),
|
||||
keywords=pdfmark_utf16.get('keywords', ''))
|
||||
return result
|
||||
|
||||
|
||||
def _get_postscript_icc_path():
|
||||
"Parse Ghostscript's help message to find where iccprofiles are stored"
|
||||
|
||||
p_gs = Popen([get_program('gs'), '--help'], close_fds=True,
|
||||
universal_newlines=True,
|
||||
stdout=PIPE, stderr=PIPE)
|
||||
out, _ = p_gs.communicate()
|
||||
lines = out.splitlines()
|
||||
|
||||
def search_paths(lines):
|
||||
seeking = True
|
||||
for line in lines:
|
||||
if seeking:
|
||||
if line.startswith('Search path'):
|
||||
seeking = False
|
||||
continue
|
||||
else:
|
||||
if line.strip().startswith('/'):
|
||||
yield from (
|
||||
path.strip() for path in line.split(':')
|
||||
if path.strip() != '')
|
||||
for root in search_paths(lines):
|
||||
path = os.path.realpath(os.path.join(root, '../iccprofiles'))
|
||||
if os.path.exists(path):
|
||||
return path
|
||||
|
||||
raise FileNotFoundError("Could not find Ghostscript's iccprofiles")
|
||||
|
||||
|
||||
def generate_pdfa_def(target_filename, pdfmark, icc='sRGB'):
|
||||
if icc == 'sRGB':
|
||||
icc_profile = os.path.join(_get_postscript_icc_path(), 'srgb.icc')
|
||||
else:
|
||||
raise NotImplementedError("Only supporting sRGB")
|
||||
|
||||
ps = _get_pdfa_def(icc_profile, icc, pdfmark)
|
||||
|
||||
# Since PostScript might not handle UTF-8 (it's hard to get a clear
|
||||
# answer), insist on ascii
|
||||
with open(target_filename, 'w', encoding='ascii') as f:
|
||||
f.write(ps)
|
||||
@@ -0,0 +1,84 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
from subprocess import CalledProcessError, check_output, STDOUT, check_call
|
||||
import sys
|
||||
import os
|
||||
|
||||
from . import ExitCode, get_program
|
||||
|
||||
|
||||
def check(input_file, log):
|
||||
args_qpdf = [
|
||||
get_program('qpdf'),
|
||||
'--check',
|
||||
input_file
|
||||
]
|
||||
|
||||
try:
|
||||
check_output(args_qpdf, stderr=STDOUT, universal_newlines=True)
|
||||
except CalledProcessError as e:
|
||||
if e.returncode == 2:
|
||||
print("{0}: not a valid PDF, and could not repair it.".format(
|
||||
input_file))
|
||||
print("Details:")
|
||||
print(e.output)
|
||||
elif e.returncode == 3:
|
||||
log.info("qpdf --check returned warnings:")
|
||||
log.info(e.output)
|
||||
else:
|
||||
print(e.output)
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def repair(input_file, output_file, log):
|
||||
args_qpdf = [
|
||||
get_program('qpdf'), input_file, output_file
|
||||
]
|
||||
try:
|
||||
check_output(args_qpdf, stderr=STDOUT, universal_newlines=True)
|
||||
except CalledProcessError as e:
|
||||
if e.returncode == 3 and e.output.find("operation succeeded"):
|
||||
log.debug('qpdf found and fixed errors:')
|
||||
log.debug(e.output)
|
||||
print(e.output)
|
||||
return
|
||||
|
||||
if e.returncode == 2 and e.output.find("invalid password"):
|
||||
print("{0}: this PDF is password-protected - password must "
|
||||
"be removed for OCR".format(input_file))
|
||||
sys.exit(ExitCode.input_file)
|
||||
elif e.returncode == 2:
|
||||
print("{0}: not a valid PDF, and could not repair it.".format(
|
||||
input_file))
|
||||
print("Details:")
|
||||
print(e.output)
|
||||
sys.exit(ExitCode.input_file)
|
||||
else:
|
||||
print("{0}: unknown error".format(
|
||||
input_file))
|
||||
print(e.output)
|
||||
sys.exit(ExitCode.unknown)
|
||||
|
||||
|
||||
def get_npages(input_file):
|
||||
pages = check_output(
|
||||
[get_program('qpdf'), '--show-npages', input_file],
|
||||
universal_newlines=True, close_fds=True)
|
||||
return int(pages)
|
||||
|
||||
|
||||
def split_pages(input_file, work_folder, npages):
|
||||
"""Split multipage PDF into individual pages.
|
||||
|
||||
Incredibly enough, this multiple process approach is about 70 times
|
||||
faster than using Ghostscript.
|
||||
"""
|
||||
for n in range(int(npages)):
|
||||
args_qpdf = [
|
||||
get_program('qpdf'), input_file,
|
||||
'--pages', input_file, '{0}'.format(n + 1), '--',
|
||||
os.path.join(work_folder, '{0:06d}.page.pdf'.format(n + 1))
|
||||
]
|
||||
check_call(args_qpdf)
|
||||
@@ -0,0 +1,181 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
import sys
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
from functools import lru_cache
|
||||
from . import ExitCode, get_program
|
||||
|
||||
from subprocess import Popen, PIPE, CalledProcessError, \
|
||||
TimeoutExpired, check_output, STDOUT
|
||||
try:
|
||||
from subprocess import DEVNULL
|
||||
except ImportError:
|
||||
DEVNULL = open(os.devnull, 'wb')
|
||||
|
||||
|
||||
HOCR_TEMPLATE = '''<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||
<head>
|
||||
<title></title>
|
||||
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 3.02.02' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word'/>
|
||||
</head>
|
||||
<body>
|
||||
<div class='ocr_page' id='page_1' title='image "x.tif"; bbox 0 0 {0} {1}; ppageno 0'>
|
||||
<div class='ocr_carea' id='block_1_1' title="bbox 0 1 {0} {1}">
|
||||
<p class='ocr_par' dir='ltr' id='par_1' title="bbox 0 1 {0} {1}">
|
||||
<span class='ocr_line' id='line_1' title="bbox 0 1 {0} {1}"><span class='ocrx_word' id='word_1' title="bbox 0 1 {0} {1}"> </span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>'''
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def version():
|
||||
args_tess = [
|
||||
get_program('tesseract'),
|
||||
'--version'
|
||||
]
|
||||
try:
|
||||
versions = check_output(
|
||||
args_tess, close_fds=True, universal_newlines=True,
|
||||
stderr=STDOUT)
|
||||
except CalledProcessError:
|
||||
print("Could not find Tesseract executable on system PATH.")
|
||||
sys.exit(ExitCode.missing_dependency)
|
||||
|
||||
tesseract_version = re.match(r'tesseract\s(.+)', versions).group(1)
|
||||
return tesseract_version
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def languages():
|
||||
args_tess = [
|
||||
get_program('tesseract'),
|
||||
'--list-langs'
|
||||
]
|
||||
try:
|
||||
langs = check_output(
|
||||
args_tess, close_fds=True, universal_newlines=True,
|
||||
stderr=STDOUT)
|
||||
except CalledProcessError as e:
|
||||
print("Tesseract failed to report available languages.")
|
||||
print("Output from Tesseract:")
|
||||
print("-" * 40)
|
||||
print(e.output)
|
||||
sys.exit(ExitCode.missing_dependency)
|
||||
return set(lang.strip() for lang in langs.splitlines()[1:])
|
||||
|
||||
|
||||
def generate_hocr(input_file, output_hocr, language: list, tessconfig: list,
|
||||
timeout: float, pageinfo_getter, pagesegmode: int, log):
|
||||
|
||||
badxml = os.path.splitext(output_hocr)[0] + '.badxml'
|
||||
|
||||
args_tesseract = [
|
||||
get_program('tesseract'),
|
||||
'-l', '+'.join(language)
|
||||
]
|
||||
|
||||
if pagesegmode is not None:
|
||||
args_tesseract.extend(['-psm', str(pagesegmode)])
|
||||
|
||||
args_tesseract.extend([
|
||||
input_file,
|
||||
badxml,
|
||||
'hocr'
|
||||
] + tessconfig)
|
||||
p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE,
|
||||
universal_newlines=True)
|
||||
try:
|
||||
stdout, stderr = p.communicate(timeout=timeout)
|
||||
except TimeoutExpired:
|
||||
p.kill()
|
||||
stdout, stderr = p.communicate()
|
||||
# Generate a HOCR file with no recognized text if tesseract times out
|
||||
# Temporary workaround to hocrTransform not being able to function if
|
||||
# it does not have a valid hOCR file.
|
||||
with open(output_hocr, 'w', encoding="utf-8") as f:
|
||||
pageinfo = pageinfo_getter()
|
||||
f.write(HOCR_TEMPLATE.format(
|
||||
pageinfo['width_pixels'],
|
||||
pageinfo['height_pixels']))
|
||||
else:
|
||||
if stdout:
|
||||
log.info(stdout)
|
||||
if stderr:
|
||||
log.error(stderr)
|
||||
|
||||
if p.returncode != 0:
|
||||
raise CalledProcessError(p.returncode, args_tesseract)
|
||||
|
||||
if os.path.exists(badxml + '.html'):
|
||||
# Tesseract 3.02 appends suffix ".html" on its own (.badxml.html)
|
||||
shutil.move(badxml + '.html', badxml)
|
||||
elif os.path.exists(badxml + '.hocr'):
|
||||
# Tesseract 3.03 appends suffix ".hocr" on its own (.badxml.hocr)
|
||||
shutil.move(badxml + '.hocr', badxml)
|
||||
|
||||
# Tesseract 3.03 inserts source filename into hocr file without
|
||||
# escaping it, creating invalid XML and breaking the parser.
|
||||
# As a workaround, rewrite the hocr file, replacing the filename
|
||||
# with a space. Don't know if Tesseract 3.02 does the same.
|
||||
|
||||
regex_nested_single_quotes = re.compile(
|
||||
r"""title='image "([^"]*)";""")
|
||||
with open(badxml, mode='r', encoding='utf-8') as f_in, \
|
||||
open(output_hocr, mode='w', encoding='utf-8') as f_out:
|
||||
for line in f_in:
|
||||
line = regex_nested_single_quotes.sub(
|
||||
r"""title='image " ";""", line)
|
||||
f_out.write(line)
|
||||
|
||||
|
||||
def generate_pdf(input_image, skip_pdf, output_pdf, language: list,
|
||||
tessconfig: list, timeout: float, pagesegmode: int, log):
|
||||
'''Use Tesseract to render a PDF.
|
||||
|
||||
input_image -- image to analyze
|
||||
skip_pdf -- if we time out, use this file as output
|
||||
language -- list of languages to consider
|
||||
tessconfig -- tesseract configuration
|
||||
timeout -- timeout (seconds)
|
||||
log -- logger object
|
||||
'''
|
||||
|
||||
args_tesseract = [
|
||||
get_program('tesseract'),
|
||||
'-l', '+'.join(language)
|
||||
]
|
||||
|
||||
if pagesegmode is not None:
|
||||
args_tesseract.extend(['-psm', str(pagesegmode)])
|
||||
|
||||
args_tesseract.extend([
|
||||
input_image,
|
||||
os.path.splitext(output_pdf)[0], # Tesseract appends suffix
|
||||
'pdf'
|
||||
] + tessconfig)
|
||||
p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE,
|
||||
universal_newlines=True)
|
||||
|
||||
try:
|
||||
stdout, stderr = p.communicate(timeout=timeout)
|
||||
if stdout:
|
||||
log.info(stdout)
|
||||
if stderr:
|
||||
log.error(stderr)
|
||||
except TimeoutExpired:
|
||||
p.kill()
|
||||
log.info("Tesseract - page timed out")
|
||||
shutil.copy(skip_pdf, output_pdf)
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
# unpaper documentation:
|
||||
# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md
|
||||
|
||||
from subprocess import Popen, PIPE
|
||||
from tempfile import NamedTemporaryFile
|
||||
import sys
|
||||
import os
|
||||
from functools import lru_cache
|
||||
from . import ExitCode, get_program
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def version():
|
||||
args_unpaper = [
|
||||
get_program('unpaper'),
|
||||
'--version'
|
||||
]
|
||||
p_unpaper = Popen(args_unpaper, close_fds=True, universal_newlines=True,
|
||||
stdout=PIPE, stderr=PIPE)
|
||||
version, _ = p_unpaper.communicate(timeout=5)
|
||||
|
||||
return version.strip()
|
||||
|
||||
|
||||
try:
|
||||
from PIL import Image
|
||||
except ImportError:
|
||||
print("Could not find Python3 imaging library", file=sys.stderr)
|
||||
raise
|
||||
|
||||
|
||||
def run(input_file, output_file, dpi, log, mode_args):
|
||||
args_unpaper = [
|
||||
get_program('unpaper'),
|
||||
'-v',
|
||||
'--dpi', str(dpi)
|
||||
] + mode_args
|
||||
|
||||
SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'}
|
||||
|
||||
im = Image.open(input_file)
|
||||
if im.mode not in SUFFIXES.keys():
|
||||
log.info("Converting image to other colorspace")
|
||||
try:
|
||||
if im.mode == 'P' and len(im.getcolors()) == 2:
|
||||
im = im.convert(mode='1')
|
||||
else:
|
||||
im = im.convert(mode='RGB')
|
||||
except IOError:
|
||||
log.error(
|
||||
"Could not convert image with type " + im.mode)
|
||||
sys.exit(ExitCode.missing_dependency)
|
||||
|
||||
try:
|
||||
suffix = SUFFIXES[im.mode]
|
||||
except KeyError:
|
||||
log.error(
|
||||
"Failed to convert image to a supported format.")
|
||||
sys.exit(ExitCode.missing_dependency)
|
||||
|
||||
with NamedTemporaryFile(suffix=suffix) as input_pnm, \
|
||||
NamedTemporaryFile(suffix=suffix, mode="r+b") as output_pnm:
|
||||
im.save(input_pnm, format='PPM')
|
||||
im.close()
|
||||
|
||||
os.unlink(output_pnm.name)
|
||||
|
||||
args_unpaper.extend([input_pnm.name, output_pnm.name])
|
||||
p_unpaper = Popen(
|
||||
args_unpaper, close_fds=True,
|
||||
universal_newlines=True, stdout=PIPE, stderr=PIPE
|
||||
)
|
||||
out, err = p_unpaper.communicate()
|
||||
log.debug(out)
|
||||
log.debug(err)
|
||||
|
||||
Image.open(output_pnm.name).save(output_file)
|
||||
|
||||
|
||||
def deskew(input_file, output_file, dpi, log):
|
||||
run(input_file, output_file, dpi, log, [
|
||||
'--mask-scan-size', '100', # don't blank out narrow columns
|
||||
'--no-border-align', # don't align visible content to borders
|
||||
'--no-mask-center', # don't center visible content within page
|
||||
'--no-grayfilter', # don't remove light gray areas
|
||||
'--no-blackfilter', # don't remove solid black areas
|
||||
'--no-noisefilter', # don't remove salt and pepper noise
|
||||
'--no-blurfilter' # don't remove blurry objects/debris
|
||||
])
|
||||
|
||||
|
||||
def clean(input_file, output_file, dpi, log):
|
||||
run(input_file, output_file, dpi, log, [
|
||||
'--mask-scan-size', '100', # don't blank out narrow columns
|
||||
'--no-border-align', # don't align visible content to borders
|
||||
'--no-mask-center', # don't center visible content within page
|
||||
'--no-grayfilter', # don't remove light gray areas
|
||||
'--no-blackfilter', # don't remove solid black areas
|
||||
'--no-deskew', # don't deskew
|
||||
])
|
||||
+266
@@ -0,0 +1,266 @@
|
||||
<?xml version="1.0" encoding="UTF-8" standalone="no"?>
|
||||
<!DOCTYPE svg PUBLIC "-//W3C//DTD SVG 1.1//EN"
|
||||
"http://www.w3.org/Graphics/SVG/1.1/DTD/svg11.dtd">
|
||||
<!-- Generated by graphviz version 2.38.0 (20140413.2041)
|
||||
-->
|
||||
<!-- Title: Pipeline: Pages: 1 -->
|
||||
<svg width="1132pt" height="708pt"
|
||||
viewBox="0.00 0.00 1132.00 708.08" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink">
|
||||
<g id="graph0" class="graph" transform="scale(1 1) rotate(0) translate(4 704.083)">
|
||||
<title>Pipeline:</title>
|
||||
<polygon fill="white" stroke="none" points="-4,4 -4,-704.083 1128,-704.083 1128,4 -4,4"/>
|
||||
<g id="clust1" class="cluster"><title>clustertasks</title>
|
||||
<polygon fill="none" stroke="black" points="8,-8 8,-692.083 1116,-692.083 1116,-8 8,-8"/>
|
||||
<text text-anchor="middle" x="562" y="-664.083" font-family="Times,serif" font-size="30.00" fill="#ff3232">Pipeline:</text>
|
||||
</g>
|
||||
<!-- t0 -->
|
||||
<g id="node1" class="node"><title>t0</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="936.535,-646.083 713.465,-646.083 709.465,-642.083 709.465,-610.083 932.535,-610.083 936.535,-614.083 936.535,-646.083"/>
|
||||
<polyline fill="none" stroke="black" points="932.535,-642.083 709.465,-642.083 "/>
|
||||
<polyline fill="none" stroke="black" points="932.535,-642.083 932.535,-610.083 "/>
|
||||
<polyline fill="none" stroke="black" points="932.535,-642.083 936.535,-646.083 "/>
|
||||
<text text-anchor="middle" x="823" y="-622.083" font-family="Times,serif" font-size="20.00">repair_pdf</text>
|
||||
</g>
|
||||
<!-- t1 -->
|
||||
<g id="node2" class="node"><title>t1</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="914.112,-567.155 710,-584.057 505.888,-567.155 506.078,-539.806 913.922,-539.806 914.112,-567.155"/>
|
||||
<polygon fill="none" stroke="black" points="918.134,-570.834 710,-588.069 501.866,-570.834 502.11,-535.808 917.89,-535.808 918.134,-570.834"/>
|
||||
<text text-anchor="middle" x="710" y="-553.596" font-family="Times,serif" font-size="20.00">split_pages</text>
|
||||
</g>
|
||||
<!-- t0->t1 -->
|
||||
<g id="edge1" class="edge"><title>t0->t1</title>
|
||||
<path fill="none" stroke="#0044a0" d="M793.9,-609.961C783.582,-603.89 771.656,-596.873 760.092,-590.069"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="761.747,-586.982 751.353,-584.927 758.197,-593.015 761.747,-586.982"/>
|
||||
</g>
|
||||
<!-- t12 -->
|
||||
<g id="node14" class="node"><title>t12</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="1108.08,-509.109 769.918,-509.109 765.918,-505.109 765.918,-473.109 1104.08,-473.109 1108.08,-477.109 1108.08,-509.109"/>
|
||||
<polyline fill="none" stroke="black" points="1104.08,-505.109 765.918,-505.109 "/>
|
||||
<polyline fill="none" stroke="black" points="1104.08,-505.109 1104.08,-473.109 "/>
|
||||
<polyline fill="none" stroke="black" points="1104.08,-505.109 1108.08,-509.109 "/>
|
||||
<text text-anchor="middle" x="937" y="-485.109" font-family="Times,serif" font-size="20.00">generate_postscript_stub</text>
|
||||
</g>
|
||||
<!-- t0->t12 -->
|
||||
<g id="edge19" class="edge"><title>t0->t12</title>
|
||||
<path fill="none" stroke="#0044a0" d="M899.32,-610.004C909.979,-604.592 919.748,-597.466 927,-588.083 941.916,-568.78 943.099,-540.385 941.345,-519.486"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="944.8,-518.881 940.218,-509.328 937.843,-519.653 944.8,-518.881"/>
|
||||
</g>
|
||||
<!-- t2 -->
|
||||
<g id="node3" class="node"><title>t2</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="592.299,-509.109 241.701,-509.109 237.701,-505.109 237.701,-473.109 588.299,-473.109 592.299,-477.109 592.299,-509.109"/>
|
||||
<polyline fill="none" stroke="black" points="588.299,-505.109 237.701,-505.109 "/>
|
||||
<polyline fill="none" stroke="black" points="588.299,-505.109 588.299,-473.109 "/>
|
||||
<polyline fill="none" stroke="black" points="588.299,-505.109 592.299,-509.109 "/>
|
||||
<text text-anchor="middle" x="415" y="-485.109" font-family="Times,serif" font-size="20.00">rasterize_with_ghostscript</text>
|
||||
</g>
|
||||
<!-- t1->t2 -->
|
||||
<g id="edge2" class="edge"><title>t1->t2</title>
|
||||
<path fill="none" stroke="#0044a0" d="M608.89,-535.808C573.672,-527.87 534.524,-519.048 500.679,-511.42"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="501.345,-507.982 490.82,-509.198 499.806,-514.811 501.345,-507.982"/>
|
||||
</g>
|
||||
<!-- t7 -->
|
||||
<g id="node9" class="node"><title>t7</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="314.109,-277.109 19.8906,-277.109 15.8906,-273.109 15.8906,-241.109 310.109,-241.109 314.109,-245.109 314.109,-277.109"/>
|
||||
<polyline fill="none" stroke="black" points="310.109,-273.109 15.8906,-273.109 "/>
|
||||
<polyline fill="none" stroke="black" points="310.109,-273.109 310.109,-241.109 "/>
|
||||
<polyline fill="none" stroke="black" points="310.109,-273.109 314.109,-277.109 "/>
|
||||
<text text-anchor="middle" x="165" y="-253.109" font-family="Times,serif" font-size="20.00">select_image_layer</text>
|
||||
</g>
|
||||
<!-- t1->t7 -->
|
||||
<g id="edge11" class="edge"><title>t1->t7</title>
|
||||
<path fill="none" stroke="#0044a0" d="M501.985,-548.028C416.651,-540.897 317.351,-528.985 229,-509.109 131.492,-487.174 17,-534.054 17,-434.109 17,-434.109 17,-434.109 17,-374.109 17,-340.481 4.97908,-324.525 27,-299.109 32.8004,-292.415 39.6539,-286.828 47.1559,-282.17"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="49.1993,-285.038 56.2361,-277.118 45.7959,-278.921 49.1993,-285.038"/>
|
||||
</g>
|
||||
<!-- t13 -->
|
||||
<g id="node12" class="node"><title>t13</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="1029.34,-451.109 808.662,-451.109 804.662,-447.109 804.662,-415.109 1025.34,-415.109 1029.34,-419.109 1029.34,-451.109"/>
|
||||
<polyline fill="none" stroke="black" points="1025.34,-447.109 804.662,-447.109 "/>
|
||||
<polyline fill="none" stroke="black" points="1025.34,-447.109 1025.34,-415.109 "/>
|
||||
<polyline fill="none" stroke="black" points="1025.34,-447.109 1029.34,-451.109 "/>
|
||||
<text text-anchor="middle" x="917" y="-427.109" font-family="Times,serif" font-size="20.00">skip_page</text>
|
||||
</g>
|
||||
<!-- t1->t13 -->
|
||||
<g id="edge16" class="edge"><title>t1->t13</title>
|
||||
<path fill="none" stroke="#0044a0" d="M716.849,-535.484C723.825,-515.965 736.562,-488.763 757,-473.109 768.342,-464.422 781.352,-457.641 794.947,-452.352"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="796.223,-455.612 804.44,-448.923 793.845,-449.029 796.223,-455.612"/>
|
||||
</g>
|
||||
<!-- t11 -->
|
||||
<g id="node13" class="node"><title>t11</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="1068.24,-335.109 687.76,-335.109 683.76,-331.109 683.76,-299.109 1064.24,-299.109 1068.24,-303.109 1068.24,-335.109"/>
|
||||
<polyline fill="none" stroke="black" points="1064.24,-331.109 683.76,-331.109 "/>
|
||||
<polyline fill="none" stroke="black" points="1064.24,-331.109 1064.24,-299.109 "/>
|
||||
<polyline fill="none" stroke="black" points="1064.24,-331.109 1068.24,-335.109 "/>
|
||||
<text text-anchor="middle" x="876" y="-311.109" font-family="Times,serif" font-size="20.00">tesseract_ocr_and_render_pdf</text>
|
||||
</g>
|
||||
<!-- t1->t11 -->
|
||||
<g id="edge18" class="edge"><title>t1->t11</title>
|
||||
<path fill="none" stroke="#0044a0" d="M710.717,-535.657C712.098,-517.744 715.879,-492.693 726,-473.109 754.651,-417.673 809.56,-368.731 844.403,-341.318"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="846.607,-344.038 852.372,-335.148 842.322,-338.503 846.607,-344.038"/>
|
||||
</g>
|
||||
<!-- t3 -->
|
||||
<g id="node4" class="node"><title>t3</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="564.742,-451.109 269.258,-451.109 265.258,-447.109 265.258,-415.109 560.742,-415.109 564.742,-419.109 564.742,-451.109"/>
|
||||
<polyline fill="none" stroke="black" points="560.742,-447.109 265.258,-447.109 "/>
|
||||
<polyline fill="none" stroke="black" points="560.742,-447.109 560.742,-415.109 "/>
|
||||
<polyline fill="none" stroke="black" points="560.742,-447.109 564.742,-451.109 "/>
|
||||
<text text-anchor="middle" x="415" y="-427.109" font-family="Times,serif" font-size="20.00">preprocess_deskew</text>
|
||||
</g>
|
||||
<!-- t2->t3 -->
|
||||
<g id="edge3" class="edge"><title>t2->t3</title>
|
||||
<path fill="none" stroke="#0044a0" d="M415,-473.003C415,-469.312 415,-465.322 415,-461.352"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="418.5,-461.111 415,-451.111 411.5,-461.111 418.5,-461.111"/>
|
||||
</g>
|
||||
<!-- t6 -->
|
||||
<g id="node8" class="node"><title>t6</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="354.119,-335.109 39.8808,-335.109 35.8808,-331.109 35.8808,-299.109 350.119,-299.109 354.119,-303.109 354.119,-335.109"/>
|
||||
<polyline fill="none" stroke="black" points="350.119,-331.109 35.8808,-331.109 "/>
|
||||
<polyline fill="none" stroke="black" points="350.119,-331.109 350.119,-299.109 "/>
|
||||
<polyline fill="none" stroke="black" points="350.119,-331.109 354.119,-335.109 "/>
|
||||
<text text-anchor="middle" x="195" y="-311.109" font-family="Times,serif" font-size="20.00">select_image_for_pdf</text>
|
||||
</g>
|
||||
<!-- t2->t6 -->
|
||||
<g id="edge9" class="edge"><title>t2->t6</title>
|
||||
<path fill="none" stroke="#0044a0" d="M294.315,-473.06C280.461,-467.585 267.283,-460.434 256,-451.109 223.245,-424.039 207.251,-375.559 200.099,-345.207"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="203.486,-344.31 197.925,-335.292 196.648,-345.809 203.486,-344.31"/>
|
||||
</g>
|
||||
<!-- t4 -->
|
||||
<g id="node5" class="node"><title>t4</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="587.95,-393.109 310.05,-393.109 306.05,-389.109 306.05,-357.109 583.95,-357.109 587.95,-361.109 587.95,-393.109"/>
|
||||
<polyline fill="none" stroke="black" points="583.95,-389.109 306.05,-389.109 "/>
|
||||
<polyline fill="none" stroke="black" points="583.95,-389.109 583.95,-357.109 "/>
|
||||
<polyline fill="none" stroke="black" points="583.95,-389.109 587.95,-393.109 "/>
|
||||
<text text-anchor="middle" x="447" y="-369.109" font-family="Times,serif" font-size="20.00">preprocess_clean</text>
|
||||
</g>
|
||||
<!-- t3->t4 -->
|
||||
<g id="edge4" class="edge"><title>t3->t4</title>
|
||||
<path fill="none" stroke="#0044a0" d="M424.775,-415.003C427.132,-410.878 429.703,-406.379 432.232,-401.952"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="435.362,-403.53 437.285,-393.111 429.285,-400.057 435.362,-403.53"/>
|
||||
</g>
|
||||
<!-- t3->t6 -->
|
||||
<g id="edge8" class="edge"><title>t3->t6</title>
|
||||
<path fill="none" stroke="#0044a0" d="M351.041,-415.02C333.086,-409.151 313.871,-401.82 297,-393.109 269.828,-379.08 242.065,-358.149 222.377,-341.96"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="224.378,-339.071 214.46,-335.347 219.891,-344.444 224.378,-339.071"/>
|
||||
</g>
|
||||
<!-- t5 -->
|
||||
<g id="node6" class="node"><title>t5</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="665.666,-335.109 376.334,-335.109 372.334,-331.109 372.334,-299.109 661.666,-299.109 665.666,-303.109 665.666,-335.109"/>
|
||||
<polyline fill="none" stroke="black" points="661.666,-331.109 372.334,-331.109 "/>
|
||||
<polyline fill="none" stroke="black" points="661.666,-331.109 661.666,-299.109 "/>
|
||||
<polyline fill="none" stroke="black" points="661.666,-331.109 665.666,-335.109 "/>
|
||||
<text text-anchor="middle" x="519" y="-311.109" font-family="Times,serif" font-size="20.00">ocr_tesseract_hocr</text>
|
||||
</g>
|
||||
<!-- t4->t5 -->
|
||||
<g id="edge5" class="edge"><title>t4->t5</title>
|
||||
<path fill="none" stroke="#0044a0" d="M468.994,-357.003C475.274,-352.118 482.229,-346.709 488.905,-341.516"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="491.396,-344.013 497.141,-335.111 487.099,-338.487 491.396,-344.013"/>
|
||||
</g>
|
||||
<!-- t4->t6 -->
|
||||
<g id="edge7" class="edge"><title>t4->t6</title>
|
||||
<path fill="none" stroke="#0044a0" d="M370.365,-357.079C342.386,-350.862 310.55,-343.787 281.752,-337.388"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="282.225,-333.907 271.704,-335.155 280.707,-340.741 282.225,-333.907"/>
|
||||
</g>
|
||||
<!-- t4->t11 -->
|
||||
<g id="edge17" class="edge"><title>t4->t11</title>
|
||||
<path fill="none" stroke="#0044a0" d="M577.461,-357.079C627.38,-350.563 684.512,-343.106 735.342,-336.47"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="735.957,-339.92 745.42,-335.155 735.051,-332.979 735.957,-339.92"/>
|
||||
</g>
|
||||
<!-- t8 -->
|
||||
<g id="node7" class="node"><title>t8</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="986.109,-277.109 701.891,-277.109 697.891,-273.109 697.891,-241.109 982.109,-241.109 986.109,-245.109 986.109,-277.109"/>
|
||||
<polyline fill="none" stroke="black" points="982.109,-273.109 697.891,-273.109 "/>
|
||||
<polyline fill="none" stroke="black" points="982.109,-273.109 982.109,-241.109 "/>
|
||||
<polyline fill="none" stroke="black" points="982.109,-273.109 986.109,-277.109 "/>
|
||||
<text text-anchor="middle" x="842" y="-253.109" font-family="Times,serif" font-size="20.00">render_hocr_page</text>
|
||||
</g>
|
||||
<!-- t5->t8 -->
|
||||
<g id="edge6" class="edge"><title>t5->t8</title>
|
||||
<path fill="none" stroke="#0044a0" d="M617.226,-299.079C654.028,-292.699 696.036,-285.416 733.699,-278.886"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="734.429,-282.312 743.685,-277.155 733.234,-275.415 734.429,-282.312"/>
|
||||
</g>
|
||||
<!-- t9 -->
|
||||
<g id="node11" class="node"><title>t9</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="679.486,-277.109 336.514,-277.109 332.514,-273.109 332.514,-241.109 675.486,-241.109 679.486,-245.109 679.486,-277.109"/>
|
||||
<polyline fill="none" stroke="black" points="675.486,-273.109 332.514,-273.109 "/>
|
||||
<polyline fill="none" stroke="black" points="675.486,-273.109 675.486,-241.109 "/>
|
||||
<polyline fill="none" stroke="black" points="675.486,-273.109 679.486,-277.109 "/>
|
||||
<text text-anchor="middle" x="506" y="-253.109" font-family="Times,serif" font-size="20.00">render_hocr_debug_page</text>
|
||||
</g>
|
||||
<!-- t5->t9 -->
|
||||
<g id="edge15" class="edge"><title>t5->t9</title>
|
||||
<path fill="none" stroke="#0044a0" d="M515.029,-299.003C514.147,-295.204 513.191,-291.087 512.243,-287.002"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="515.617,-286.06 509.947,-277.111 508.799,-287.643 515.617,-286.06"/>
|
||||
</g>
|
||||
<!-- t10 -->
|
||||
<g id="node10" class="node"><title>t10</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="973.082,-219.109 714.918,-219.109 710.918,-215.109 710.918,-183.109 969.082,-183.109 973.082,-187.109 973.082,-219.109"/>
|
||||
<polyline fill="none" stroke="black" points="969.082,-215.109 710.918,-215.109 "/>
|
||||
<polyline fill="none" stroke="black" points="969.082,-215.109 969.082,-183.109 "/>
|
||||
<polyline fill="none" stroke="black" points="969.082,-215.109 973.082,-219.109 "/>
|
||||
<text text-anchor="middle" x="842" y="-195.109" font-family="Times,serif" font-size="20.00">add_text_layer</text>
|
||||
</g>
|
||||
<!-- t8->t10 -->
|
||||
<g id="edge12" class="edge"><title>t8->t10</title>
|
||||
<path fill="none" stroke="#0044a0" d="M842,-241.003C842,-237.312 842,-233.322 842,-229.352"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="845.5,-229.111 842,-219.111 838.5,-229.111 845.5,-229.111"/>
|
||||
</g>
|
||||
<!-- t6->t7 -->
|
||||
<g id="edge10" class="edge"><title>t6->t7</title>
|
||||
<path fill="none" stroke="#0044a0" d="M185.836,-299.003C183.626,-294.878 181.216,-290.379 178.845,-285.952"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="181.915,-284.273 174.108,-277.111 175.745,-287.578 181.915,-284.273"/>
|
||||
</g>
|
||||
<!-- t6->t9 -->
|
||||
<g id="edge14" class="edge"><title>t6->t9</title>
|
||||
<path fill="none" stroke="#0044a0" d="M289.577,-299.079C324.86,-292.726 365.115,-285.478 401.259,-278.969"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="402.116,-282.372 411.337,-277.155 400.875,-275.482 402.116,-282.372"/>
|
||||
</g>
|
||||
<!-- t7->t10 -->
|
||||
<g id="edge13" class="edge"><title>t7->t10</title>
|
||||
<path fill="none" stroke="#0044a0" d="M314.366,-242.002C317.606,-241.697 320.821,-241.399 324,-241.109 451.228,-229.521 596.262,-218.816 700.44,-211.565"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="700.976,-215.036 710.71,-210.852 700.491,-208.053 700.976,-215.036"/>
|
||||
</g>
|
||||
<!-- t14 -->
|
||||
<g id="node15" class="node"><title>t14</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="774.472,-105.333 939,-78.005 1103.53,-105.333 1103.37,-149.551 774.625,-149.551 774.472,-105.333"/>
|
||||
<polygon fill="none" stroke="black" points="770.46,-101.94 939,-73.9453 1107.54,-101.94 1107.36,-153.556 770.639,-153.556 770.46,-101.94"/>
|
||||
<text text-anchor="middle" x="939" y="-111.555" font-family="Times,serif" font-size="20.00">merge_pages</text>
|
||||
</g>
|
||||
<!-- t10->t14 -->
|
||||
<g id="edge23" class="edge"><title>t10->t14</title>
|
||||
<path fill="none" stroke="#0044a0" d="M862.571,-182.814C870.479,-176.165 879.888,-168.254 889.356,-160.293"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="891.89,-162.736 897.292,-153.622 887.385,-157.378 891.89,-162.736"/>
|
||||
</g>
|
||||
<!-- t9->t14 -->
|
||||
<g id="edge24" class="edge"><title>t9->t14</title>
|
||||
<path fill="none" stroke="#0044a0" d="M546.87,-241.103C586.226,-225.054 647.626,-200.869 702,-183.109 730.59,-173.771 761.415,-164.679 791.074,-156.405"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="792.38,-159.675 801.082,-153.632 790.511,-152.93 792.38,-159.675"/>
|
||||
</g>
|
||||
<!-- t13->t14 -->
|
||||
<g id="edge20" class="edge"><title>t13->t14</title>
|
||||
<path fill="none" stroke="#0044a0" d="M986.636,-414.956C1033.58,-398.748 1087,-369.103 1087,-318.109 1087,-318.109 1087,-318.109 1087,-258.109 1087,-215.948 1054.81,-182.761 1020.46,-159.346"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="1021.98,-156.158 1011.7,-153.604 1018.15,-162.013 1021.98,-156.158"/>
|
||||
</g>
|
||||
<!-- t11->t14 -->
|
||||
<g id="edge22" class="edge"><title>t11->t14</title>
|
||||
<path fill="none" stroke="#0044a0" d="M969.056,-299.033C979.177,-293.579 988.239,-286.44 995,-277.109 1020.03,-242.568 998.243,-195.803 974.929,-162.021"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="977.661,-159.825 968.996,-153.728 971.968,-163.897 977.661,-159.825"/>
|
||||
</g>
|
||||
<!-- t12->t14 -->
|
||||
<g id="edge21" class="edge"><title>t12->t14</title>
|
||||
<path fill="none" stroke="#0044a0" d="M1006.64,-472.956C1053.58,-456.748 1107,-427.103 1107,-376.109 1107,-376.109 1107,-376.109 1107,-258.109 1107,-220.742 1095.53,-209.426 1069,-183.109 1059.8,-173.988 1049.06,-165.927 1037.76,-158.869"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="1039.41,-155.777 1029.03,-153.665 1035.83,-161.791 1039.41,-155.777"/>
|
||||
</g>
|
||||
<!-- t15 -->
|
||||
<g id="node16" class="node"><title>t15</title>
|
||||
<polygon fill="#efa03b" stroke="black" points="1053.18,-52 828.822,-52 824.822,-48 824.822,-16 1049.18,-16 1053.18,-20 1053.18,-52"/>
|
||||
<polyline fill="none" stroke="black" points="1049.18,-48 824.822,-48 "/>
|
||||
<polyline fill="none" stroke="black" points="1049.18,-48 1049.18,-16 "/>
|
||||
<polyline fill="none" stroke="black" points="1049.18,-48 1053.18,-52 "/>
|
||||
<text text-anchor="middle" x="939" y="-28" font-family="Times,serif" font-size="20.00">copy_final</text>
|
||||
</g>
|
||||
<!-- t14->t15 -->
|
||||
<g id="edge25" class="edge"><title>t14->t15</title>
|
||||
<path fill="none" stroke="#0044a0" d="M939,-73.8665C939,-69.8921 939,-65.942 939,-62.1676"/>
|
||||
<polygon fill="#0044a0" stroke="#0044a0" points="942.5,-62.1213 939,-52.1214 935.5,-62.1214 942.5,-62.1213"/>
|
||||
</g>
|
||||
</g>
|
||||
</svg>
|
||||
|
After Width: | Height: | Size: 18 KiB |
@@ -0,0 +1,5 @@
|
||||
ruffus>=2.6.3
|
||||
Pillow>=2.4.0
|
||||
reportlab>=3.1.44
|
||||
PyPDF2>=1.25.1
|
||||
git+https://github.com/jbarlow83/img2pdf.git@e9bcce0afc3720752ca53a991db93f911a1df709#egg=img2pdf-0.1.5.dev
|
||||
@@ -0,0 +1,226 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
from __future__ import print_function, unicode_literals
|
||||
from setuptools import setup
|
||||
from subprocess import STDOUT, check_output, CalledProcessError
|
||||
from collections.abc import Mapping
|
||||
import re
|
||||
import sys
|
||||
|
||||
|
||||
if sys.version_info < (3, 4):
|
||||
print("Python 3.4 or newer is required")
|
||||
sys.exit(1)
|
||||
|
||||
missing_program = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH.
|
||||
'''
|
||||
|
||||
unknown_version = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher. Your system has
|
||||
'{program}' but we cannot tell what version is installed. Contact the
|
||||
package maintainer.
|
||||
'''
|
||||
|
||||
old_version = '''
|
||||
OCRmyPDF requires '{program}' {need_version} or higher. Your system appears
|
||||
to have {found_version}. Please update this program.
|
||||
'''
|
||||
|
||||
okay_its_optional = '''
|
||||
This program is OPTIONAL, so installation of OCRmyPDF can proceed, but
|
||||
some functionality may be missing.
|
||||
'''
|
||||
|
||||
not_okay_its_required = '''
|
||||
This program is REQUIRED for OCRmyPDF to work. Installation will abort.
|
||||
'''
|
||||
|
||||
osx_install_advice = '''
|
||||
If you have homebrew installed, try these command to install the missing
|
||||
packages:
|
||||
brew update
|
||||
brew upgrade
|
||||
brew install {package}
|
||||
'''
|
||||
|
||||
linux_install_advice = '''
|
||||
On systems with the aptitude package manager (Debian, Ubuntu), try these
|
||||
commands:
|
||||
sudo apt-get update
|
||||
sudo apt-get install {package}
|
||||
|
||||
On RPM-based systems (Red Hat, Fedora), search for instructions on
|
||||
installing the RPM for {program}.
|
||||
'''
|
||||
|
||||
|
||||
def get_platform():
|
||||
if sys.platform.startswith('freebsd'):
|
||||
return 'freebsd'
|
||||
elif sys.platform.startswith('linux'):
|
||||
return 'linux'
|
||||
return sys.platform
|
||||
|
||||
|
||||
def _error_trailer(program, package, optional, **kwargs):
|
||||
if optional:
|
||||
print(okay_its_optional.format(**locals()), file=sys.stderr)
|
||||
else:
|
||||
print(not_okay_its_required.format(**locals()), file=sys.stderr)
|
||||
|
||||
if isinstance(package, Mapping):
|
||||
package = package[get_platform()]
|
||||
|
||||
if get_platform() == 'darwin':
|
||||
print(osx_install_advice.format(**locals()), file=sys.stderr)
|
||||
elif get_platform() == 'linux':
|
||||
print(linux_install_advice.format(**locals()), file=sys.stderr)
|
||||
|
||||
|
||||
def error_missing_program(
|
||||
program,
|
||||
package,
|
||||
optional
|
||||
):
|
||||
print(missing_program.format(**locals()), file=sys.stderr)
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def error_unknown_version(
|
||||
program,
|
||||
package,
|
||||
optional,
|
||||
need_version
|
||||
):
|
||||
print(unknown_version.format(**locals()), file=sys.stderr)
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def error_old_version(
|
||||
program,
|
||||
package,
|
||||
optional,
|
||||
need_version,
|
||||
found_version
|
||||
):
|
||||
print(old_version.format(**locals()), file=sys.stderr)
|
||||
_error_trailer(**locals())
|
||||
|
||||
|
||||
def check_external_program(
|
||||
program,
|
||||
need_version,
|
||||
package,
|
||||
version_check_args=['--version'],
|
||||
version_scrape_regex=re.compile(r'(\d+\.\d+(?:\.\d+)?)'),
|
||||
optional=False):
|
||||
|
||||
print('Checking for {program} >= {need_version}...'.format(
|
||||
program=program, need_version=need_version))
|
||||
try:
|
||||
result = check_output(
|
||||
[program] + version_check_args,
|
||||
universal_newlines=True, stderr=STDOUT)
|
||||
except (CalledProcessError, FileNotFoundError):
|
||||
error_missing_program(program, package, optional)
|
||||
if not optional:
|
||||
sys.exit(1)
|
||||
print('Continuing install without {program}'.format(program=program))
|
||||
return
|
||||
|
||||
try:
|
||||
found_version = version_scrape_regex.search(result).group(1)
|
||||
except AttributeError:
|
||||
error_unknown_version(program, package, optional, need_version)
|
||||
sys.exit(1)
|
||||
|
||||
if found_version < need_version:
|
||||
error_old_version(program, package, optional, need_version,
|
||||
found_version)
|
||||
|
||||
print('Found {program} {found_version}'.format(
|
||||
program=program, found_version=found_version))
|
||||
|
||||
|
||||
command = next((arg for arg in sys.argv[1:] if not arg.startswith('-')), '')
|
||||
|
||||
|
||||
if command.startswith('install') or \
|
||||
command in ['check', 'test', 'nosetests', 'easy_install', 'egg_info']:
|
||||
check_external_program(
|
||||
program='tesseract',
|
||||
need_version='3.02.02',
|
||||
package={'darwin': 'tesseract', 'linux': 'tesseract-ocr'}
|
||||
)
|
||||
check_external_program(
|
||||
program='gs',
|
||||
need_version='9.14',
|
||||
package='ghostscript'
|
||||
)
|
||||
check_external_program(
|
||||
program='unpaper',
|
||||
need_version='6.1',
|
||||
package='unpaper',
|
||||
optional=True
|
||||
)
|
||||
check_external_program(
|
||||
program='qpdf',
|
||||
need_version='5.0.0',
|
||||
package='qpdf',
|
||||
version_check_args=['--version']
|
||||
)
|
||||
|
||||
|
||||
if 'upload' in sys.argv[1:]:
|
||||
print('Use twine to upload the package - setup.py upload is insecure')
|
||||
sys.exit(1)
|
||||
|
||||
tests_require = open('test_requirements.txt').read().splitlines()
|
||||
|
||||
setup(
|
||||
name='ocrmypdf',
|
||||
description='OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched',
|
||||
url='https://github.com/jbarlow83/OCRmyPDF',
|
||||
author='James R. Barlow',
|
||||
author_email='jim@purplerock.ca',
|
||||
license='Public Domain',
|
||||
packages=['ocrmypdf'],
|
||||
keywords=['PDF', 'OCR', 'optical character recognition', 'PDF/A', 'scanning'],
|
||||
classifiers=[
|
||||
"Programming Language :: Python :: 3",
|
||||
"Development Status :: 5 - Production/Stable",
|
||||
"Environment :: Console",
|
||||
"Intended Audience :: End Users/Desktop",
|
||||
"Intended Audience :: Science/Research",
|
||||
"Intended Audience :: System Administrators",
|
||||
"License :: Public Domain",
|
||||
"Operating System :: MacOS :: MacOS X",
|
||||
"Operating System :: POSIX",
|
||||
"Operating System :: POSIX :: BSD",
|
||||
"Operating System :: POSIX :: Linux",
|
||||
"Topic :: Scientific/Engineering :: Image Recognition",
|
||||
"Topic :: Text Processing :: Indexing",
|
||||
"Topic :: Text Processing :: Linguistic",
|
||||
],
|
||||
setup_requires=[
|
||||
'setuptools_scm'
|
||||
],
|
||||
use_scm_version={'version_scheme': 'post-release'},
|
||||
install_requires=[
|
||||
'ruffus',
|
||||
'Pillow',
|
||||
'reportlab',
|
||||
'PyPDF2',
|
||||
'img2pdf'
|
||||
],
|
||||
tests_require=tests_require,
|
||||
entry_points={
|
||||
'console_scripts': [
|
||||
'ocrmypdf = ocrmypdf.main:run_pipeline'
|
||||
],
|
||||
},
|
||||
include_package_data=True,
|
||||
zip_safe=False)
|
||||
@@ -1,53 +0,0 @@
|
||||
#! /bin/bash
|
||||
|
||||
set -x
|
||||
set -e
|
||||
|
||||
chmod +x OCRmyPDF*.AppImage
|
||||
|
||||
# run OCRmyPDF to test if the AppImage can ocr a test file
|
||||
run_appimage()
|
||||
{
|
||||
echo ""
|
||||
./OCRmyPDF*.AppImage --help
|
||||
echo ""
|
||||
./OCRmyPDF*.AppImage --list-programs
|
||||
echo ""
|
||||
./OCRmyPDF*.AppImage --list-licenses
|
||||
echo ""
|
||||
./OCRmyPDF*.AppImage ocrmypdf -l deu -s -d --jbig2-lossy --optimize 1 "$TRAVIS_BUILD_DIR"/test/test.pdf output.pdf
|
||||
echo ""
|
||||
}
|
||||
|
||||
|
||||
# check AppImage for common issues
|
||||
run_appimagelint()
|
||||
{
|
||||
wget https://github.com/TheAssassin/appimagelint/releases/download/continuous/appimagelint-x86_64.AppImage
|
||||
chmod +x appimagelint-x86_64.AppImage
|
||||
./appimagelint-x86_64.AppImage OCRmyPDF*.AppImage
|
||||
}
|
||||
|
||||
|
||||
# extract the OCRmyPDF AppImage, install pytest & test requirements and run pytest
|
||||
run_pytest()
|
||||
{
|
||||
git clone --depth=1 --branch "v$OCRMYPDF_VERSION" https://github.com/jbarlow83/OCRmyPDF.git
|
||||
./OCRmyPDF*.AppImage --appimage-extract
|
||||
|
||||
pushd squashfs-root
|
||||
./AppRun python3 -m pip install pytest
|
||||
./AppRun python3 -m pip install -r ../OCRmyPDF/requirements/test.txt
|
||||
./AppRun python3 -m pytest ../OCRmyPDF -n auto
|
||||
popd
|
||||
}
|
||||
|
||||
|
||||
run_appimage
|
||||
|
||||
run_appimagelint
|
||||
|
||||
# run_pytest
|
||||
|
||||
|
||||
|
||||
Binary file not shown.
@@ -0,0 +1 @@
|
||||
pytest>=2.7.2
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 1.4 MiB |
@@ -0,0 +1,34 @@
|
||||
All test resources must come from free public domain sources for
|
||||
copyright reasons.
|
||||
|
||||
Test files do not necessarily produce perfect (or even good) OCR
|
||||
results.
|
||||
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| File | Source |
|
||||
+=====================+================================================================================+
|
||||
| graph.pdf | Wikimedia |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| c02-22.pdf | Project Gutenberg: https://www.gutenberg.org/files/76/76-h/images/c02-22.jpg |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| LinnSequencer.jpg | Wikimedia_ |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| congress.jpg | http://www.baxleystamps.com/litho/meiji/courts_1871.jpg |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| blank.pdf | Blank page from Adobe Illustrator CC 2015 |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| enormous.pdf | PNG file saved to PDF using img2pdf |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| invalid.pdf | PDF file header followed by EOF marker; not valid |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| multipage.pdf | several other files concatenated |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| skew.pdf | skewed version of c02-22.PDF |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| Test_Issue_28.pdf | file with some syntax errors |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
| missing_docinfo.pdf | file missing its DocumentInfo dictionary |
|
||||
+---------------------+--------------------------------------------------------------------------------+
|
||||
|
||||
|
||||
.. _Wikimedia: https://upload.wikimedia.org/wikipedia/en/b/b7/LinnSequencer_hardware_MIDI_sequencer_brochure_page_2_300dpi.jpg
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because it is too large
Load Diff
Binary file not shown.
Binary file not shown.
File diff suppressed because one or more lines are too long
Binary file not shown.
|
After Width: | Height: | Size: 188 KiB |
Binary file not shown.
File diff suppressed because one or more lines are too long
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,3 @@
|
||||
%PDF-1.3
|
||||
This is not a valid PDF file
|
||||
%%EOF
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Executable
+11
@@ -0,0 +1,11 @@
|
||||
#!/usr/bin/env python3
|
||||
import sys
|
||||
|
||||
|
||||
def main():
|
||||
print('qpdf dummy')
|
||||
sys.exit(2)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Executable
+78
@@ -0,0 +1,78 @@
|
||||
#!/usr/bin/env python3
|
||||
import sys
|
||||
import os
|
||||
import hashlib
|
||||
import shutil
|
||||
import subprocess
|
||||
|
||||
|
||||
CACHE_PATH = os.path.abspath(os.path.join(
|
||||
os.path.dirname(__file__), '..', 'cache'))
|
||||
|
||||
|
||||
def main():
|
||||
operation = sys.argv[-1]
|
||||
# For anything except a hocr or pdf, defer to real tesseract
|
||||
if operation != 'hocr' and operation != 'pdf':
|
||||
tess_args = ['tesseract'] + sys.argv[1:]
|
||||
os.execvp("tesseract", tess_args)
|
||||
return # Not reachable
|
||||
|
||||
try:
|
||||
os.makedirs(CACHE_PATH)
|
||||
except FileExistsError:
|
||||
pass
|
||||
|
||||
m = hashlib.sha1()
|
||||
|
||||
version = subprocess.check_output(
|
||||
['tesseract', '--version'],
|
||||
stderr=subprocess.STDOUT)
|
||||
|
||||
m.update(version)
|
||||
m.update(operation.encode())
|
||||
|
||||
try:
|
||||
lang = sys.argv[sys.argv.index('-l') + 1]
|
||||
m.update(lang.encode())
|
||||
except ValueError:
|
||||
pass
|
||||
try:
|
||||
psm = sys.argv[sys.argv.index('-psm') + 1]
|
||||
m.update(psm.encode())
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
input_file = sys.argv[-3]
|
||||
output_file = sys.argv[-2]
|
||||
|
||||
if operation == 'hocr':
|
||||
output_file += '.hocr'
|
||||
elif operation == 'pdf':
|
||||
output_file += '.pdf'
|
||||
|
||||
with open(input_file, 'rb') as f:
|
||||
m.update(f.read())
|
||||
|
||||
cache_name = os.path.join(CACHE_PATH, m.hexdigest())
|
||||
if os.path.exists(cache_name):
|
||||
# Cache hit
|
||||
print("Tesseract cache hit", file=sys.stderr)
|
||||
shutil.copy(cache_name, output_file)
|
||||
sys.exit(0)
|
||||
|
||||
# Cache miss
|
||||
print("Tesseract cache miss", file=sys.stderr)
|
||||
|
||||
# Call tesseract
|
||||
subprocess.check_call(['tesseract'] + sys.argv[1:])
|
||||
|
||||
# Insert file into cache
|
||||
if os.path.exists(output_file):
|
||||
shutil.copy(output_file, cache_name)
|
||||
else:
|
||||
print("Could not find output file", file=sys.stderr)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
Executable
+65
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env python3
|
||||
import sys
|
||||
import img2pdf
|
||||
from PIL import Image
|
||||
|
||||
|
||||
VERSION_STRING = '''tesseract 3.04.00
|
||||
leptonica-1.72
|
||||
libjpeg 8d : libpng 1.6.19 : libtiff 4.0.6 : zlib 1.2.5
|
||||
SPOOFED
|
||||
'''
|
||||
|
||||
HOCR_TEMPLATE = '''<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||
<head>
|
||||
<title></title>
|
||||
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 3.02.02' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word'/>
|
||||
</head>
|
||||
<body>
|
||||
<div class='ocr_page' id='page_1' title='image "x.tif"; bbox 0 0 {0} {1}; ppageno 0'>
|
||||
<div class='ocr_carea' id='block_1_1' title="bbox 0 1 {0} {1}">
|
||||
<p class='ocr_par' dir='ltr' id='par_1' title="bbox 0 1 {0} {1}">
|
||||
<span class='ocr_line' id='line_1' title="bbox 0 1 {0} {1}"><span class='ocrx_word' id='word_1' title="bbox 0 1 {0} {1}"> </span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>'''
|
||||
|
||||
|
||||
def main():
|
||||
if sys.argv[1] == '--version':
|
||||
print(VERSION_STRING, file=sys.stderr)
|
||||
sys.exit(0)
|
||||
elif sys.argv[1] == '--list-langs':
|
||||
print('List of available languages (1):\neng', file=sys.stderr)
|
||||
sys.exit(0)
|
||||
elif sys.argv[-1] == 'hocr':
|
||||
inputf = sys.argv[-3]
|
||||
output = sys.argv[-2]
|
||||
with Image.open(inputf) as im, \
|
||||
open(output + '.hocr', 'w', encoding='utf-8') as f:
|
||||
w, h = im.size
|
||||
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
|
||||
elif sys.argv[-1] == 'pdf':
|
||||
inputf = sys.argv[-3]
|
||||
output = sys.argv[-2]
|
||||
pdf_bytes = img2pdf.convert([inputf], dpi=300)
|
||||
with open(output + '.pdf', 'wb') as f:
|
||||
f.write(pdf_bytes)
|
||||
else:
|
||||
print("Spoof doesn't understand arguments", file=sys.stderr)
|
||||
print(sys.argv, file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
sys.exit(0)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -0,0 +1,61 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
from ocrmypdf import hocrtransform
|
||||
from ocrmypdf.tesseract import HOCR_TEMPLATE
|
||||
from reportlab.pdfgen.canvas import Canvas
|
||||
from PIL import Image
|
||||
from tempfile import NamedTemporaryFile
|
||||
from contextlib import suppress
|
||||
import os
|
||||
import shutil
|
||||
import pytest
|
||||
import img2pdf
|
||||
import pytest
|
||||
import sys
|
||||
|
||||
|
||||
if sys.version_info.major < 3:
|
||||
print("Requires Python 3.4+")
|
||||
sys.exit(1)
|
||||
|
||||
TESTS_ROOT = os.path.abspath(os.path.dirname(__file__))
|
||||
SPOOF_PATH = os.path.join(TESTS_ROOT, 'spoof')
|
||||
PROJECT_ROOT = os.path.dirname(TESTS_ROOT)
|
||||
OCRMYPDF = os.path.join(PROJECT_ROOT, 'OCRmyPDF.sh')
|
||||
TEST_RESOURCES = os.path.join(PROJECT_ROOT, 'tests', 'resources')
|
||||
TEST_OUTPUT = os.environ.get(
|
||||
'OCRMYPDF_TEST_OUTPUT',
|
||||
default=os.path.join(PROJECT_ROOT, 'tests', 'output', 'hocrtransform'))
|
||||
|
||||
|
||||
def setup_module():
|
||||
with suppress(FileNotFoundError):
|
||||
shutil.rmtree(TEST_OUTPUT)
|
||||
with suppress(FileExistsError):
|
||||
os.makedirs(TEST_OUTPUT)
|
||||
with open(_make_output('blank.hocr'), 'w') as f:
|
||||
f.write(HOCR_TEMPLATE)
|
||||
|
||||
|
||||
def _make_input(input_basename):
|
||||
return os.path.join(TEST_RESOURCES, input_basename)
|
||||
|
||||
|
||||
def _make_output(output_basename):
|
||||
return os.path.join(TEST_OUTPUT, output_basename)
|
||||
|
||||
|
||||
def test_mono_image():
|
||||
im = Image.new('1', (8, 8), 0)
|
||||
for n in range(8):
|
||||
im.putpixel((n, n), 1)
|
||||
im.save(_make_output('mono.tif'), format='TIFF')
|
||||
|
||||
hocr = hocrtransform.HocrTransform(_make_output('blank.hocr'), 300)
|
||||
hocr.to_pdf(_make_output('mono.pdf'), imageFileName=_make_output('mono.tif'))
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,369 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
from __future__ import print_function
|
||||
from subprocess import Popen, PIPE, check_output, check_call
|
||||
import os
|
||||
import shutil
|
||||
from contextlib import suppress
|
||||
import sys
|
||||
import pytest
|
||||
from ocrmypdf.pageinfo import pdf_get_all_pageinfo
|
||||
import PyPDF2 as pypdf
|
||||
from ocrmypdf import ExitCode
|
||||
|
||||
|
||||
if sys.version_info.major < 3:
|
||||
print("Requires Python 3.4+")
|
||||
sys.exit(1)
|
||||
|
||||
TESTS_ROOT = os.path.abspath(os.path.dirname(__file__))
|
||||
SPOOF_PATH = os.path.join(TESTS_ROOT, 'spoof')
|
||||
PROJECT_ROOT = os.path.dirname(TESTS_ROOT)
|
||||
OCRMYPDF = os.path.join(PROJECT_ROOT, 'OCRmyPDF.sh')
|
||||
TEST_RESOURCES = os.path.join(PROJECT_ROOT, 'tests', 'resources')
|
||||
TEST_OUTPUT = os.environ.get(
|
||||
'OCRMYPDF_TEST_OUTPUT',
|
||||
default=os.path.join(PROJECT_ROOT, 'tests', 'output', 'main'))
|
||||
|
||||
|
||||
def setup_module():
|
||||
with suppress(FileNotFoundError):
|
||||
shutil.rmtree(TEST_OUTPUT)
|
||||
with suppress(FileExistsError):
|
||||
os.makedirs(TEST_OUTPUT)
|
||||
|
||||
|
||||
def run_ocrmypdf_sh(input_file, output_file, *args, env=None):
|
||||
sh_args = ['sh', OCRMYPDF] + list(args) + [input_file, output_file]
|
||||
sh = Popen(
|
||||
sh_args, close_fds=True, stdout=PIPE, stderr=PIPE,
|
||||
universal_newlines=True, env=env)
|
||||
out, err = sh.communicate()
|
||||
return sh, out, err
|
||||
|
||||
|
||||
def _make_input(input_basename):
|
||||
return os.path.join(TEST_RESOURCES, input_basename)
|
||||
|
||||
|
||||
def _make_output(output_basename):
|
||||
return os.path.join(TEST_OUTPUT, output_basename)
|
||||
|
||||
|
||||
def check_ocrmypdf(input_basename, output_basename, *args, env=None):
|
||||
input_file = _make_input(input_basename)
|
||||
output_file = _make_output(output_basename)
|
||||
|
||||
sh, out, err = run_ocrmypdf_sh(input_file, output_file, *args, env=env)
|
||||
assert sh.returncode == 0, dict(stdout=out, stderr=err)
|
||||
assert os.path.exists(output_file), "Output file not created"
|
||||
assert os.stat(output_file).st_size > 100, "PDF too small or empty"
|
||||
return output_file
|
||||
|
||||
|
||||
def run_ocrmypdf_env(input_basename, output_basename, *args, env=None):
|
||||
input_file = _make_input(input_basename)
|
||||
output_file = _make_output(output_basename)
|
||||
|
||||
if env is None:
|
||||
env = os.environ
|
||||
|
||||
p_args = ['ocrmypdf'] + list(args) + [input_file, output_file]
|
||||
p = Popen(
|
||||
p_args, close_fds=True, stdout=PIPE, stderr=PIPE,
|
||||
universal_newlines=True, env=env)
|
||||
out, err = p.communicate()
|
||||
return p, out, err
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def spoof_tesseract_noop():
|
||||
env = os.environ.copy()
|
||||
program = os.path.join(SPOOF_PATH, 'tesseract_noop.py')
|
||||
check_call(['chmod', "+x", program])
|
||||
env['OCRMYPDF_TESSERACT'] = program
|
||||
return env
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def spoof_tesseract_cache():
|
||||
env = os.environ.copy()
|
||||
program = os.path.join(SPOOF_PATH, "tesseract_cache.py")
|
||||
check_call(['chmod', '+x', program])
|
||||
env['OCRMYPDF_TESSERACT'] = program
|
||||
return env
|
||||
|
||||
|
||||
def test_quick(spoof_tesseract_noop):
|
||||
check_ocrmypdf('c02-22.pdf', 'test_quick.pdf', env=spoof_tesseract_noop)
|
||||
|
||||
|
||||
def test_deskew(spoof_tesseract_noop):
|
||||
# Run with deskew
|
||||
deskewed_pdf = check_ocrmypdf(
|
||||
'skew.pdf', 'test_deskew.pdf', '-d', env=spoof_tesseract_noop)
|
||||
|
||||
# Now render as an image again and use Leptonica to find the skew angle
|
||||
# to confirm that it was deskewed
|
||||
from ocrmypdf.ghostscript import rasterize_pdf
|
||||
import logging
|
||||
log = logging.getLogger()
|
||||
|
||||
deskewed_png = _make_output('deskewed.png')
|
||||
|
||||
rasterize_pdf(
|
||||
deskewed_pdf,
|
||||
deskewed_png,
|
||||
xres=150,
|
||||
yres=150,
|
||||
raster_device='pngmono',
|
||||
log=log)
|
||||
|
||||
from ocrmypdf.leptonica import pixRead, pixDestroy, pixFindSkew
|
||||
pix = pixRead(deskewed_png)
|
||||
skew_angle, skew_confidence = pixFindSkew(pix)
|
||||
pix = pixDestroy(pix)
|
||||
|
||||
print(skew_angle)
|
||||
assert -0.5 < skew_angle < 0.5, "Deskewing failed"
|
||||
|
||||
|
||||
def test_clean(spoof_tesseract_noop):
|
||||
check_ocrmypdf('skew.pdf', 'test_clean.pdf', '-c', env=spoof_tesseract_noop)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("pdf,renderer", [
|
||||
('palette.pdf', 'hocr'),
|
||||
('palette.pdf', 'tesseract'),
|
||||
('cmyk.pdf', 'hocr'),
|
||||
('cmyk.pdf', 'tesseract'),
|
||||
('ccitt.pdf', 'hocr'),
|
||||
('ccitt.pdf', 'tesseract'),
|
||||
('jbig2.pdf', 'hocr'),
|
||||
('jbig2.pdf', 'tesseract')
|
||||
])
|
||||
def test_exotic_image(spoof_tesseract_cache, pdf, renderer):
|
||||
check_ocrmypdf(
|
||||
pdf,
|
||||
'test_{0}_{1}.pdf'.format(pdf, renderer),
|
||||
'-dc',
|
||||
'-v', '1',
|
||||
'--pdf-renderer', renderer, env=spoof_tesseract_cache)
|
||||
|
||||
|
||||
def test_preserve_metadata(spoof_tesseract_noop):
|
||||
pdf_before = pypdf.PdfFileReader(_make_input('graph.pdf'))
|
||||
|
||||
output = check_ocrmypdf('graph.pdf', 'test_metadata_preserve.pdf',
|
||||
env=spoof_tesseract_noop)
|
||||
|
||||
pdf_after = pypdf.PdfFileReader(output)
|
||||
|
||||
for key in ('/Title', '/Author'):
|
||||
assert pdf_before.documentInfo[key] == pdf_after.documentInfo[key]
|
||||
|
||||
|
||||
def test_override_metadata(spoof_tesseract_noop):
|
||||
input_file = _make_input('c02-22.pdf')
|
||||
output_file = _make_output('test_override_metadata.pdf')
|
||||
|
||||
german = 'Du siehst den Wald vor lauter Bäumen nicht.'
|
||||
chinese = '孔子'
|
||||
high_unicode = 'U+1030C is: 𐌌'
|
||||
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
input_file, output_file,
|
||||
'--title', german,
|
||||
'--author', chinese,
|
||||
'--subject', high_unicode,
|
||||
env=spoof_tesseract_noop)
|
||||
|
||||
assert p.returncode == ExitCode.ok
|
||||
|
||||
pdf = output_file
|
||||
|
||||
out_pdfinfo = check_output(['pdfinfo', pdf], universal_newlines=True)
|
||||
lines_pdfinfo = out_pdfinfo.splitlines()
|
||||
pdfinfo = {}
|
||||
for line in lines_pdfinfo:
|
||||
k, v = line.strip().split(':', maxsplit=1)
|
||||
pdfinfo[k.strip()] = v.strip()
|
||||
|
||||
assert pdfinfo['Title'] == german
|
||||
assert pdfinfo['Author'] == chinese
|
||||
assert pdfinfo['Subject'] == high_unicode
|
||||
assert pdfinfo.get('Keywords', '') == ''
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', [
|
||||
'hocr',
|
||||
'tesseract',
|
||||
])
|
||||
def test_oversample(spoof_tesseract_cache, renderer):
|
||||
oversampled_pdf = check_ocrmypdf(
|
||||
'skew.pdf', 'test_oversample_%s.pdf' % renderer, '--oversample', '300',
|
||||
'-f',
|
||||
'--pdf-renderer', renderer, env=spoof_tesseract_cache)
|
||||
|
||||
pdfinfo = pdf_get_all_pageinfo(oversampled_pdf)
|
||||
|
||||
print(pdfinfo[0]['xres'])
|
||||
assert abs(pdfinfo[0]['xres'] - 300) < 1
|
||||
|
||||
|
||||
def test_repeat_ocr():
|
||||
sh, _, _ = run_ocrmypdf_sh('graph_ocred.pdf', 'wontwork.pdf')
|
||||
assert sh.returncode != 0
|
||||
|
||||
|
||||
def test_force_ocr(spoof_tesseract_cache):
|
||||
out = check_ocrmypdf('graph_ocred.pdf', 'test_force.pdf', '-f',
|
||||
env=spoof_tesseract_cache)
|
||||
pdfinfo = pdf_get_all_pageinfo(out)
|
||||
assert pdfinfo[0]['has_text']
|
||||
|
||||
|
||||
def test_skip_ocr(spoof_tesseract_cache):
|
||||
check_ocrmypdf('graph_ocred.pdf', 'test_skip.pdf', '-s',
|
||||
env=spoof_tesseract_cache)
|
||||
|
||||
|
||||
def test_argsfile(spoof_tesseract_noop):
|
||||
with open(_make_output('test_argsfile.txt'), 'w') as argsfile:
|
||||
print('--title', 'ArgsFile Test', '--author', 'Test Cases',
|
||||
sep='\n', end='\n', file=argsfile)
|
||||
check_ocrmypdf('graph.pdf', 'test_argsfile.pdf',
|
||||
'@' + _make_output('test_argsfile.txt'),
|
||||
env=spoof_tesseract_noop)
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', [
|
||||
'hocr',
|
||||
'tesseract',
|
||||
])
|
||||
def test_ocr_timeout(renderer):
|
||||
out = check_ocrmypdf('skew.pdf', 'test_timeout_%s.pdf' % renderer,
|
||||
'--tesseract-timeout', '1.0')
|
||||
pdfinfo = pdf_get_all_pageinfo(out)
|
||||
assert pdfinfo[0]['has_text'] == False
|
||||
|
||||
|
||||
def test_skip_big(spoof_tesseract_cache):
|
||||
out = check_ocrmypdf('enormous.pdf', 'test_enormous.pdf',
|
||||
'--skip-big', '10', env=spoof_tesseract_cache)
|
||||
pdfinfo = pdf_get_all_pageinfo(out)
|
||||
assert pdfinfo[0]['has_text'] == False
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', [
|
||||
'hocr',
|
||||
'tesseract',
|
||||
])
|
||||
def test_maximum_options(spoof_tesseract_cache, renderer):
|
||||
check_ocrmypdf(
|
||||
'multipage.pdf', 'test_multipage%s.pdf' % renderer,
|
||||
'-d', '-c', '-i', '-g', '-f', '-k', '--oversample', '300',
|
||||
'--skip-big', '10', '--title', 'Too Many Weird Files',
|
||||
'--author', 'py.test', '--pdf-renderer', renderer,
|
||||
env=spoof_tesseract_cache)
|
||||
|
||||
|
||||
def test_tesseract_missing_tessdata():
|
||||
env = os.environ.copy()
|
||||
env['TESSDATA_PREFIX'] = '/tmp'
|
||||
|
||||
p, _, err = run_ocrmypdf_env(
|
||||
'graph_ocred.pdf', 'not_a_pdfa.pdf', '-v', '1', '--skip-text', env=env)
|
||||
assert p.returncode == ExitCode.missing_dependency, err
|
||||
|
||||
|
||||
def test_invalid_input_pdf():
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
'invalid.pdf', 'wont_be_created.pdf')
|
||||
assert p.returncode == ExitCode.input_file, err
|
||||
|
||||
|
||||
def test_blank_input_pdf():
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
'blank.pdf', 'still_blank.pdf')
|
||||
assert p.returncode == ExitCode.ok
|
||||
|
||||
|
||||
def test_french(spoof_tesseract_cache):
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
'francais.pdf', 'francais.pdf', '-l', 'fra', env=spoof_tesseract_cache)
|
||||
assert p.returncode == ExitCode.ok, \
|
||||
"This test may fail if Tesseract language packs are missing"
|
||||
|
||||
|
||||
def test_klingon():
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
'francais.pdf', 'francais.pdf', '-l', 'klz')
|
||||
assert p.returncode == ExitCode.bad_args
|
||||
|
||||
|
||||
def test_missing_docinfo(spoof_tesseract_noop):
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
'missing_docinfo.pdf', 'missing_docinfo.pdf', '-l', 'eng', '-c',
|
||||
env=spoof_tesseract_noop)
|
||||
assert p.returncode == ExitCode.ok, err
|
||||
|
||||
|
||||
def test_uppercase_extension(spoof_tesseract_noop):
|
||||
shutil.copy(_make_input("skew.pdf"), _make_input("UPPERCASE.PDF"))
|
||||
try:
|
||||
check_ocrmypdf("UPPERCASE.PDF", "UPPERCASE_OUT.PDF",
|
||||
env=spoof_tesseract_noop)
|
||||
finally:
|
||||
os.unlink(_make_input("UPPERCASE.PDF"))
|
||||
|
||||
|
||||
def test_input_file_not_found():
|
||||
input_file = "does not exist.pdf"
|
||||
sh, out, err = run_ocrmypdf_sh(
|
||||
_make_input(input_file),
|
||||
_make_output("will not happen.pdf"))
|
||||
assert sh.returncode == ExitCode.input_file
|
||||
assert (input_file in out or input_file in err)
|
||||
|
||||
|
||||
def test_input_file_not_a_pdf():
|
||||
input_file = __file__ # Try to OCR this file
|
||||
sh, out, err = run_ocrmypdf_sh(
|
||||
_make_input(input_file),
|
||||
_make_output("will not happen.pdf"))
|
||||
assert sh.returncode == ExitCode.input_file
|
||||
assert (input_file in out or input_file in err)
|
||||
|
||||
|
||||
def test_qpdf_repair_fails():
|
||||
env = os.environ.copy()
|
||||
env['OCRMYPDF_QPDF'] = os.path.abspath('./spoof/qpdf_dummy_return2.py')
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
'-v', '1',
|
||||
'c02-22.pdf', 'wont_be_created.pdf', env=env)
|
||||
print(out)
|
||||
print(err)
|
||||
assert p.returncode == ExitCode.input_file
|
||||
|
||||
|
||||
def test_encrypted():
|
||||
p, out, err = run_ocrmypdf_env('skew-encrypted.pdf', 'wont_be_created.pdf')
|
||||
assert p.returncode == ExitCode.input_file
|
||||
assert out.find('password')
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', [
|
||||
'hocr',
|
||||
'tesseract',
|
||||
])
|
||||
def test_pagesegmode(renderer, spoof_tesseract_cache):
|
||||
check_ocrmypdf(
|
||||
'skew.pdf', 'test_psm_%s.pdf' % renderer,
|
||||
'--tesseract-pagesegmode', '7',
|
||||
'-v', '1',
|
||||
'--pdf-renderer', renderer, env=spoof_tesseract_cache)
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
#!/usr/bin/env python3
|
||||
# © 2015 James R. Barlow: github.com/jbarlow83
|
||||
|
||||
from ocrmypdf import pageinfo
|
||||
from reportlab.pdfgen.canvas import Canvas
|
||||
from PIL import Image
|
||||
from tempfile import NamedTemporaryFile
|
||||
from contextlib import suppress
|
||||
import os
|
||||
import shutil
|
||||
import pytest
|
||||
import img2pdf
|
||||
import pytest
|
||||
import sys
|
||||
|
||||
|
||||
if sys.version_info.major < 3:
|
||||
print("Requires Python 3.4+")
|
||||
sys.exit(1)
|
||||
|
||||
TESTS_ROOT = os.path.abspath(os.path.dirname(__file__))
|
||||
SPOOF_PATH = os.path.join(TESTS_ROOT, 'spoof')
|
||||
PROJECT_ROOT = os.path.dirname(TESTS_ROOT)
|
||||
OCRMYPDF = os.path.join(PROJECT_ROOT, 'OCRmyPDF.sh')
|
||||
TEST_RESOURCES = os.path.join(PROJECT_ROOT, 'tests', 'resources')
|
||||
TEST_OUTPUT = os.environ.get(
|
||||
'OCRMYPDF_TEST_OUTPUT',
|
||||
default=os.path.join(PROJECT_ROOT, 'tests', 'output', 'pageinfo'))
|
||||
|
||||
|
||||
def setup_module():
|
||||
with suppress(FileNotFoundError):
|
||||
shutil.rmtree(TEST_OUTPUT)
|
||||
with suppress(FileExistsError):
|
||||
os.makedirs(TEST_OUTPUT)
|
||||
|
||||
|
||||
def _make_input(input_basename):
|
||||
return os.path.join(TEST_RESOURCES, input_basename)
|
||||
|
||||
|
||||
def _make_output(output_basename):
|
||||
return os.path.join(TEST_OUTPUT, output_basename)
|
||||
|
||||
|
||||
def test_single_page_text():
|
||||
filename = os.path.join(TEST_OUTPUT, 'text.pdf')
|
||||
pdf = Canvas(filename, pagesize=(8*72, 6*72))
|
||||
text = pdf.beginText()
|
||||
text.setFont('Helvetica', 12)
|
||||
text.setTextOrigin(1*72, 3*72)
|
||||
text.textLine("Methink'st thou art a general offence and every"
|
||||
" man should beat thee.")
|
||||
pdf.drawText(text)
|
||||
pdf.showPage()
|
||||
pdf.save()
|
||||
|
||||
pdfinfo = pageinfo.pdf_get_all_pageinfo(filename)
|
||||
|
||||
assert len(pdfinfo) == 1
|
||||
page = pdfinfo[0]
|
||||
|
||||
assert page['has_text']
|
||||
assert len(page['images']) == 0
|
||||
|
||||
|
||||
def test_single_page_image():
|
||||
filename = os.path.join(TEST_OUTPUT, 'image-mono.pdf')
|
||||
|
||||
with NamedTemporaryFile() as im_tmp:
|
||||
im = Image.new('1', (8, 8), 0)
|
||||
for n in range(8):
|
||||
im.putpixel((n, n), 1)
|
||||
im.save(im_tmp.name, format='PNG')
|
||||
|
||||
pdf_bytes = img2pdf.convert([im_tmp.name], dpi=8)
|
||||
with open(filename, 'wb') as pdf:
|
||||
pdf.write(pdf_bytes)
|
||||
|
||||
pdfinfo = pageinfo.pdf_get_all_pageinfo(filename)
|
||||
|
||||
assert len(pdfinfo) == 1
|
||||
page = pdfinfo[0]
|
||||
|
||||
assert not page['has_text']
|
||||
assert len(page['images']) == 1
|
||||
|
||||
pdfimage = page['images'][0]
|
||||
assert pdfimage['width'] == 8
|
||||
assert pdfimage['color'] == 'gray'
|
||||
|
||||
# While unexpected, this is correct
|
||||
# PDF spec says /FlateDecode image must have /BitsPerComponent 8
|
||||
# So mono images get upgraded to 8-bit
|
||||
assert pdfimage['bpc'] == 8
|
||||
|
||||
# DPI in a 1"x1" is the image width
|
||||
assert pdfimage['dpi_w'] == 8
|
||||
assert pdfimage['dpi_h'] == 8
|
||||
|
||||
|
||||
def test_single_page_inline_image():
|
||||
filename = os.path.join(TEST_OUTPUT, 'image-mono-inline.pdf')
|
||||
pdf = Canvas(filename, pagesize=(8*72, 6*72))
|
||||
with NamedTemporaryFile() as im_tmp:
|
||||
im = Image.new('1', (8, 8), 0)
|
||||
for n in range(8):
|
||||
im.putpixel((n, n), 1)
|
||||
im.save(im_tmp.name, format='PNG')
|
||||
# Draw image in a 72x72 pt or 1"x1" area
|
||||
pdf.drawInlineImage(im_tmp.name, 0, 0, width=72, height=72)
|
||||
pdf.showPage()
|
||||
pdf.save()
|
||||
|
||||
with pytest.raises(NotImplementedError):
|
||||
pageinfo.pdf_get_all_pageinfo(filename)
|
||||
|
||||
|
||||
def test_jpeg():
|
||||
filename = _make_input('c02-22.pdf')
|
||||
|
||||
pdfinfo = pageinfo.pdf_get_all_pageinfo(filename)
|
||||
|
||||
pdfimage = pdfinfo[0]['images'][0]
|
||||
assert pdfimage['enc'] == 'jpeg'
|
||||
|
||||
Reference in New Issue
Block a user