Compare commits
868
Commits
v14.0.2
...
release/v17
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0a0756b33e | ||
|
|
c5d3ef4b17 | ||
|
|
6b37583674 | ||
|
|
de5f2b80f0 | ||
|
|
d951b4f0f7 | ||
|
|
b386d39b3b | ||
|
|
ec595a395b | ||
|
|
bd29269c00 | ||
|
|
6fb7c5d95f | ||
|
|
d57552c4f8 | ||
|
|
f017c982cf | ||
|
|
7ac51ac1a7 | ||
|
|
db9f94de14 | ||
|
|
37e7131a01 | ||
|
|
bc745d4d81 | ||
|
|
c818ad5e75 | ||
|
|
4b16228a4a | ||
|
|
d40fca2590 | ||
|
|
99f8106936 | ||
|
|
ef88ba3f95 | ||
|
|
2f4280b66c | ||
|
|
6cf9d1c6ee | ||
|
|
6a7164a76c | ||
|
|
3f328785f0 | ||
|
|
5acf21651f | ||
|
|
7bfe3ecd5b | ||
|
|
5371cc5e39 | ||
|
|
4c7086c609 | ||
|
|
bf76c8270c | ||
|
|
740f67091c | ||
|
|
36dea181e6 | ||
|
|
c69f293322 | ||
|
|
e9fe061c30 | ||
|
|
c9ea07e954 | ||
|
|
0c3745a1a4 | ||
|
|
664c3e2a8e | ||
|
|
315d0df0e9 | ||
|
|
3c94ada857 | ||
|
|
fcbdbac602 | ||
|
|
122450c19e | ||
|
|
0c4ee5af4e | ||
|
|
bdc50e9470 | ||
|
|
4cb488d0fc | ||
|
|
bb5238e524 | ||
|
|
900a60fd10 | ||
|
|
f5617ce44e | ||
|
|
0e946a7498 | ||
|
|
b2b6a7c4b1 | ||
|
|
75c664793e | ||
|
|
bbd263ff48 | ||
|
|
7a4b98974c | ||
|
|
d72a494979 | ||
|
|
64726f97b3 | ||
|
|
83a43408c2 | ||
|
|
2cb0973540 | ||
|
|
0d6e0c4560 | ||
|
|
94d7735862 | ||
|
|
c540967429 | ||
|
|
195344d307 | ||
|
|
de63d6eac9 | ||
|
|
6ada11ddae | ||
|
|
fc30cb8903 | ||
|
|
01a3706281 | ||
|
|
e613db6a82 | ||
|
|
742a4bac17 | ||
|
|
4c1ef0b471 | ||
|
|
eace567f7b | ||
|
|
e9bfce34f1 | ||
|
|
16c2604a07 | ||
|
|
9ebba91466 | ||
|
|
aec995aced | ||
|
|
be425e7405 | ||
|
|
b4f9673364 | ||
|
|
9ea804aff5 | ||
|
|
e162361d28 | ||
|
|
22d00837e3 | ||
|
|
0faba42d36 | ||
|
|
57e2600566 | ||
|
|
41758766a1 | ||
|
|
3e46b039ed | ||
|
|
ae783b4ae6 | ||
|
|
b9f488d65c | ||
|
|
ed813cec67 | ||
|
|
938ce8e285 | ||
|
|
cf3fb6e89b | ||
|
|
3482ea5fe5 | ||
|
|
e85c5bbb4d | ||
|
|
740b0bddc6 | ||
|
|
a4ee513cd4 | ||
|
|
0ad7f5fc13 | ||
|
|
47cea37487 | ||
|
|
b89bb3b524 | ||
|
|
95d9c3ed18 | ||
|
|
f91e41a209 | ||
|
|
f6fcdfa618 | ||
|
|
01ea6c2b8b | ||
|
|
f02d733d31 | ||
|
|
28d6ea0f10 | ||
|
|
6913ec7cb8 | ||
|
|
40f01d85ae | ||
|
|
b7640bdb9c | ||
|
|
62ad37b276 | ||
|
|
b1de6a6ad4 | ||
|
|
e4fa9dbc8f | ||
|
|
b7737446e4 | ||
|
|
42891346d1 | ||
|
|
4ed0e4510c | ||
|
|
4f9c4c3e52 | ||
|
|
08ee5690bc | ||
|
|
3f38ea4d80 | ||
|
|
f04b5504e8 | ||
|
|
69185e5819 | ||
|
|
ade3ecd5a1 | ||
|
|
e4f8ba8edc | ||
|
|
1225c0a45e | ||
|
|
f0c292f4e1 | ||
|
|
1f493ba789 | ||
|
|
e1d976168c | ||
|
|
60182ac8a8 | ||
|
|
53db7b384b | ||
|
|
d77d63f1dc | ||
|
|
ff250afa51 | ||
|
|
cdb976db41 | ||
|
|
9535b52d06 | ||
|
|
e1216eddb0 | ||
|
|
21d69ffe87 | ||
|
|
3baeb83533 | ||
|
|
134f4fcc28 | ||
|
|
a869a4ac42 | ||
|
|
48a2fdb0f2 | ||
|
|
91a2d39845 | ||
|
|
3a0a7c546b | ||
|
|
d0a46a0359 | ||
|
|
cb22a35834 | ||
|
|
9ff7ab491c | ||
|
|
0c3110857e | ||
|
|
aad90bcb54 | ||
|
|
5f685aef6e | ||
|
|
65b89cafde | ||
|
|
eeda99636a | ||
|
|
afc85333ac | ||
|
|
3987a610e1 | ||
|
|
dab969f97d | ||
|
|
480a8253eb | ||
|
|
8a06dd478a | ||
|
|
4dbd34f06a | ||
|
|
8668cf4524 | ||
|
|
1d74c2831f | ||
|
|
530186b468 | ||
|
|
7b37f57b1c | ||
|
|
d4b7165d72 | ||
|
|
f5bfd2fd3e | ||
|
|
cdf956ffc4 | ||
|
|
c6b21d4dea | ||
|
|
7575dddefc | ||
|
|
66a3e8508e | ||
|
|
7bb3a97208 | ||
|
|
d2add01217 | ||
|
|
4476e81240 | ||
|
|
a373fcd649 | ||
|
|
87478bc240 | ||
|
|
04ad78f01d | ||
|
|
62c3ae80c7 | ||
|
|
d556014185 | ||
|
|
d18efcbbf1 | ||
|
|
f9a4a2e240 | ||
|
|
5f89100dc3 | ||
|
|
1ef9aaf659 | ||
|
|
4c4a1cfa17 | ||
|
|
5251e21f7e | ||
|
|
28eb923d9f | ||
|
|
1579337ebe | ||
|
|
f673da9ab9 | ||
|
|
8d715c4157 | ||
|
|
0f3c7765aa | ||
|
|
9dbce33ee6 | ||
|
|
54ce09496c | ||
|
|
f4c6c8121b | ||
|
|
057eaff36d | ||
|
|
b88d63bdf7 | ||
|
|
a385cd967d | ||
|
|
2f72f8e94a | ||
|
|
ee47e986f3 | ||
|
|
e44063da15 | ||
|
|
abc2d41e2d | ||
|
|
38d60ea89b | ||
|
|
35ec90af44 | ||
|
|
aa1cc8ae04 | ||
|
|
eaceb66030 | ||
|
|
b1dcc2c445 | ||
|
|
ab3855af48 | ||
|
|
5c6cc4031f | ||
|
|
f181307e50 | ||
|
|
b213efb030 | ||
|
|
f59e68911f | ||
|
|
9605656a2f | ||
|
|
599fb1a1f6 | ||
|
|
9a2c0cf6ff | ||
|
|
414d80fc16 | ||
|
|
7ca4ae4e16 | ||
|
|
7e7e2f2e91 | ||
|
|
d07231a7aa | ||
|
|
0e831db9f4 | ||
|
|
650ca1c65b | ||
|
|
a7b0c0df6c | ||
|
|
d735791524 | ||
|
|
66308c2813 | ||
|
|
d81de57bbc | ||
|
|
a9a8b39dba | ||
|
|
fd5b8132ae | ||
|
|
63675c21ce | ||
|
|
6af22051a8 | ||
|
|
8318ebbaec | ||
|
|
4fc0c3a0d5 | ||
|
|
74305e8741 | ||
|
|
d6b069d3fa | ||
|
|
194ca699a8 | ||
|
|
175b743ffe | ||
|
|
080b73e7c0 | ||
|
|
df6079c06d | ||
|
|
45cf92f40b | ||
|
|
5b1900beec | ||
|
|
c0208f0da1 | ||
|
|
61163c2aa9 | ||
|
|
332369f1b0 | ||
|
|
7ea940a3a6 | ||
|
|
8a784d6052 | ||
|
|
5cf86a7c2e | ||
|
|
3beabf55e7 | ||
|
|
6f6448f286 | ||
|
|
9f6e5a48ad | ||
|
|
ee3da07710 | ||
|
|
45043f6a8c | ||
|
|
b166e86216 | ||
|
|
1e2d76b931 | ||
|
|
6851ea7f11 | ||
|
|
4143154e91 | ||
|
|
7c5bed41f1 | ||
|
|
9865f01f47 | ||
|
|
be3971e755 | ||
|
|
3304498bdc | ||
|
|
e4a8f7a354 | ||
|
|
d1a45e4abc | ||
|
|
3b9367fc69 | ||
|
|
92a78f611e | ||
|
|
6f16d0130a | ||
|
|
8b1443c482 | ||
|
|
d84c47816c | ||
|
|
15a77c9d69 | ||
|
|
43c84ca268 | ||
|
|
4125b8a456 | ||
|
|
07e774cce9 | ||
|
|
553a20a8e6 | ||
|
|
172ba4cad1 | ||
|
|
0f5ccb71ca | ||
|
|
0970cebfea | ||
|
|
6de6749062 | ||
|
|
7b2dd892e5 | ||
|
|
c05ed7297c | ||
|
|
c29f58a8b7 | ||
|
|
eb303fef1a | ||
|
|
2a55ceadd0 | ||
|
|
71991ad09b | ||
|
|
bd60d6ccd9 | ||
|
|
83b4469ef1 | ||
|
|
d2a7caf496 | ||
|
|
ff0ea45bf2 | ||
|
|
b5bc1d209c | ||
|
|
53270b8eb1 | ||
|
|
3049a10757 | ||
|
|
53002f65d9 | ||
|
|
acea9529ea | ||
|
|
32322a9fe9 | ||
|
|
6b09129911 | ||
|
|
e4274a956d | ||
|
|
19af116034 | ||
|
|
a5896c45e8 | ||
|
|
b7d63f3dc1 | ||
|
|
137b054f43 | ||
|
|
e6daa28c6d | ||
|
|
2512093076 | ||
|
|
66bc4a3733 | ||
|
|
65df44f670 | ||
|
|
6edc749023 | ||
|
|
cff98d258e | ||
|
|
d1fc77e1b6 | ||
|
|
17eed0529a | ||
|
|
f02353686d | ||
|
|
32813a3c3d | ||
|
|
073a434ab3 | ||
|
|
f390e7f9d1 | ||
|
|
bfbe571f12 | ||
|
|
368568b8ea | ||
|
|
55e7177dbe | ||
|
|
b486df7e2d | ||
|
|
74a84b6ae9 | ||
|
|
cfebf1dc8b | ||
|
|
1aaff4af6f | ||
|
|
36c82e0659 | ||
|
|
522f9d5f56 | ||
|
|
796e424ee5 | ||
|
|
d87db6cad0 | ||
|
|
dd6ed4c5f8 | ||
|
|
206bab74bc | ||
|
|
b333480749 | ||
|
|
f71a5ffd61 | ||
|
|
b7c3ea70ed | ||
|
|
636623ab49 | ||
|
|
74253e5fc8 | ||
|
|
02d85ff070 | ||
|
|
179c36151b | ||
|
|
3c4b099cb1 | ||
|
|
15df9c370c | ||
|
|
86d92ef490 | ||
|
|
8f44b29ca3 | ||
|
|
5a08a6cfeb | ||
|
|
6d2d870711 | ||
|
|
cc058be4b2 | ||
|
|
7565d20c0a | ||
|
|
9a075039b5 | ||
|
|
5a1c043331 | ||
|
|
fe89be5dc0 | ||
|
|
d70296b97a | ||
|
|
7d7658018d | ||
|
|
8fb8e9f72c | ||
|
|
85d6fb8ce9 | ||
|
|
828e741c24 | ||
|
|
36837f8353 | ||
|
|
12fd4f70f1 | ||
|
|
250615561d | ||
|
|
a659f83d67 | ||
|
|
08f95c0b13 | ||
|
|
dbd3c93757 | ||
|
|
5d128a91d2 | ||
|
|
a1b8113d56 | ||
|
|
f052e910c9 | ||
|
|
116e2692d0 | ||
|
|
b2669c7d71 | ||
|
|
c8c53d38a3 | ||
|
|
d303b42c86 | ||
|
|
f77f701a50 | ||
|
|
1c3b7d1507 | ||
|
|
bf62562787 | ||
|
|
6c6cbfd4d6 | ||
|
|
ee5acbe94e | ||
|
|
5e478a7774 | ||
|
|
92c5200ad2 | ||
|
|
86a102f8e6 | ||
|
|
2463b91051 | ||
|
|
07f7c6b812 | ||
|
|
8138664287 | ||
|
|
120ca72393 | ||
|
|
f9b3e9a97b | ||
|
|
1e87930bbb | ||
|
|
fe4725658e | ||
|
|
9d042767cc | ||
|
|
23bc247b9c | ||
|
|
e44bf46d77 | ||
|
|
f50620c244 | ||
|
|
6f755321b8 | ||
|
|
706681deb8 | ||
|
|
c283cf0a0d | ||
|
|
0f82d7223e | ||
|
|
9a6150ae53 | ||
|
|
fec0948a13 | ||
|
|
18b59c57b4 | ||
|
|
a67a11e61c | ||
|
|
6ca4940a32 | ||
|
|
0e4cce2642 | ||
|
|
8fca0c71dc | ||
|
|
944d99bdc1 | ||
|
|
5bb6e1c5d7 | ||
|
|
8d7a8f0f98 | ||
|
|
b9dd0a5e3c | ||
|
|
6949ad2c5d | ||
|
|
b3324c3b4e | ||
|
|
b38cac6931 | ||
|
|
bb4c47e707 | ||
|
|
5e1e2497ab | ||
|
|
cd910fbf21 | ||
|
|
1225269a4b | ||
|
|
3a75b20740 | ||
|
|
d35d008806 | ||
|
|
f5662d5eb0 | ||
|
|
39010dd255 | ||
|
|
fbaad570c7 | ||
|
|
f974e3b3c1 | ||
|
|
46b49cc176 | ||
|
|
5256e74d0c | ||
|
|
621d6a0b89 | ||
|
|
08be7c8bbe | ||
|
|
980a5472b6 | ||
|
|
51c618e357 | ||
|
|
4dde3786c2 | ||
|
|
d544342602 | ||
|
|
fac91fca2a | ||
|
|
6edf756849 | ||
|
|
4fb1bb4de6 | ||
|
|
6a8eb7daaa | ||
|
|
0544d06c3d | ||
|
|
34c285c9ac | ||
|
|
2f53b27651 | ||
|
|
772677746b | ||
|
|
f0bad87ea6 | ||
|
|
44e71f8c14 | ||
|
|
964b30ca26 | ||
|
|
214a333e2d | ||
|
|
ec6401ab57 | ||
|
|
cbc5e8ce8d | ||
|
|
a1c4cfe8f1 | ||
|
|
3a721e6578 | ||
|
|
e6b716cdde | ||
|
|
02c39998b8 | ||
|
|
0774bc7f14 | ||
|
|
c6a98b3d0b | ||
|
|
981bbf1105 | ||
|
|
2b0c6cfd40 | ||
|
|
59f6bc8306 | ||
|
|
653c4ffb45 | ||
|
|
d947ca258e | ||
|
|
d5ff7f7db9 | ||
|
|
579cef3649 | ||
|
|
cb2f090c60 | ||
|
|
f3d6387bca | ||
|
|
abf9729c61 | ||
|
|
442e9c9f0d | ||
|
|
397fad249d | ||
|
|
9a3c5a3f7c | ||
|
|
950c700274 | ||
|
|
26432c38a9 | ||
|
|
28be50136c | ||
|
|
0c62f2de5d | ||
|
|
5caf654f22 | ||
|
|
205593445e | ||
|
|
f25fb8c63a | ||
|
|
99c78650b6 | ||
|
|
69355886a8 | ||
|
|
08e89e2dbe | ||
|
|
0e013df161 | ||
|
|
9ba4e3ab46 | ||
|
|
5fdcb7602b | ||
|
|
b4db1b741f | ||
|
|
7a8cc21e31 | ||
|
|
0674829d8f | ||
|
|
315aa0474b | ||
|
|
df3451e779 | ||
|
|
3ba42802d1 | ||
|
|
d6342cb8c2 | ||
|
|
065bddbc6c | ||
|
|
067f429dde | ||
|
|
6895c2d70f | ||
|
|
686481982a | ||
|
|
a9e1d19b78 | ||
|
|
f95aa63718 | ||
|
|
855de287b2 | ||
|
|
feeb9f213f | ||
|
|
e7eb8fa805 | ||
|
|
8a747f005a | ||
|
|
16ab4a8b4e | ||
|
|
8d30cff4ef | ||
|
|
59d5b0d1bd | ||
|
|
9ec0745ab8 | ||
|
|
3a3635f7f9 | ||
|
|
6a746a1cbb | ||
|
|
906c130f96 | ||
|
|
4a78458821 | ||
|
|
fddf3ce2f4 | ||
|
|
353b34e695 | ||
|
|
7d63355c3c | ||
|
|
42ff7fc842 | ||
|
|
26470fe16a | ||
|
|
3b9d4b7f0a | ||
|
|
11f53fe9a9 | ||
|
|
123c0c766f | ||
|
|
6a9be2142e | ||
|
|
0bc350f55e | ||
|
|
7a6edf62ba | ||
|
|
07b6f06f11 | ||
|
|
2005f622bb | ||
|
|
cca04fd799 | ||
|
|
75bf8e4ba2 | ||
|
|
daabb5b100 | ||
|
|
035ebea72f | ||
|
|
a499956462 | ||
|
|
74d2a156c4 | ||
|
|
f87fc7b12d | ||
|
|
602f5632cb | ||
|
|
9fbbcf7599 | ||
|
|
9498f01f59 | ||
|
|
2c59aca5a1 | ||
|
|
51301d69c9 | ||
|
|
7e608fd1df | ||
|
|
ecc79315df | ||
|
|
14365d10b8 | ||
|
|
5e5320020f | ||
|
|
103c3e0cd6 | ||
|
|
7a1c89edd9 | ||
|
|
a5ff3d2f42 | ||
|
|
b71d16dd96 | ||
|
|
fd593eb5e9 | ||
|
|
a0b98abb94 | ||
|
|
18353e1e94 | ||
|
|
9adcad84da | ||
|
|
f2714586d8 | ||
|
|
0b6fb62967 | ||
|
|
1db8b0b943 | ||
|
|
f38aebb3d5 | ||
|
|
7162c36d37 | ||
|
|
f4d4ea46c8 | ||
|
|
2fd1a0f178 | ||
|
|
73ed33a086 | ||
|
|
e6095a9949 | ||
|
|
16f05af401 | ||
|
|
1631afc878 | ||
|
|
63d87fc440 | ||
|
|
9489c01259 | ||
|
|
30d92ad83f | ||
|
|
a4987733c4 | ||
|
|
39eee05230 | ||
|
|
5b2f2e6290 | ||
|
|
445617a1a5 | ||
|
|
f6e90a5934 | ||
|
|
43618e6b3f | ||
|
|
e97f89de3b | ||
|
|
11d3e32f1e | ||
|
|
2affa83efe | ||
|
|
c90d5cd84b | ||
|
|
aacaba3d26 | ||
|
|
fec53be841 | ||
|
|
3f7b540f76 | ||
|
|
d217856166 | ||
|
|
e2be457e9b | ||
|
|
4850f486d2 | ||
|
|
729c7febd9 | ||
|
|
6c6aca2f1e | ||
|
|
c69823f496 | ||
|
|
73f8f6aac8 | ||
|
|
d944254e45 | ||
|
|
f7ddffe554 | ||
|
|
8a73ed5d5a | ||
|
|
03669183d7 | ||
|
|
74e101a2fa | ||
|
|
532cf18ad3 | ||
|
|
0b90b697e2 | ||
|
|
6be7c5f7c8 | ||
|
|
db2e5132e6 | ||
|
|
b14f6f778a | ||
|
|
415de77457 | ||
|
|
a9466c4f58 | ||
|
|
d9ae453a63 | ||
|
|
9841e09233 | ||
|
|
0ca314e066 | ||
|
|
d7680cae27 | ||
|
|
491b6bdb1f | ||
|
|
c591f9601a | ||
|
|
8d1e75017e | ||
|
|
94615f7ad4 | ||
|
|
e5df8e1315 | ||
|
|
d739b91aef | ||
|
|
686cfb2539 | ||
|
|
2633716bb7 | ||
|
|
0a07c0a44e | ||
|
|
2ca6e110ca | ||
|
|
334a07c839 | ||
|
|
a57c39358d | ||
|
|
30a0c315fb | ||
|
|
b860f0d94c | ||
|
|
14f4c19f5a | ||
|
|
7ab5c55d46 | ||
|
|
8b6ecd5971 | ||
|
|
7b0871ae4c | ||
|
|
b73af7ce10 | ||
|
|
60645717e2 | ||
|
|
1cbf578538 | ||
|
|
e966c1fceb | ||
|
|
d0133f8641 | ||
|
|
6d30b497dc | ||
|
|
f3b89e66eb | ||
|
|
04154e207c | ||
|
|
9898904be7 | ||
|
|
27d5229842 | ||
|
|
4a9a575ef0 | ||
|
|
52fd9a630d | ||
|
|
a596ccf844 | ||
|
|
e7fa97731f | ||
|
|
290aa28108 | ||
|
|
a95640ed9e | ||
|
|
f69267bb67 | ||
|
|
e36d5a309f | ||
|
|
55566d9830 | ||
|
|
f02ea20678 | ||
|
|
372c22d42b | ||
|
|
949265bbd0 | ||
|
|
916106733c | ||
|
|
44bcafd3aa | ||
|
|
71166f7be8 | ||
|
|
580252a1a0 | ||
|
|
c0b60dae6a | ||
|
|
ae123fd209 | ||
|
|
454ad0acc5 | ||
|
|
0c306ac328 | ||
|
|
52d99732b1 | ||
|
|
5b5827983b | ||
|
|
56f9bc311d | ||
|
|
eb17dc1ecf | ||
|
|
6f8115a052 | ||
|
|
aac913c666 | ||
|
|
b5e73ac4e4 | ||
|
|
9e98c90891 | ||
|
|
ca2592c1d9 | ||
|
|
a31f17bb9d | ||
|
|
1cb46afa94 | ||
|
|
5a759947dd | ||
|
|
db3df13e95 | ||
|
|
2a8bc03167 | ||
|
|
d2297b39d0 | ||
|
|
e4cd081d4d | ||
|
|
d2dbea6cf8 | ||
|
|
46a279a49a | ||
|
|
299f0c4003 | ||
|
|
9ffb45f283 | ||
|
|
cd61c4efd9 | ||
|
|
a06ab2a1c5 | ||
|
|
dfa4ebf1a6 | ||
|
|
58f388c69d | ||
|
|
990b462a94 | ||
|
|
b928dc0808 | ||
|
|
8916955f45 | ||
|
|
82bef40aa6 | ||
|
|
1c45f32941 | ||
|
|
fadc0cf69b | ||
|
|
7ce9d08b2d | ||
|
|
eb3a51e33a | ||
|
|
f3dd733773 | ||
|
|
4dbc5e1dba | ||
|
|
c0637c287e | ||
|
|
6127f7abd6 | ||
|
|
a4059762e6 | ||
|
|
40afcd68a7 | ||
|
|
f238e721ed | ||
|
|
16eb5627a7 | ||
|
|
fbf0674189 | ||
|
|
62c4f65fc3 | ||
|
|
e400112f32 | ||
|
|
7935914f55 | ||
|
|
ad3a1dbbad | ||
|
|
0655f8e7ae | ||
|
|
04a9372584 | ||
|
|
b9646b6f85 | ||
|
|
53c953a561 | ||
|
|
c278fecb34 | ||
|
|
23951c9e38 | ||
|
|
e8ae370ceb | ||
|
|
67be4d1904 | ||
|
|
6f82097d14 | ||
|
|
fc6f959d21 | ||
|
|
e38d569d8f | ||
|
|
0856750ee2 | ||
|
|
05721ba84a | ||
|
|
38c3422e5e | ||
|
|
d153a6f6df | ||
|
|
1a7738a925 | ||
|
|
8985c0dfe9 | ||
|
|
ebfe008432 | ||
|
|
1f16eb6f50 | ||
|
|
cbb0868ae3 | ||
|
|
68bb38d0ad | ||
|
|
0443e87345 | ||
|
|
b3de5833d3 | ||
|
|
95b14ee282 | ||
|
|
07b89e6a19 | ||
|
|
8991d2cb33 | ||
|
|
86a20c4130 | ||
|
|
6827a6efe8 | ||
|
|
c6b5332699 | ||
|
|
68610046c6 | ||
|
|
880326868d | ||
|
|
c6be3ba076 | ||
|
|
0565cb0b10 | ||
|
|
dc49906704 | ||
|
|
93fda0dd00 | ||
|
|
d4110e78cb | ||
|
|
5285d68fcc | ||
|
|
2b0e149809 | ||
|
|
b7ce5b0d7d | ||
|
|
ffd6a64ce9 | ||
|
|
5727f1e081 | ||
|
|
b75a7eca2a | ||
|
|
2b01676434 | ||
|
|
e11c386c58 | ||
|
|
9346d1f970 | ||
|
|
012cbef865 | ||
|
|
0687568e1b | ||
|
|
3086cfc3d9 | ||
|
|
91a14660b3 | ||
|
|
539f0ee0ce | ||
|
|
7172817cd6 | ||
|
|
d9cc759142 | ||
|
|
364799fc3e | ||
|
|
f4c211fa2d | ||
|
|
113a6b45bd | ||
|
|
e9419d2c40 | ||
|
|
fb006ef39f | ||
|
|
890b994403 | ||
|
|
01bbf7d144 | ||
|
|
468de5324a | ||
|
|
072db75fa3 | ||
|
|
8519b3f625 | ||
|
|
dd7c4f3eaa | ||
|
|
c8e6f20f8d | ||
|
|
10530a8698 | ||
|
|
207866abf5 | ||
|
|
3829af16fb | ||
|
|
24db31b4c5 | ||
|
|
8132a4ae10 | ||
|
|
d5128c5cf5 | ||
|
|
270e31fa67 | ||
|
|
85e31d0a19 | ||
|
|
ea36aedb5f | ||
|
|
bd4d44e182 | ||
|
|
8fcf358934 | ||
|
|
47b0f28564 | ||
|
|
7018e2b247 | ||
|
|
8d12ecb798 | ||
|
|
0ab29ec0ba | ||
|
|
179714770a | ||
|
|
f04f45545c | ||
|
|
a3a083c125 | ||
|
|
d855f63985 | ||
|
|
d4863cbf0f | ||
|
|
7b8f081fbf | ||
|
|
7d33039bcd | ||
|
|
fde886baf4 | ||
|
|
146da79c00 | ||
|
|
2fc3b0d973 | ||
|
|
5667424530 | ||
|
|
8add531ffd | ||
|
|
0388c23ae7 | ||
|
|
9b77daae7c | ||
|
|
3e1b3ec98d | ||
|
|
0f0ca6f517 | ||
|
|
c93349c350 | ||
|
|
0c287929c2 | ||
|
|
23a37fc35c | ||
|
|
162a47f98e | ||
|
|
0239f69912 | ||
|
|
2ad8961d0b | ||
|
|
eec8a2b574 | ||
|
|
6c78076bea | ||
|
|
2637e84691 | ||
|
|
e8c82ee4b6 | ||
|
|
de2bb5ce8c | ||
|
|
ec1c377532 | ||
|
|
173428e81a | ||
|
|
67ed29dcea | ||
|
|
3454c050ed | ||
|
|
5ee99b26e7 | ||
|
|
ac3aa67d8a | ||
|
|
1768a1eda9 | ||
|
|
5902fe45c1 | ||
|
|
78981641f0 | ||
|
|
c77ae4b34c | ||
|
|
b2cbbf0099 | ||
|
|
6b6c34af01 | ||
|
|
be12f7a728 | ||
|
|
e3c813fc67 | ||
|
|
35a1eaf62a | ||
|
|
d393d18c13 | ||
|
|
54e622ad10 | ||
|
|
330352aeed | ||
|
|
ac2fc49208 | ||
|
|
4bee7355e9 | ||
|
|
86f2b1f9a7 | ||
|
|
3002409e49 | ||
|
|
0cf6828c20 | ||
|
|
331c829b6e | ||
|
|
06a5e0c3f6 | ||
|
|
811f23381a | ||
|
|
a371655052 | ||
|
|
a6ce35b13a | ||
|
|
45added738 | ||
|
|
6e20439c91 | ||
|
|
72e056436c | ||
|
|
e02ba19097 | ||
|
|
d3b858f994 | ||
|
|
19045c4f21 | ||
|
|
f4d89fe6cc | ||
|
|
ab85c0f5a9 | ||
|
|
32693b683d | ||
|
|
b5dc276ba1 | ||
|
|
7c38c71794 | ||
|
|
a80e7a127b | ||
|
|
cf3309555f | ||
|
|
f80dd0d86a | ||
|
|
1ba2bce486 | ||
|
|
050dd1f5a8 | ||
|
|
e44a57aec0 | ||
|
|
d94d2671c3 | ||
|
|
5124daa79f | ||
|
|
7293847da7 | ||
|
|
59fd0ac587 | ||
|
|
90619b308c | ||
|
|
d0d49ce989 | ||
|
|
bf0224faa4 | ||
|
|
ae2f8ed8f1 | ||
|
|
14ac9b0560 | ||
|
|
dbe6148d41 | ||
|
|
36d4c2dbbc | ||
|
|
adbffb7bd9 | ||
|
|
05ecb6ca46 | ||
|
|
0a7b60cda5 | ||
|
|
5f211ecf6f | ||
|
|
417ee067a2 | ||
|
|
c4649dabef | ||
|
|
6eadd65dfb | ||
|
|
e8ed510543 | ||
|
|
22d35c199d | ||
|
|
5a82ad63c9 | ||
|
|
4769a6c50b | ||
|
|
9004009adc | ||
|
|
11221f9912 | ||
|
|
177349cc84 | ||
|
|
070c9772ce | ||
|
|
1bc09045a5 | ||
|
|
e46a18dd2f | ||
|
|
c64871c2ed | ||
|
|
de909fb99a | ||
|
|
731b2fc477 | ||
|
|
214f6ec759 | ||
|
|
080aa4dbd1 | ||
|
|
7af5dcd4a4 | ||
|
|
fe9f52fbe7 | ||
|
|
fcbdeb8dbe | ||
|
|
cb251a8d03 | ||
|
|
3731fdfd72 | ||
|
|
b2e6a6431e | ||
|
|
9ff1e56bf6 | ||
|
|
2b30f74fce | ||
|
|
10f4c48e0b | ||
|
|
37d5c086bb | ||
|
|
91830627e5 | ||
|
|
f2fc37b257 | ||
|
|
a99e40fa84 | ||
|
|
9ce692a6f1 | ||
|
|
a3c49b8f31 | ||
|
|
33b70be7d5 | ||
|
|
4924b11b6b | ||
|
|
1d0e4e7c9f | ||
|
|
9b8d14d16e | ||
|
|
b7eb93eb79 | ||
|
|
42c0d0f48f | ||
|
|
5fce50ff7c | ||
|
|
765ed4c386 | ||
|
|
1b2849ec0a | ||
|
|
4f604591b4 | ||
|
|
01dc8e23ff | ||
|
|
b432770cfc | ||
|
|
5502fb8d9f | ||
|
|
e66922b030 | ||
|
|
00e9759b16 | ||
|
|
ba10c5345b | ||
|
|
8a5f94988a | ||
|
|
aa73e3c69f | ||
|
|
9d5fa05a00 | ||
|
|
997380e567 | ||
|
|
2685f910b1 | ||
|
|
bfcc586032 | ||
|
|
2d77b95fd9 |
+40
-28
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM ubuntu:22.04 as base
|
||||
FROM ubuntu:25.04 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
@@ -9,57 +9,67 @@ RUN echo 'debconf debconf/frontend select Noninteractive' | debconf-set-selectio
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
python3 \
|
||||
libqpdf-dev \
|
||||
zlib1g \
|
||||
liblept5
|
||||
python-is-python3
|
||||
|
||||
FROM base as builder
|
||||
FROM base AS builder
|
||||
|
||||
# Note we need leptonica here to build jbig2
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential autoconf automake libtool \
|
||||
libleptonica-dev \
|
||||
zlib1g-dev \
|
||||
python3-dev \
|
||||
python3-distutils \
|
||||
libffi-dev \
|
||||
ca-certificates \
|
||||
curl \
|
||||
git
|
||||
|
||||
# Get the latest pip (Ubuntu version doesn't support manylinux2010)
|
||||
RUN \
|
||||
curl https://bootstrap.pypa.io/get-pip.py | python3
|
||||
git \
|
||||
libcairo2-dev \
|
||||
pkg-config
|
||||
|
||||
# Compile and install jbig2
|
||||
# Needs libleptonica-dev, zlib1g-dev
|
||||
RUN \
|
||||
mkdir jbig2 \
|
||||
&& curl -L https://github.com/agl/jbig2enc/archive/ea6a40a.tar.gz | \
|
||||
&& curl -L https://github.com/agl/jbig2enc/archive/c0141bf.tar.gz | \
|
||||
tar xz -C jbig2 --strip-components=1 \
|
||||
&& cd jbig2 \
|
||||
&& ./autogen.sh && ./configure && make && make install \
|
||||
&& cd .. \
|
||||
&& rm -rf jbig2
|
||||
|
||||
COPY . /app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
RUN pip3 install --no-cache-dir .[test,webservice,watcher]
|
||||
# Copy uv from ghcr
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9.8 /uv /uvx /bin/
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
# Install the project's dependencies using the lockfile and settings
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
--mount=type=bind,source=uv.lock,target=uv.lock \
|
||||
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
|
||||
uv sync --frozen --no-install-project --no-dev
|
||||
|
||||
# Then, add the rest of the project source code and install it
|
||||
# Installing separately from its dependencies allows optimal layer caching
|
||||
COPY . /app
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen \
|
||||
--extra webservice --extra watcher --no-dev \
|
||||
--no-install-package pyarrow
|
||||
|
||||
FROM base
|
||||
|
||||
# For Tesseract 5
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
software-properties-common gpg-agent
|
||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr-devel
|
||||
RUN apt-get update && apt-get install -y software-properties-common
|
||||
|
||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr5
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ghostscript \
|
||||
fonts-droid-fallback \
|
||||
fonts-noto-core \
|
||||
fonts-noto-cjk \
|
||||
jbig2dec \
|
||||
img2pdf \
|
||||
libsm6 libxext6 libxrender-dev \
|
||||
pngquant \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-chi-sim \
|
||||
@@ -76,11 +86,13 @@ WORKDIR /app
|
||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
||||
COPY --from=builder /usr/local/bin/ /usr/local/bin/
|
||||
|
||||
COPY --from=builder /app/misc/webservice.py /app/
|
||||
COPY --from=builder /app/misc/watcher.py /app/
|
||||
COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
# Copy minimal project files to get the test suite.
|
||||
COPY --from=builder /app/pyproject.toml /app/README.md /app/
|
||||
COPY --from=builder /app/tests /app/tests
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
|
||||
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
FROM alpine:3.22 AS base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
ENV TZ=UTC
|
||||
|
||||
RUN apk add --no-cache \
|
||||
python3 \
|
||||
zlib
|
||||
|
||||
FROM base AS builder
|
||||
|
||||
# Yes it really is python3-dev, and py3-package
|
||||
RUN apk add --no-cache \
|
||||
ca-certificates \
|
||||
git \
|
||||
python3-dev \
|
||||
py3-pyarrow \
|
||||
curl
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=ghcr.io/astral-sh/uv:0.9.8 /uv /uvx /bin/
|
||||
|
||||
ENV UV_COMPILE_BYTECODE=1 UV_LINK_MODE=copy
|
||||
|
||||
RUN uv venv --system-site-packages .venv
|
||||
|
||||
# Install the project's dependencies using the lockfile and settings
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
--mount=type=bind,source=uv.lock,target=uv.lock \
|
||||
--mount=type=bind,source=pyproject.toml,target=pyproject.toml \
|
||||
uv sync --frozen --no-install-project --no-dev
|
||||
|
||||
# Then, add the rest of the project source code and install it
|
||||
# Installing separately from its dependencies allows optimal layer caching
|
||||
COPY . /app
|
||||
RUN --mount=type=cache,target=/root/.cache/uv \
|
||||
uv sync --frozen \
|
||||
--extra webservice --extra watcher --no-dev \
|
||||
--no-install-package pyarrow
|
||||
|
||||
FROM base
|
||||
|
||||
RUN apk add --no-cache \
|
||||
ghostscript \
|
||||
jbig2dec \
|
||||
jbig2enc \
|
||||
pngquant \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-data-chi_sim \
|
||||
tesseract-ocr-data-deu \
|
||||
tesseract-ocr-data-eng \
|
||||
tesseract-ocr-data-fra \
|
||||
tesseract-ocr-data-osd \
|
||||
tesseract-ocr-data-por \
|
||||
tesseract-ocr-data-spa \
|
||||
font-noto \
|
||||
ttf-droid \
|
||||
unpaper \
|
||||
&& rm -rf /var/cache/apk/*
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder --chown=app:app /app /app
|
||||
|
||||
RUN rm -rf /app/.git && \
|
||||
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||
ln -s /app/misc/watcher.py /app/watcher.py
|
||||
|
||||
ENV PATH="/app/.venv/bin:${PATH}"
|
||||
|
||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||
@@ -13,5 +13,6 @@
|
||||
*.jpg binary
|
||||
*.bin binary
|
||||
*.afdesign binary
|
||||
*.ttf binary
|
||||
|
||||
.git_archival.txt export-subst
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
name: Installation, packaging, dependencies
|
||||
description: Installation, packages, dependencies, "nothing works", test suite failures...
|
||||
title: "[Bug]: "
|
||||
labels: ["triage"]
|
||||
assignees:
|
||||
- jbarlow83
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Thanks for taking the time to fill out this bug report!
|
||||
|
||||
If your issue involves using OCRmyPDF on specific file(s) and not getting
|
||||
good results, this is the *wrong* issue template. Please use the recommended
|
||||
template to ensure we have enough information to help.
|
||||
- type: textarea
|
||||
id: what-happened
|
||||
attributes:
|
||||
label: What were you trying to do?
|
||||
description: Also tell us, what did you expect to happen?
|
||||
placeholder: Tell us what you see!
|
||||
validations:
|
||||
required: true
|
||||
- type: dropdown
|
||||
id: packaging-system
|
||||
attributes:
|
||||
label: Where are you installing/running from?
|
||||
multiple: true
|
||||
options:
|
||||
- PyPI (pip, poetry, pipx, etc.)
|
||||
- Linux package manager (apt, dnf, etc.)
|
||||
- Wndows package manager (chocolatey, etc.)
|
||||
- Homebrew
|
||||
- Docker container
|
||||
- Ubuntu snap
|
||||
- Conda
|
||||
- source build
|
||||
validations:
|
||||
required: true
|
||||
- type: input
|
||||
id: version
|
||||
attributes:
|
||||
label: OCRmyPDF version
|
||||
description: Paste "ocrmypdf --version" here
|
||||
- type: dropdown
|
||||
id: operating-system
|
||||
attributes:
|
||||
label: What operating system are you working on?
|
||||
multiple: true
|
||||
options:
|
||||
- Linux
|
||||
- Windows
|
||||
- macOS
|
||||
- BSD
|
||||
- type: input
|
||||
id: os_version
|
||||
attributes:
|
||||
label: Operating system details and version
|
||||
- type: checkboxes
|
||||
attributes:
|
||||
label: Simple sanity checks
|
||||
description: Select all that apply
|
||||
options:
|
||||
- label: Operating system is currently supported by its vendor (not end of life)
|
||||
- label: Python version is compatible with OCRmyPDF
|
||||
- label: This issue is not about a specific input file
|
||||
- type: textarea
|
||||
id: logs
|
||||
attributes:
|
||||
label: Relevant log output
|
||||
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||
render: plain text
|
||||
@@ -0,0 +1,75 @@
|
||||
name: Problem with specific file
|
||||
description: Something went wrong while trying to OCR a specific file
|
||||
title: "[Bug]: "
|
||||
labels: ["triage"]
|
||||
assignees:
|
||||
- jbarlow83
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Thanks for taking the time to describe this issue with a particular file.
|
||||
- type: textarea
|
||||
id: what-happened
|
||||
attributes:
|
||||
label: Describe the bug
|
||||
description: A clear and concise description of what the bug is.
|
||||
placeholder: Tell us what you see!
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
id: reproduce
|
||||
attributes:
|
||||
label: Steps to reproduce
|
||||
description: Please include steps to reproduce.
|
||||
value: |
|
||||
1. Run ocrmypdf -v1 ...arguments... input.pdf output.pdf
|
||||
2. Open output.pdf
|
||||
3. ...
|
||||
render: plain text
|
||||
- type: textarea
|
||||
id: files
|
||||
attributes:
|
||||
label: Files
|
||||
description: |
|
||||
Please attach the input and output files, or any screenshots that may be helpful.
|
||||
|
||||
If you cannot provide a test file, we probably won't be able to help with the issue.
|
||||
PDF is a complex file format, and there may be technical details in the PDF that are
|
||||
causing the issue. There's really no substitute for a test file.
|
||||
|
||||
We understand files may contain personal or sensitive information. Here are some options:
|
||||
- Try reproducing the issue with a file from the OCRmyPDF test suite. (See tests/resources)
|
||||
- Try to create another file in the same way as your private file.
|
||||
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
||||
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
||||
omits personal information.
|
||||
placeholder: |
|
||||
Drag and drop files here.
|
||||
- type: dropdown
|
||||
id: packaging-system
|
||||
attributes:
|
||||
label: How did you download and install the software?
|
||||
multiple: true
|
||||
options:
|
||||
- PyPI (pip, poetry, pipx, etc.)
|
||||
- Linux package manager (apt, dnf, etc.)
|
||||
- Windows package manager (chocolatey, etc.)
|
||||
- Homebrew
|
||||
- Docker container
|
||||
- Ubuntu snap
|
||||
- Conda
|
||||
- source build
|
||||
- type: input
|
||||
id: version
|
||||
attributes:
|
||||
label: OCRmyPDF version
|
||||
description: Paste "ocrmypdf --version" here
|
||||
placeholder: ocrmypdf --version
|
||||
- type: textarea
|
||||
id: logs
|
||||
attributes:
|
||||
label: Relevant log output
|
||||
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||
placeholder: Run OCRmyPDF with verbosity `-v1` to get more detailed logging output.
|
||||
render: plain text
|
||||
@@ -0,0 +1,83 @@
|
||||
name: Problem with third party app that uses OCRmyPDF
|
||||
description: |
|
||||
For PDF generation issues with third party software such as Paperless-ngx that
|
||||
uses OCRmyPDF to perform OCR or generate PDFs.
|
||||
title: "[3rdparty]: "
|
||||
labels: ["triage"]
|
||||
assignees:
|
||||
- jbarlow83
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Thanks for taking the time to describe this issue with a particular file
|
||||
and third party app.
|
||||
|
||||
If you are comfortable using OCRmyPDF, please trying to install OCRmyPDF,
|
||||
run it on your file, and see if it works. It's easier for everyone
|
||||
if you can confirm that the issue occurs with OCRmyPDF and not with
|
||||
the third party app.
|
||||
- type: checkboxes
|
||||
attributes:
|
||||
label: Simple sanity checks
|
||||
description: Select all that apply
|
||||
options:
|
||||
- label: This is an issue with an app that uses OCRmyPDF for OCR
|
||||
- label: I am using a recent version of the third party app
|
||||
- label: I will include a file that reproduces the issuse
|
||||
- type: input
|
||||
id: thirdparty-app-name-version
|
||||
attributes:
|
||||
label: Third party app name and version
|
||||
description: e.g. Paperless-ngx 2.9.0
|
||||
- type: textarea
|
||||
id: what-happened
|
||||
attributes:
|
||||
label: Describe the bug
|
||||
description: A clear and concise description of what the bug is.
|
||||
placeholder: Tell us what you see!
|
||||
validations:
|
||||
required: true
|
||||
- type: textarea
|
||||
id: reproduce
|
||||
attributes:
|
||||
label: Steps to reproduce
|
||||
description: Please include steps to reproduce.
|
||||
value: |
|
||||
1. Import attached file into Paperless-ngx
|
||||
2. Trigger OCR
|
||||
3. Check log file
|
||||
4. ...
|
||||
render: plain text
|
||||
- type: textarea
|
||||
id: files
|
||||
attributes:
|
||||
label: Files
|
||||
description: |
|
||||
Please attach the input and output files, or any screenshots that may be helpful.
|
||||
|
||||
If you cannot provide a test file, we probably won't be able to help with the issue.
|
||||
PDF is a complex file format, and there may be technical details in the PDF that are
|
||||
causing the issue. There's really no substitute for a test file.
|
||||
|
||||
We understand files may contain personal or sensitive information. Here are some options:
|
||||
- Try reproducing the issue with a file from the test suite. (See tests/resources)
|
||||
- Try to create another file in the same way as your private file.
|
||||
- Encrypt the file to OCRmyPDF's private GPG key, and then zip the GPG file.
|
||||
- Use ``qpdf --json yourfile.pdf`` to produce a JSON representation of your file that
|
||||
omits personal information.
|
||||
placeholder: |
|
||||
Drag and drop files here.
|
||||
- type: input
|
||||
id: version
|
||||
attributes:
|
||||
label: OCRmyPDF version
|
||||
description: Paste "ocrmypdf --version" here
|
||||
placeholder: ocrmypdf --version
|
||||
- type: textarea
|
||||
id: logs
|
||||
attributes:
|
||||
label: Relevant log output
|
||||
description: Please copy and paste any relevant log output. This will be automatically formatted into code, so no need for backticks.
|
||||
placeholder: Run OCRmyPDF with verbosity `-v1` to get more detailed logging output.
|
||||
render: plain text
|
||||
@@ -0,0 +1,12 @@
|
||||
name: Feature request
|
||||
description: Suggest an idea for this project
|
||||
title: "[Feature]: "
|
||||
labels: ["enhancement", "triage"]
|
||||
assignees:
|
||||
- jbarlow83
|
||||
body:
|
||||
- type: textarea
|
||||
id: feature
|
||||
attributes:
|
||||
label: Describe the proposed feature
|
||||
description: A clear and concise description of what the desired is.
|
||||
@@ -1,23 +0,0 @@
|
||||
---
|
||||
name: Feature request
|
||||
about: Suggest an idea for this project
|
||||
title: ''
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**Is your feature request related to a problem? Please describe.**
|
||||
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
||||
|
||||
**Describe the solution you'd like**
|
||||
A clear and concise description of what you want to happen.
|
||||
|
||||
**Describe alternatives you've considered**
|
||||
A clear and concise description of any alternative solutions or features you've considered. Please include the versions of OCRmyPDF and other supporting programs (Tesseract OCR, Ghostscript) - maybe an alternative already exists in a newer version.
|
||||
|
||||
**Example file**
|
||||
If your issue concerns how OCRmyPDF processes certain files, and please provide an example file that helps illustrate how OCRmyPDF's output could be improve. You could also look in ``tests/resources`` and see if any of those files demonstrates your issue.
|
||||
|
||||
**Additional context**
|
||||
Add any other context or screenshots about the feature request here.
|
||||
@@ -1,33 +0,0 @@
|
||||
---
|
||||
name: General issues
|
||||
about: Installation, packages, dependencies, "nothing works", test suite failures...
|
||||
title: "[BUG]"
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**Describe the bug**
|
||||
What's the problem?
|
||||
|
||||
**To Reproduce**
|
||||
Steps to reproduce the behavior.
|
||||
|
||||
**Expected behavior**
|
||||
What did you expected to happen?
|
||||
|
||||
**Screenshots**
|
||||
If applicable, add screenshots to help explain your problem.
|
||||
|
||||
**System (please complete the following information):**
|
||||
- OS:
|
||||
- Python version:
|
||||
- OCRmyPDF version:
|
||||
- Platform: x64 or ARM
|
||||
|
||||
**Installation**
|
||||
How did you install OCRmyPDF? Did you install it from your operating system's
|
||||
package manager, or using pip?
|
||||
|
||||
**Additional context**
|
||||
Add any other context about the problem here.
|
||||
@@ -1,40 +0,0 @@
|
||||
---
|
||||
name: Problem with specific file
|
||||
about: Something went wrong while trying to OCR a specific file
|
||||
title: "[BUG]"
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
**To Reproduce**
|
||||
What command line or API call were you trying to run?
|
||||
|
||||
```bash
|
||||
ocrmypdf ...arguments... input.pdf output.pdf
|
||||
```
|
||||
|
||||
Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful.
|
||||
|
||||
**Example file**
|
||||
If your issue is a problem that affects only certain files, and we will require an input file (PDF or image) that demonstrates your issue.
|
||||
|
||||
Please provide an input file with no personal or confidential information. At your option you may [GPG-encrypt the file](https://github.com/ocrmypdf/OCRmyPDF/wiki) for OCRmyPDF's author only.
|
||||
|
||||
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
||||
|
||||
*(Issues without example files usually cannot be resolved. It's like reporting an issue against a web browser without providing a URL.)*
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Screenshots**
|
||||
If applicable, add screenshots to help explain your problem.
|
||||
|
||||
**System**
|
||||
- OS: [e.g. Linux, Windows, macOS]
|
||||
- OCRmyPDF Version: ``ocrmypdf --version``
|
||||
- How did you install ocrmypdf? Did you use a system package manager, `pip`, or a Docker image?
|
||||
+169
-85
@@ -5,7 +5,7 @@ name: Test and deploy
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
- main
|
||||
- ci
|
||||
- release/*
|
||||
- feature/*
|
||||
@@ -21,53 +21,48 @@ jobs:
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
os: [ubuntu-22.04, ubuntu-24.04]
|
||||
python: ["3.11", "3.12", "3.13", "3.14"]
|
||||
include:
|
||||
- os: ubuntu-20.04
|
||||
python: "3.8"
|
||||
- os: ubuntu-20.04
|
||||
python: "3.9"
|
||||
- os: ubuntu-20.04
|
||||
python: "3.10"
|
||||
- os: ubuntu-latest
|
||||
python: "3.9"
|
||||
- os: ubuntu-latest
|
||||
python: "3.10"
|
||||
- os: ubuntu-latest
|
||||
- os: ubuntu-22.04
|
||||
tesseract_ppa: "ppa"
|
||||
python: "3.11"
|
||||
# - os: ubuntu-latest
|
||||
# python: "pypy3.8"
|
||||
#- os: ubuntu-latest
|
||||
# python: "pypy3.9"
|
||||
- os: ubuntu-latest
|
||||
python: "3.9"
|
||||
tesseract5: true
|
||||
|
||||
env:
|
||||
OS: ${{ matrix.os }}
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
- uses: actions/setup-python@v4
|
||||
name: Install Python
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
- name: Install Tesseract 5
|
||||
if: matrix.tesseract5
|
||||
- name: Install Tesseract from PPA
|
||||
if: matrix.tesseract_ppa == 'ppa'
|
||||
run: |
|
||||
sudo add-apt-repository ppa:alex-p/tesseract-ocr-devel
|
||||
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr5
|
||||
|
||||
- name: Install common packages
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
curl \
|
||||
fonts-noto-core \
|
||||
fonts-noto-cjk \
|
||||
ghostscript \
|
||||
jbig2dec \
|
||||
img2pdf \
|
||||
libexempi8 \
|
||||
libffi-dev \
|
||||
libsm6 libxext6 libxrender-dev \
|
||||
pngquant \
|
||||
@@ -79,24 +74,9 @@ jobs:
|
||||
unpaper \
|
||||
zlib1g
|
||||
|
||||
- name: Install Ubuntu 20.04 packages
|
||||
if: matrix.os == 'ubuntu-20.04' || matrix.os == 'ubuntu-latest'
|
||||
run: |
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
libexempi8
|
||||
|
||||
- name: Install Ubuntu packages for PyPy
|
||||
if: startsWith(matrix.python, 'pypy')
|
||||
run: |
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
libxml2-dev \
|
||||
libxslt1-dev \
|
||||
pypy3-dev
|
||||
|
||||
- name: Install Python packages
|
||||
run: |
|
||||
python -m pip install --upgrade pip wheel
|
||||
python -m pip install --prefer-binary .[test]
|
||||
uv sync --group test
|
||||
|
||||
- name: Report versions
|
||||
run: |
|
||||
@@ -104,14 +84,16 @@ jobs:
|
||||
gs --version
|
||||
pngquant --version
|
||||
unpaper --version
|
||||
img2pdf --version
|
||||
uv run --no-dev img2pdf --version
|
||||
|
||||
- name: Test
|
||||
run: |
|
||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v3
|
||||
uses: codecov/codecov-action@v5
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
files: ./coverage.xml
|
||||
env_vars: OS,PYTHON
|
||||
@@ -122,14 +104,14 @@ jobs:
|
||||
strategy:
|
||||
matrix:
|
||||
os: [macos-latest]
|
||||
python: ["3.10", "3.11"]
|
||||
python: ["3.11", "3.12", "3.13", "3.14"]
|
||||
|
||||
env:
|
||||
OS: ${{ matrix.os }}
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
@@ -143,31 +125,39 @@ jobs:
|
||||
jbig2enc \
|
||||
openjpeg \
|
||||
pngquant \
|
||||
tesseract
|
||||
poppler \
|
||||
tesseract \
|
||||
verapdf
|
||||
|
||||
- uses: actions/setup-python@v4
|
||||
name: Install Python
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
- name: Install Python packages
|
||||
run: |
|
||||
python -m pip install --upgrade pip wheel
|
||||
python -m pip install --prefer-binary .[test]
|
||||
uv sync --group test
|
||||
|
||||
- name: Report versions
|
||||
run: |
|
||||
tesseract --version
|
||||
gs --version
|
||||
pngquant --version
|
||||
img2pdf --version
|
||||
uv run --no-dev img2pdf --version
|
||||
|
||||
- name: Test
|
||||
run: |
|
||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v3
|
||||
uses: codecov/codecov-action@v5
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
files: ./coverage.xml
|
||||
env_vars: OS,PYTHON
|
||||
@@ -178,38 +168,45 @@ jobs:
|
||||
strategy:
|
||||
matrix:
|
||||
os: [windows-latest]
|
||||
python: ["3.10", "3.11"]
|
||||
python: ["3.11", "3.12", "3.13", "3.14"]
|
||||
|
||||
env:
|
||||
OS: ${{ matrix.os }}
|
||||
PYTHON: ${{ matrix.python }}
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
- uses: actions/setup-python@v4
|
||||
name: Install Python
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: "0.9.x"
|
||||
|
||||
- name: "Set up Python"
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
- name: Install system packages
|
||||
run: |
|
||||
choco install --yes --no-progress --pre tesseract
|
||||
choco install --yes --no-progress --ignore-checksums ghostscript
|
||||
choco install --yes --no-progress tesseract
|
||||
choco install --yes --no-progress --ignore-checksums ghostscript --version 9.56.1
|
||||
choco install --yes --no-progress poppler --version=25.11.0
|
||||
|
||||
- name: Install Python packages
|
||||
run: |
|
||||
python -m pip install --upgrade pip wheel
|
||||
python -m pip install --prefer-binary .[test]
|
||||
uv sync --group test
|
||||
|
||||
- name: Test
|
||||
run: |
|
||||
python -m pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
uv run --no-dev pytest --cov-report xml --cov=ocrmypdf --cov=tests/ -n0 tests/
|
||||
|
||||
- name: Upload coverage to Codecov
|
||||
uses: codecov/codecov-action@v3
|
||||
uses: codecov/codecov-action@v5
|
||||
env:
|
||||
CODECOV_TOKEN: ${{ secrets.CODECOV_TOKEN }}
|
||||
with:
|
||||
files: ./coverage.xml
|
||||
env_vars: OS,PYTHON
|
||||
@@ -218,22 +215,22 @@ jobs:
|
||||
name: Build sdist and wheels
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
- uses: actions/setup-python@v4
|
||||
name: Install Python
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
python-version: "3.7"
|
||||
version: "0.9.x"
|
||||
|
||||
- name: Make wheels and sdist
|
||||
run: |
|
||||
python -m pip install --upgrade pip wheel build
|
||||
python -m build --sdist --wheel
|
||||
uv build --sdist --wheel
|
||||
|
||||
- uses: actions/upload-artifact@v3
|
||||
- uses: actions/upload-artifact@v6
|
||||
with:
|
||||
name: artifact
|
||||
path: |
|
||||
./dist/*.whl
|
||||
./dist/*.tar.gz
|
||||
@@ -242,21 +239,63 @@ jobs:
|
||||
name: Deploy artifacts to PyPI
|
||||
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||
runs-on: ubuntu-latest
|
||||
environment: release
|
||||
permissions:
|
||||
id-token: write # mandatory for PyPI publishing
|
||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||
steps:
|
||||
- uses: actions/download-artifact@v3
|
||||
- uses: actions/download-artifact@v7
|
||||
with:
|
||||
name: artifact
|
||||
path: dist
|
||||
|
||||
- uses: pypa/gh-action-pypi-publish@master
|
||||
with:
|
||||
user: __token__
|
||||
password: ${{ secrets.TOKEN_PYPI }}
|
||||
# repository_url: https://test.pypi.org/legacy/
|
||||
- name: Publish to PyPI
|
||||
uses: pypa/gh-action-pypi-publish@release/v1
|
||||
|
||||
docker:
|
||||
name: Build Docker images
|
||||
create_release:
|
||||
name: Create GitHub release
|
||||
needs: [upload_pypi]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name == 'push' && startsWith(github.event.ref, 'refs/tags/v')
|
||||
permissions:
|
||||
# Required to create a release
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/download-artifact@v7
|
||||
with:
|
||||
name: artifact
|
||||
path: dist
|
||||
|
||||
- name: Sign the dists with Sigstore
|
||||
uses: sigstore/gh-action-sigstore-python@v3.2.0
|
||||
with:
|
||||
inputs: |
|
||||
./dist/*.tar.gz
|
||||
./dist/*.whl
|
||||
|
||||
- name: Create GitHub Release
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: >-
|
||||
gh release create
|
||||
"$GITHUB_REF_NAME"
|
||||
--repo "$GITHUB_REPOSITORY"
|
||||
--notes ""
|
||||
|
||||
- name: Upload artifact signatures to GitHub Release
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
# Upload to GitHub Release using the `gh` CLI.
|
||||
# `dist/` contains the built packages, and the
|
||||
# sigstore-produced signatures and certificates.
|
||||
run: >-
|
||||
gh release upload
|
||||
"$GITHUB_REF_NAME" dist/**
|
||||
--repo "$GITHUB_REPOSITORY"
|
||||
|
||||
docker_ubuntu:
|
||||
name: Build Ubuntu-based Docker image
|
||||
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name != 'pull_request'
|
||||
@@ -264,9 +303,9 @@ jobs:
|
||||
- name: Set image tag to release or branch
|
||||
run: echo "DOCKER_IMAGE_TAG=${GITHUB_REF##*/}" >> $GITHUB_ENV
|
||||
|
||||
- name: If master, set to latest
|
||||
- name: If main, set to latest
|
||||
run: echo 'DOCKER_IMAGE_TAG=latest' >> $GITHUB_ENV
|
||||
if: env.DOCKER_IMAGE_TAG == 'master'
|
||||
if: env.DOCKER_IMAGE_TAG == 'main'
|
||||
|
||||
- name: Set Docker Hub repository to username
|
||||
run: echo "DOCKER_REPOSITORY=jbarlow83" >> $GITHUB_ENV
|
||||
@@ -274,22 +313,22 @@ jobs:
|
||||
- name: Set image name
|
||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf" >> $GITHUB_ENV
|
||||
|
||||
- uses: actions/checkout@v3
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v2
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
username: jbarlow83
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v2
|
||||
uses: docker/setup-qemu-action@v3
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@v2
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Print image tag
|
||||
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
||||
@@ -300,4 +339,49 @@ jobs:
|
||||
--push \
|
||||
--platform linux/arm64/v8,linux/amd64 \
|
||||
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \
|
||||
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}-ubuntu:${DOCKER_IMAGE_TAG}" \
|
||||
--file .docker/Dockerfile .
|
||||
|
||||
docker_alpine:
|
||||
name: Build Alpine-based Docker images
|
||||
needs: [wheel_sdist_linux, test_linux, test_macos, test_windows]
|
||||
runs-on: ubuntu-latest
|
||||
if: github.event_name != 'pull_request'
|
||||
steps:
|
||||
- name: Set image tag to release or branch
|
||||
run: echo "DOCKER_IMAGE_TAG=${GITHUB_REF##*/}" >> $GITHUB_ENV
|
||||
|
||||
- name: If main, set to latest
|
||||
run: echo 'DOCKER_IMAGE_TAG=latest' >> $GITHUB_ENV
|
||||
if: env.DOCKER_IMAGE_TAG == 'main'
|
||||
|
||||
- name: Set Docker Hub repository to username
|
||||
run: echo "DOCKER_REPOSITORY=jbarlow83" >> $GITHUB_ENV
|
||||
|
||||
- name: Set image name
|
||||
run: echo "DOCKER_IMAGE_NAME=ocrmypdf-alpine" >> $GITHUB_ENV
|
||||
|
||||
- uses: actions/checkout@v6
|
||||
with:
|
||||
fetch-depth: "0" # 0=all, needed for setuptools-scm to resolve version tags
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
username: jbarlow83
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Print image tag
|
||||
run: echo "Building image ${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}"
|
||||
|
||||
- name: Build
|
||||
run: |
|
||||
docker buildx build \
|
||||
--push \
|
||||
--platform linux/amd64,linux/arm64 \
|
||||
--tag "${DOCKER_REPOSITORY}/${DOCKER_IMAGE_NAME}:${DOCKER_IMAGE_TAG}" \
|
||||
--file .docker/Dockerfile.alpine .
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
name: Remove Triage Label on Reply
|
||||
|
||||
on:
|
||||
issue_comment:
|
||||
types:
|
||||
- created
|
||||
|
||||
jobs:
|
||||
remove-triage-label:
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- name: Check if comment is by the repository owner
|
||||
id: check_comment
|
||||
run: |
|
||||
echo "::set-output name=is_owner::$(
|
||||
if [[ '${{ github.event.comment.user.login }}' == 'jbarlow83' ]]; then
|
||||
echo 'true';
|
||||
else
|
||||
echo 'false';
|
||||
fi
|
||||
)"
|
||||
|
||||
- name: Remove 'triage' label
|
||||
if: ${{ steps.check_comment.outputs.is_owner == 'true' }}
|
||||
uses: actions-ecosystem/action-remove-labels@v1
|
||||
with:
|
||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
||||
labels: triage
|
||||
@@ -6,6 +6,7 @@
|
||||
.venv*/
|
||||
.tox/
|
||||
.vscode/
|
||||
.hypothesis/
|
||||
.ipynb_checkpoints/
|
||||
.mypy_cache/
|
||||
.pytest_cache/
|
||||
@@ -26,6 +27,7 @@ venv*/
|
||||
*.traineddata
|
||||
/private
|
||||
/coverage.xml
|
||||
/issuepdf
|
||||
|
||||
# Package building
|
||||
*.egg-info/
|
||||
@@ -42,3 +44,8 @@ docs/_build/
|
||||
docs/_static/
|
||||
docs/_templates/
|
||||
docs/Makefile
|
||||
src/ocrmypdf/_version.py
|
||||
|
||||
.idea/
|
||||
.aider*
|
||||
CLAUDE.md
|
||||
|
||||
+7
-20
@@ -3,34 +3,21 @@
|
||||
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v4.3.0
|
||||
rev: v4.4.0
|
||||
hooks:
|
||||
- id: check-case-conflict
|
||||
- id: check-merge-conflict
|
||||
- id: check-toml
|
||||
- id: check-yaml
|
||||
- id: debug-statements
|
||||
- repo: https://github.com/pycqa/isort
|
||||
rev: 5.10.1
|
||||
- repo: https://github.com/astral-sh/ruff-pre-commit
|
||||
rev: "v0.14.11"
|
||||
hooks:
|
||||
- id: isort
|
||||
args: ["--profile", "black", "-a", "from __future__ import annotations"]
|
||||
- repo: https://github.com/psf/black
|
||||
rev: 22.6.0
|
||||
hooks:
|
||||
- id: black
|
||||
language_version: python
|
||||
- repo: https://github.com/asottile/setup-cfg-fmt
|
||||
rev: v1.20.2
|
||||
hooks:
|
||||
- id: setup-cfg-fmt
|
||||
- repo: https://github.com/asottile/pyupgrade
|
||||
rev: v2.37.2
|
||||
hooks:
|
||||
- id: pyupgrade
|
||||
args: ["--py38-plus"]
|
||||
- id: ruff-check
|
||||
args: [--fix]
|
||||
- id: ruff-format
|
||||
- repo: https://github.com/pre-commit/mirrors-mypy
|
||||
rev: v0.971
|
||||
rev: v1.2.0
|
||||
hooks:
|
||||
- id: mypy
|
||||
additional_dependencies:
|
||||
|
||||
+5
-5
@@ -11,13 +11,13 @@ version: 2
|
||||
sphinx:
|
||||
configuration: docs/conf.py
|
||||
|
||||
# Optionally build your docs in additional formats such as PDF
|
||||
formats:
|
||||
- pdf
|
||||
|
||||
# Optionally set the version of Python and requirements required to build your docs
|
||||
build:
|
||||
os: ubuntu-22.04
|
||||
tools:
|
||||
python: "3.11"
|
||||
|
||||
python:
|
||||
version: "3.8"
|
||||
install:
|
||||
- method: pip
|
||||
path: .
|
||||
|
||||
-132
@@ -1,132 +0,0 @@
|
||||
Format: https://www.debian.org/doc/packaging-manuals/copyright-format/1.0/
|
||||
Upstream-Name: OCRmyPDF
|
||||
Upstream-Contact: James R. Barlow <james@purplerock.ca>
|
||||
Source: https://github.com/ocrmypdf/OCRmyPDF
|
||||
|
||||
|
||||
Files:
|
||||
.git_archival.txt
|
||||
docs/images/logo-social.png
|
||||
docs/images/logo-square-256.svg
|
||||
docs/images/logo-square.png
|
||||
docs/images/logo-square.svg
|
||||
docs/images/logo.svg
|
||||
setup.cfg
|
||||
Copyright: (C) 2022 James R. Barlow
|
||||
License: MPL-2.0
|
||||
|
||||
Files:
|
||||
.github/ISSUE_TEMPLATE/*.md
|
||||
docs/images/macos-workflow.png
|
||||
Copyright: (C) 2022 James R. Barlow
|
||||
License: CC-BY-SA-4.0
|
||||
|
||||
Files:
|
||||
tests/resources/acroform.pdf
|
||||
tests/resources/aspect.pdf
|
||||
tests/resources/blank.pdf
|
||||
tests/resources/cmyk.pdf
|
||||
tests/resources/crom.png
|
||||
tests/resources/enormous.pdf
|
||||
tests/resources/formxobject.pdf
|
||||
tests/resources/francais.pdf
|
||||
tests/resources/hugemono.pdf
|
||||
tests/resources/invalid.pdf
|
||||
tests/resources/kcs.pdf
|
||||
tests/resources/livecycle.pdf
|
||||
tests/resources/missing_docinfo.pdf
|
||||
tests/resources/negzero.pdf
|
||||
tests/resources/no_contents.pdf
|
||||
tests/resources/toc.pdf
|
||||
tests/resources/trivial.pdf
|
||||
tests/resources/truetype_font_nomapping.pdf
|
||||
tests/resources/type3_font_nomapping.pdf
|
||||
Copyright: (C) 2022 James R. Barlow
|
||||
License: CC-BY-SA-4.0
|
||||
|
||||
Files:
|
||||
tests/resources/graph.pdf
|
||||
tests/resources/graph_ocred.pdf
|
||||
Copyright: (C) 2012 SmokeyJoe
|
||||
License: GFDL-1.2-or-later or CC-BY-SA-3.0
|
||||
|
||||
Files: tests/resources/c02-22.pdf
|
||||
tests/resources/congress.jpg
|
||||
tests/resources/multipage.pdf
|
||||
Copyright: Public domain
|
||||
License: public-domain
|
||||
Copyright on these files has expired.
|
||||
|
||||
Files: docs/images/bitmap_vs_svg.svg
|
||||
Copyright: (C) 2006 Yug
|
||||
License: CC-BY-SA-2.5
|
||||
|
||||
Files: tests/cache/*
|
||||
Copyright: (C) 2022 James R. Barlow
|
||||
License: CC-BY-SA-4.0
|
||||
|
||||
Files: tests/resources/linn.png
|
||||
tests/resources/linn.pdf
|
||||
tests/resources/linn.txt
|
||||
tests/resources/ccitt.pdf
|
||||
tests/resources/cardinal.pdf
|
||||
tests/resources/jbig2.pdf
|
||||
tests/resources/skew.pdf
|
||||
tests/resources/rotated_skew.pdf
|
||||
tests/resources/poster.pdf
|
||||
Copyright: (C) 1985 Forat Electronics
|
||||
License: GFDL-1.2-or-later or CC-BY-SA-3.0
|
||||
|
||||
Files: tests/resources/lichtenstein.pdf
|
||||
Copyright: (C) 2001 Andreas Tille
|
||||
(C) 2007 Alessio Damato
|
||||
License: GFDL-1.2-or-later or CC-BY-SA-3.0
|
||||
|
||||
Files: tests/resources/masks.pdf
|
||||
Copyright: held by the contributors to the German Wikipedia article "Linux"
|
||||
see: https://de.wikipedia.org/w/index.php?title=Linux&action=history
|
||||
(masks.pdf generated from Wikipedia article as of 2016-08-24)
|
||||
License: CC-BY-SA-3.0
|
||||
|
||||
Files: tests/resources/epson.pdf
|
||||
Copyright: held by the contributors to the Wikipedia article "Optical character recognition"
|
||||
see: https://en.wikipedia.org/w/index.php?title=Optical_character_recognition&action=history
|
||||
(epson.pdf generated from Wikipedia article as of 2016-09-14)
|
||||
License: CC-BY-SA-3.0
|
||||
|
||||
Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf
|
||||
Copyright: (C) 2005 Ellywa
|
||||
License: GFDL-1.2-or-later or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0
|
||||
Comment:
|
||||
Obtained from: https://commons.wikimedia.org/wiki/File:Triumph.typewriter_text_Linzensoep.gif
|
||||
|
||||
Files: tests/resources/overlay.pdf
|
||||
Copyright: (C) 2017 Max Anderson
|
||||
License: MIT
|
||||
|
||||
Files:
|
||||
tests/resources/baiona*.png
|
||||
tests/resources/baiona*.jpg
|
||||
tests/resources/link.pdf
|
||||
tests/resources/palette.pdf
|
||||
Copyright: (C) 2014 Euskaldunaa
|
||||
License: CC-BY-SA-4.0
|
||||
|
||||
Files: tests/resources/vector.pdf
|
||||
Copyright: (C) 2018 Catscratch
|
||||
License: MIT
|
||||
|
||||
Files: src/ocrmypdf/data/sRGB.icc
|
||||
Copyright: Kai-Uwe Behrmann <www.behrmann.name>
|
||||
Marti Maria <www.littlecms.com>
|
||||
Photogamut <www.photogamut.org>
|
||||
Graeme Gill <www.argyllcms.com>
|
||||
ColorSolutions <www.basICColor.com>
|
||||
License: Zlib
|
||||
|
||||
Files: tests/resources/3small.pdf
|
||||
Copyright: (C) 2014 Euskaldunaa
|
||||
(C) 2017 James R. Barlow
|
||||
(C) 2005 Ellywa
|
||||
License: CC-BY-SA-4.0 and (GFDL-1.2-or-later or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0)
|
||||
Comment: concatenation of baiona_gray.png, crom.png and typewriter.png/2400dpi.pdf
|
||||
@@ -0,0 +1,73 @@
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction, and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all other entities that control, are controlled by, or are under common control with that entity. For the purposes of this definition, "control" means (i) the power, direct or indirect, to cause the direction or management of such entity, whether by contract or otherwise, or (ii) ownership of fifty percent (50%) or more of the outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications, including but not limited to software source code, documentation source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or Object form, made available under the License, as indicated by a copyright notice that is included in or attached to the work (an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object form, that is based on (or derived from) the Work and for which the editorial revisions, annotations, elaborations, or other modifications represent, as a whole, an original work of authorship. For the purposes of this License, Derivative Works shall not include works that remain separable from, or merely link (or bind by name) to the interfaces of, the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including the original version of the Work and any modifications or additions to that Work or Derivative Works thereof, that is intentionally submitted to Licensor for inclusion in the Work by the copyright owner or by an individual or Legal Entity authorized to submit on behalf of the copyright owner. For the purposes of this definition, "submitted" means any form of electronic, verbal, or written communication sent to the Licensor or its representatives, including but not limited to communication on electronic mailing lists, source code control systems, and issue tracking systems that are managed by, or on behalf of, the Licensor for the purpose of discussing and improving the Work, but excluding communication that is conspicuously marked or otherwise designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity on behalf of whom a Contribution has been received by Licensor and subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable copyright license to reproduce, prepare Derivative Works of, publicly display, publicly perform, sublicense, and distribute the Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable (except as stated in this section) patent license to make, have made, use, offer to sell, sell, import, and otherwise transfer the Work, where such license applies only to those patent claims licensable by such Contributor that are necessarily infringed by their Contribution(s) alone or by combination of their Contribution(s) with the Work to which such Contribution(s) was submitted. If You institute patent litigation against any entity (including a cross-claim or counterclaim in a lawsuit) alleging that the Work or a Contribution incorporated within the Work constitutes direct or contributory patent infringement, then any patent licenses granted to You under this License for that Work shall terminate as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the Work or Derivative Works thereof in any medium, with or without modifications, and in Source or Object form, provided that You meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works that You distribute, all copyright, patent, trademark, and attribution notices from the Source form of the Work, excluding those notices that do not pertain to any part of the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its distribution, then any Derivative Works that You distribute must include a readable copy of the attribution notices contained within such NOTICE file, excluding those notices that do not pertain to any part of the Derivative Works, in at least one of the following places: within a NOTICE text file distributed as part of the Derivative Works; within the Source form or documentation, if provided along with the Derivative Works; or, within a display generated by the Derivative Works, if and wherever such third-party notices normally appear. The contents of the NOTICE file are for informational purposes only and do not modify the License. You may add Your own attribution notices within Derivative Works that You distribute, alongside or as an addendum to the NOTICE text from the Work, provided that such additional attribution notices cannot be construed as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of Your modifications, or for any such Derivative Works as a whole, provided Your use, reproduction, and distribution of the Work otherwise complies with the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise, any Contribution intentionally submitted for inclusion in the Work by You to the Licensor shall be under the terms and conditions of this License, without any additional terms or conditions. Notwithstanding the above, nothing herein shall supersede or modify the terms of any separate license agreement you may have executed with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade names, trademarks, service marks, or product names of the Licensor, except as required for reasonable and customary use in describing the origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or agreed to in writing, Licensor provides the Work (and each Contributor provides its Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied, including, without limitation, any warranties or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A PARTICULAR PURPOSE. You are solely responsible for determining the appropriateness of using or redistributing the Work and assume any risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory, whether in tort (including negligence), contract, or otherwise, unless required by applicable law (such as deliberate and grossly negligent acts) or agreed to in writing, shall any Contributor be liable to You for damages, including any direct, indirect, special, incidental, or consequential damages of any character arising as a result of this License or out of the use or inability to use the Work (including but not limited to damages for loss of goodwill, work stoppage, computer failure or malfunction, or any and all other commercial damages or losses), even if such Contributor has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing the Work or Derivative Works thereof, You may choose to offer, and charge a fee for, acceptance of support, warranty, indemnity, or other liability obligations and/or rights consistent with this License. However, in accepting such obligations, You may act only on Your own behalf and on Your sole responsibility, not on behalf of any other Contributor, and only if You agree to indemnify, defend, and hold each Contributor harmless for any liability incurred by, or claims asserted against, such Contributor by reason of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following boilerplate notice, with the fields enclosed by brackets "[]" replaced with your own identifying information. (Don't include the brackets!) The text should be enclosed in the appropriate comment syntax for the file format. We also recommend that a file or class name and description of purpose be included on the same "printed page" as the copyright notice for easier identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -39,8 +39,10 @@ ocrmypdf # it's a scriptable command line program
|
||||
- Distributes work across all available CPU cores
|
||||
- Uses [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) engine to recognize more than [100 languages](https://github.com/tesseract-ocr/tessdata)
|
||||
- Keeps your private data private.
|
||||
- Scales properly to handle files with thousands of pages
|
||||
- Battle-tested on millions of PDFs
|
||||
- Scales properly to handle files with thousands of pages.
|
||||
- Battle-tested on millions of PDFs.
|
||||
|
||||
<img src="misc/screencast/demo.svg" alt="Demo of OCRmyPDF in a terminal session">
|
||||
|
||||
For details: please consult the [documentation](https://ocrmypdf.readthedocs.io/en/latest/).
|
||||
|
||||
@@ -68,10 +70,11 @@ Linux, Windows, macOS and FreeBSD are supported. Docker images are also availabl
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| Conda | ``conda install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| OpenBSD | ``pkg_add ocrmypdf`` |
|
||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||
|
||||
For everyone else, [see our documentation](https://ocrmypdf.readthedocs.io/en/latest/installation.html) for installation steps.
|
||||
@@ -90,6 +93,10 @@ apt-get install tesseract-ocr-chi-sim # Example: Install Chinese Simplified lan
|
||||
# Arch Linux users
|
||||
pacman -S tesseract-data-eng tesseract-data-deu # Example: Install the English and German language packs
|
||||
|
||||
# OpenBSD users
|
||||
pkg_info -aQ tesseract # Display a list of all Tesseract language packs
|
||||
pkg_add tesseract-cym # Example: Install the Welsh language pack
|
||||
|
||||
# brew macOS users
|
||||
brew install tesseract-lang
|
||||
```
|
||||
@@ -110,9 +117,43 @@ Our [documentation is served on Read the Docs](https://ocrmypdf.readthedocs.io/e
|
||||
|
||||
Please report issues on our [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) page, and follow the issue template for quick response.
|
||||
|
||||
## Feature demo
|
||||
|
||||
```bash
|
||||
# Add an OCR layer and require PDF/A
|
||||
ocrmypdf --output-type pdfa input.pdf output.pdf
|
||||
|
||||
# Convert an image to single page PDF
|
||||
ocrmypdf input.jpg output.pdf
|
||||
|
||||
# Add OCR to a file in place (only modifies file on success)
|
||||
ocrmypdf myfile.pdf myfile.pdf
|
||||
|
||||
# OCR with non-English languages (look up your language's ISO 639-3 code)
|
||||
ocrmypdf -l fra LeParisien.pdf LeParisien.pdf
|
||||
|
||||
# OCR multilingual documents
|
||||
ocrmypdf -l eng+fra Bilingual-English-French.pdf Bilingual-English-French.pdf
|
||||
|
||||
# Deskew (straighten crooked pages)
|
||||
ocrmypdf --deskew input.pdf output.pdf
|
||||
```
|
||||
|
||||
For more features, see the [documentation](https://ocrmypdf.readthedocs.io/en/latest/index.html).
|
||||
|
||||
## Requirements
|
||||
|
||||
In addition to the required Python version (3.7+), OCRmyPDF requires external program installations of Ghostscript and Tesseract OCR. OCRmyPDF is pure Python, and runs on pretty much everything: Linux, macOS, Windows and FreeBSD.
|
||||
In addition to the required Python version, OCRmyPDF requires external program installations of Ghostscript and Tesseract OCR. OCRmyPDF is pure Python, and runs on pretty much everything: Linux, macOS, Windows and FreeBSD.
|
||||
|
||||
## Plugins
|
||||
|
||||
OCRmyPDF provides a plugin interface allowing its capabilities to be extended or replaced. Here are some plugins we are aware of:
|
||||
|
||||
- [OCRmyPDF-AppleOCR](https://github.com/mkyt/ocrmypdf-AppleOCR): replaces the standard Tesseract OCR engine with Apple Vision Framework. Requires macOS.
|
||||
- [OCRmyPDF-EasyOCR](https://github.com/ocrmypdf/OCRmyPDF-EasyOCR): replaces the standard Tesseract OCR engine with EasyOCR, a newer OCR engine based on PyTorch. GPU strongly recommended.
|
||||
- [OCRmyPDF-PaddleOCR](https://github.com/clefru/ocrmypdf-paddleocr): replaces the standard Tesseract OCR engine with PaddleOCR, a powerful GPU accelerated OCR engine.
|
||||
|
||||
[paperless-ngx](https://docs.paperless-ngx.com/) provides integration of OCRmyPDF into a searchable document management system.
|
||||
|
||||
## Press & Media
|
||||
|
||||
@@ -123,6 +164,7 @@ In addition to the required Python version (3.7+), OCRmyPDF requires external pr
|
||||
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [Y Combinator discussion](https://news.ycombinator.com/item?id=32028752)
|
||||
|
||||
## Business enquiries
|
||||
|
||||
|
||||
+173
@@ -0,0 +1,173 @@
|
||||
# OCRmyPDF 中文说明
|
||||
|
||||
我将为您创建一个中文版的 README.md 文件。以下是基于原始英文版本的中文翻译:
|
||||
|
||||
```markdown:/Users/huapai/PycharmProjects/OCRmyPDF/README.md
|
||||
<!-- SPDX-FileCopyrightText: 2014 Julien Pfefferkorn -->
|
||||
<!-- SPDX-FileCopyrightText: 2015 James R. Barlow -->
|
||||
<!-- SPDX-License-Identifier: CC-BY-SA-4.0 -->
|
||||
|
||||
<img src="docs/images/logo.svg" width="240" alt="OCRmyPDF">
|
||||
|
||||
[](https://github.com/ocrmypdf/OCRmyPDF/actions/workflows/build.yml) [![PyPI 版本][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew 版本][homebrew] ![ReadTheDocs][docs] ![Python 版本][pyversions]
|
||||
|
||||
[pypi]: https://img.shields.io/pypi/v/ocrmypdf.svg "PyPI 版本"
|
||||
[homebrew]: https://img.shields.io/homebrew/v/ocrmypdf.svg "Homebrew 版本"
|
||||
[docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD"
|
||||
[pyversions]: https://img.shields.io/pypi/pyversions/ocrmypdf "支持的 Python 版本"
|
||||
|
||||
OCRmyPDF 为扫描的 PDF 文件添加 OCR 文本层,使其可以被搜索或复制粘贴。
|
||||
|
||||
```bash
|
||||
ocrmypdf # 这是一个可脚本化的命令行程序
|
||||
-l eng+fra # 支持多种语言
|
||||
--rotate-pages # 可以修正旋转错误的页面
|
||||
--deskew # 可以校正倾斜的 PDF!
|
||||
--title "My PDF" # 可以更改输出元数据
|
||||
--jobs 4 # 默认使用多核心处理
|
||||
--output-type pdfa # 默认生成 PDF/A 格式
|
||||
input_scanned.pdf # 接受 PDF 输入(或图像)
|
||||
output_searchable.pdf # 生成经过验证的 PDF 输出
|
||||
```
|
||||
|
||||
[查看发布说明了解最新变更的详情](https://ocrmypdf.readthedocs.io/en/latest/release_notes.html)。
|
||||
|
||||
## 主要特点
|
||||
|
||||
- 从普通 PDF 生成可搜索的 [PDF/A](https://en.wikipedia.org/?title=PDF/A) 文件
|
||||
- 准确地将 OCR 文本放置在图像下方,便于复制/粘贴
|
||||
- 保持原始嵌入图像的精确分辨率
|
||||
- 在可能的情况下,以"无损"操作方式插入 OCR 信息,不破坏任何其他内容
|
||||
- 优化 PDF 图像,通常生成比输入文件更小的文件
|
||||
- 如果需要,在执行 OCR 前对图像进行校正和/或清理
|
||||
- 验证输入和输出文件
|
||||
- 在所有可用的 CPU 核心上分配工作
|
||||
- 使用 [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) 引擎识别超过 [100 种语言](https://github.com/tesseract-ocr/tessdata)
|
||||
- 保护您的私人数据安全
|
||||
- 适当扩展以处理包含数千页的文件
|
||||
- 在数百万 PDF 上经过实战测试
|
||||
|
||||
<img src="misc/screencast/demo.svg" alt="终端会话中的 OCRmyPDF 演示">
|
||||
|
||||
详情请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/)。
|
||||
|
||||
## 开发动机
|
||||
|
||||
我在网上搜索免费的命令行工具来对 PDF 文件进行 OCR:我找到了很多,但没有一个真正令人满意:
|
||||
|
||||
- 要么它们生成的 PDF 文件中文本位置错误(使复制/粘贴变得不可能)
|
||||
- 要么它们不处理重音和多语言字符
|
||||
- 要么它们改变了嵌入图像的分辨率
|
||||
- 要么它们生成了体积巨大的 PDF 文件
|
||||
- 要么它们在尝试 OCR 时崩溃
|
||||
- 要么它们不生成有效的 PDF 文件
|
||||
- 最重要的是,它们都不生成 PDF/A 文件(专为长期存储设计的格式)
|
||||
|
||||
...所以我决定开发自己的工具。
|
||||
|
||||
## 安装
|
||||
|
||||
支持 Linux、Windows、macOS 和 FreeBSD。Docker 镜像也可用,同时支持 x64 和 ARM。
|
||||
|
||||
| 操作系统 | 安装命令 |
|
||||
| --------------------------- | ----------------------------- |
|
||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
| macOS (Homebrew) | ``brew install ocrmypdf`` |
|
||||
| macOS (MacPorts) | ``port install ocrmypdf`` |
|
||||
| macOS (nix) | ``nix-env -i ocrmypdf`` |
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
| FreeBSD | ``pkg install py-ocrmypdf`` |
|
||||
| Ubuntu Snap | ``snap install ocrmypdf`` |
|
||||
|
||||
对于其他用户,[请参阅我们的文档](https://ocrmypdf.readthedocs.io/en/latest/installation.html)了解安装步骤。
|
||||
|
||||
## 语言
|
||||
|
||||
OCRmyPDF 使用 Tesseract 进行 OCR,并依赖其语言包。对于 Linux 用户,您通常可以找到提供语言包的软件包:
|
||||
|
||||
```bash
|
||||
# 显示所有 Tesseract 语言包的列表
|
||||
apt-cache search tesseract-ocr
|
||||
|
||||
# Debian/Ubuntu 用户
|
||||
apt-get install tesseract-ocr-chi-sim # 示例:安装中文简体语言包
|
||||
|
||||
# Arch Linux 用户
|
||||
pacman -S tesseract-data-eng tesseract-data-deu # 示例:安装英语和德语语言包
|
||||
|
||||
# brew macOS 用户
|
||||
brew install tesseract-lang
|
||||
```
|
||||
|
||||
然后,您可以向 OCRmyPDF 传递 `-l LANG` 参数,提示它应该搜索哪些语言。可以请求多种语言。
|
||||
|
||||
OCRmyPDF 支持 Tesseract 4.1.1+。它会自动使用在 `PATH` 环境变量中首先找到的版本。在 Windows 上,如果 `PATH` 不提供 Tesseract 二进制文件,我们会根据 Windows 注册表使用已安装的最高版本号。
|
||||
|
||||
## 文档和支持
|
||||
|
||||
安装 OCRmyPDF 后,可以通过以下方式访问内置帮助,解释命令语法和选项:
|
||||
|
||||
```bash
|
||||
ocrmypdf --help
|
||||
```
|
||||
|
||||
我们的[文档托管在 Read the Docs 上](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
|
||||
请在我们的 [GitHub issues](https://github.com/ocrmypdf/OCRmyPDF/issues) 页面上报告问题,并遵循问题模板以获得快速响应。
|
||||
|
||||
## 功能演示
|
||||
|
||||
```bash
|
||||
# 添加 OCR 层并转换为 PDF/A
|
||||
ocrmypdf input.pdf output.pdf
|
||||
|
||||
# 将图像转换为单页 PDF
|
||||
ocrmypdf input.jpg output.pdf
|
||||
|
||||
# 就地为文件添加 OCR(仅在成功时修改文件)
|
||||
ocrmypdf myfile.pdf myfile.pdf
|
||||
|
||||
# 使用非英语语言进行 OCR(查找您语言的 ISO 639-3 代码)
|
||||
ocrmypdf -l fra LeParisien.pdf LeParisien.pdf
|
||||
|
||||
# OCR 多语言文档
|
||||
ocrmypdf -l eng+fra Bilingual-English-French.pdf Bilingual-English-French.pdf
|
||||
|
||||
# 校正(矫正倾斜的页面)
|
||||
ocrmypdf --deskew input.pdf output.pdf
|
||||
```
|
||||
|
||||
更多功能,请参阅[文档](https://ocrmypdf.readthedocs.io/en/latest/index.html)。
|
||||
|
||||
## 要求
|
||||
|
||||
除了所需的 Python 版本外,OCRmyPDF 还需要外部程序安装 Ghostscript 和 Tesseract OCR。OCRmyPDF 是纯 Python 编写的,几乎可以在所有平台上运行:Linux、macOS、Windows 和 FreeBSD。
|
||||
|
||||
## 媒体报道
|
||||
|
||||
- [使用 OCRmyPDF 实现无纸化](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a)
|
||||
- [将扫描文档转换为带有编辑的压缩可搜索 PDF](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c)
|
||||
- [c't 1-2014, 第 59 页](https://heise.de/-2279695):在德国领先的 IT 杂志 c't 中详细介绍 OCRmyPDF v1.0
|
||||
- [heise Open Source, 09/2014: 使用 OCRmyPDF 进行文本识别](https://heise.de/-2356670)
|
||||
- [heise 使用 OCRmyPDF 创建可搜索的 PDF 文档](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [优秀实用工具:OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser 使用 OCRmyPDF 和 Scanbd 自动化文本识别](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [Y Combinator 讨论](https://news.ycombinator.com/item?id=32028752)
|
||||
|
||||
## 商业咨询
|
||||
|
||||
如果没有公司和用户选择为功能开发和咨询提供支持,OCRmyPDF 就不会成为今天的软件。我们很乐意讨论所有咨询,无论是扩展现有功能集,还是将 OCRmyPDF 集成到更大的系统中。
|
||||
|
||||
## 许可证
|
||||
|
||||
OCRmyPDF 软件根据 Mozilla 公共许可证 2.0 (MPL-2.0) 授权。此许可证允许将 OCRmyPDF 与其他代码集成,包括商业和闭源代码,但要求您发布对 OCRmyPDF 所做的源代码级修改。
|
||||
|
||||
OCRmyPDF 的某些组件有其他许可证,如标准 SPDX 许可证标识符或 DEP5 版权和许可信息文件所示。一般来说,非核心代码根据 MIT 许可,文档和测试文件根据 Creative Commons ShareAlike 4.0 (CC-BY-SA 4.0) 许可。
|
||||
|
||||
## 免责声明
|
||||
|
||||
本软件按"原样"分发,不提供任何明示或暗示的保证或条件。
|
||||
|
||||
这份中文版 README.md 保留了原始文档的所有重要信息,包括功能介绍、安装说明、语言支持、使用示例等内容,同时保持了原始格式和结构。
|
||||
+184
@@ -0,0 +1,184 @@
|
||||
version = 1
|
||||
SPDX-PackageName = "OCRmyPDF"
|
||||
SPDX-PackageSupplier = "James R. Barlow <james@purplerock.ca>"
|
||||
SPDX-PackageDownloadLocation = "https://github.com/ocrmypdf/OCRmyPDF"
|
||||
|
||||
[[annotations]]
|
||||
path = ["docs/**", 'misc/screencast/**']
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2025 James R. Barlow"
|
||||
SPDX-License-Identifier = "CC-BY-SA-4.0"
|
||||
|
||||
[[annotations]]
|
||||
path = [
|
||||
"uv.lock",
|
||||
".git_archival.txt",
|
||||
"docs/images/logo-social.png",
|
||||
"docs/images/logo-square-256.svg",
|
||||
"docs/images/logo-square.png",
|
||||
"docs/images/logo-square.svg",
|
||||
"docs/images/logo.svg",
|
||||
]
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2025 James R. Barlow"
|
||||
SPDX-License-Identifier = "MPL-2.0"
|
||||
|
||||
[[annotations]]
|
||||
path = [".github/ISSUE_TEMPLATE/**.yml"]
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2025 James R. Barlow"
|
||||
SPDX-License-Identifier = "CC-BY-SA-4.0"
|
||||
|
||||
[[annotations]]
|
||||
path = [
|
||||
"tests/resources/acroform.pdf",
|
||||
"tests/resources/aspect.pdf",
|
||||
"tests/resources/blank.pdf",
|
||||
"tests/resources/cmyk.pdf",
|
||||
"tests/resources/crom.png",
|
||||
"tests/resources/enormous.pdf",
|
||||
"tests/resources/formxobject.pdf",
|
||||
"tests/resources/francais.pdf",
|
||||
"tests/resources/hugemono.pdf",
|
||||
"tests/resources/invalid.pdf",
|
||||
"tests/resources/kcs.pdf",
|
||||
"tests/resources/livecycle.pdf",
|
||||
"tests/resources/meta.pdf",
|
||||
"tests/resources/missing_docinfo.pdf",
|
||||
"tests/resources/negzero.pdf",
|
||||
"tests/resources/no_contents.pdf",
|
||||
"tests/resources/tagged**",
|
||||
"tests/resources/toc.pdf",
|
||||
"tests/resources/trivial.pdf",
|
||||
"tests/resources/truetype_font_nomapping.pdf",
|
||||
"tests/resources/type3_font_nomapping.pdf",
|
||||
]
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2025 James R. Barlow"
|
||||
SPDX-License-Identifier = "CC-BY-SA-4.0"
|
||||
|
||||
[[annotations]]
|
||||
path = ["tests/resources/graph.pdf", "tests/resources/graph_ocred.pdf"]
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2012 SmokeyJoe"
|
||||
SPDX-License-Identifier = "GFDL-1.2-or-later or CC-BY-SA-3.0"
|
||||
|
||||
[[annotations]]
|
||||
path = ["tests/resources/c02-22.pdf", "tests/resources/multipage.pdf"]
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "Public domain"
|
||||
SPDX-License-Identifier = "public-domain"
|
||||
|
||||
[[annotations]]
|
||||
path = "docs/images/bitmap_vs_svg.svg"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2006 Yug"
|
||||
SPDX-License-Identifier = "CC-BY-SA-2.5"
|
||||
|
||||
[[annotations]]
|
||||
path = "tests/cache/**"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2025 James R. Barlow"
|
||||
SPDX-License-Identifier = "CC-BY-SA-4.0"
|
||||
|
||||
[[annotations]]
|
||||
path = [
|
||||
"tests/resources/linn.png",
|
||||
"tests/resources/linn.pdf",
|
||||
"tests/resources/linn.txt",
|
||||
"tests/resources/ccitt.pdf",
|
||||
"tests/resources/cardinal.pdf",
|
||||
"tests/resources/jbig2.pdf",
|
||||
"tests/resources/jbig2_baddevicen.pdf",
|
||||
"tests/resources/skew.pdf",
|
||||
"tests/resources/rotated_skew.pdf",
|
||||
"tests/resources/poster.pdf",
|
||||
]
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 1985 Forat Electronics"
|
||||
SPDX-License-Identifier = "GFDL-1.2-or-later or CC-BY-SA-3.0"
|
||||
|
||||
[[annotations]]
|
||||
path = "tests/resources/lichtenstein.pdf"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = ["(C) 2001 Andreas Tille", "(C) 2007 Alessio Damato"]
|
||||
SPDX-License-Identifier = "GFDL-1.2-or-later or CC-BY-SA-3.0"
|
||||
|
||||
[[annotations]]
|
||||
path = "tests/resources/masks.pdf"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = [
|
||||
"held by the contributors to the German Wikipedia article \"Linux\"",
|
||||
"see: https://de.wikipedia.org/w/index.php?title=Linux&action=history",
|
||||
"(masks.pdf generated from Wikipedia article as of 2016-08-24)",
|
||||
]
|
||||
SPDX-License-Identifier = "CC-BY-SA-3.0"
|
||||
|
||||
[[annotations]]
|
||||
path = "tests/resources/epson.pdf"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = [
|
||||
"held by the contributors to the Wikipedia article \"Optical character recognition\"",
|
||||
"see: https://en.wikipedia.org/w/index.php?title=Optical_character_recognition&action=history",
|
||||
"(epson.pdf generated from Wikipedia article as of 2016-09-14)",
|
||||
]
|
||||
SPDX-License-Identifier = "CC-BY-SA-3.0"
|
||||
|
||||
[[annotations]]
|
||||
path = ["tests/resources/typewriter.png", "tests/resources/2400dpi.pdf"]
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2005 Ellywa"
|
||||
SPDX-License-Identifier = "GFDL-1.2-or-later or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0"
|
||||
SPDX-FileComment = "\n Obtained from: https://commons.wikimedia.org/wiki/File:Triumph.typewriter_text_Linzensoep.gif"
|
||||
|
||||
[[annotations]]
|
||||
path = "tests/resources/overlay.pdf"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2017 Max Anderson"
|
||||
SPDX-License-Identifier = "MIT"
|
||||
|
||||
[[annotations]]
|
||||
path = [
|
||||
"tests/resources/baiona**.png",
|
||||
"tests/resources/baiona**.jpg",
|
||||
"tests/resources/link.pdf",
|
||||
"tests/resources/palette.pdf",
|
||||
]
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2014 Euskaldunaa"
|
||||
SPDX-License-Identifier = "CC-BY-SA-4.0"
|
||||
|
||||
[[annotations]]
|
||||
path = "tests/resources/vector.pdf"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = "(C) 2018 Catscratch"
|
||||
SPDX-License-Identifier = "MIT"
|
||||
|
||||
[[annotations]]
|
||||
path = "src/ocrmypdf/data/sRGB.icc"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = [
|
||||
"Kai-Uwe Behrmann <www.behrmann.name>",
|
||||
"Marti Maria <www.littlecms.com>",
|
||||
"Photogamut <www.photogamut.org>",
|
||||
"Graeme Gill <www.argyllcms.com>",
|
||||
"ColorSolutions <www.basICColor.com>",
|
||||
]
|
||||
SPDX-License-Identifier = "Zlib"
|
||||
|
||||
[[annotations]]
|
||||
path = "src/ocrmypdf/data/Occulta.ttf"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = ["(C) 2026 James R. Barlow"]
|
||||
SPDX-License-Identifier = "Apache-2.0"
|
||||
|
||||
[[annotations]]
|
||||
path = "tests/resources/3small.pdf"
|
||||
precedence = "aggregate"
|
||||
SPDX-FileCopyrightText = [
|
||||
"(C) 2014 Euskaldunaa",
|
||||
"(C) 2017 James R. Barlow",
|
||||
"(C) 2005 Ellywa",
|
||||
]
|
||||
SPDX-License-Identifier = "CC-BY-SA-4.0 and (GFDL-1.2-or-later or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0)"
|
||||
SPDX-FileComment = "concatenation of baiona_gray.png, crom.png and typewriter.png/2400dpi.pdf"
|
||||
@@ -0,0 +1,613 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Advanced features
|
||||
|
||||
## Control of unpaper
|
||||
|
||||
OCRmyPDF uses `unpaper` to provide the implementation of the
|
||||
`--clean` and `--clean-final` arguments.
|
||||
[unpaper](https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md)
|
||||
provides a variety of image processing filters to improve images.
|
||||
|
||||
By default, OCRmyPDF uses only `unpaper` arguments that were found to
|
||||
be safe to use on almost all files without having to inspect every page
|
||||
of the file afterwards. This is particularly true when only `--clean`
|
||||
is used, since that instructs OCRmyPDF to only clean the image before
|
||||
OCR and not the final image.
|
||||
|
||||
However, if you wish to use the more aggressive options in `unpaper`,
|
||||
you may use `--unpaper-args '...'` to override the OCRmyPDF's defaults
|
||||
and forward other arguments to unpaper. This option will forward
|
||||
arguments to `unpaper` without any knowledge of what that program
|
||||
considers to be valid arguments. The string of arguments must be quoted
|
||||
as shown in the examples below. No filename arguments may be included.
|
||||
OCRmyPDF will assume it can append input and output filename of
|
||||
intermediate images to the `--unpaper-args` string.
|
||||
|
||||
In this example, we tell `unpaper` to expect two pages of text on a
|
||||
sheet (image), such as occurs when two facing pages of a book are
|
||||
scanned. `unpaper` uses this information to deskew each independently
|
||||
and clean up the margins of both.
|
||||
|
||||
```bash
|
||||
ocrmypdf --clean --clean-final --unpaper-args '--layout double' input.pdf output.pdf
|
||||
ocrmypdf --clean --clean-final --unpaper-args '--layout double --no-noisefilter' input.pdf output.pdf
|
||||
```
|
||||
|
||||
:::{warning}
|
||||
Some `unpaper` features will reposition text within the image.
|
||||
`--clean-final` is recommended to avoid this issue.
|
||||
:::
|
||||
|
||||
:::{warning}
|
||||
Some `unpaper` features cause multiple input or output files to be
|
||||
consumed or produced. OCRmyPDF requires `unpaper` to consume one
|
||||
file and produce one file; errors will result if this assumption is not
|
||||
met.
|
||||
:::
|
||||
|
||||
:::{note}
|
||||
`unpaper` uses uncompressed PBM/PGM/PPM files for its intermediate
|
||||
files. For large images or documents, it can take a lot of temporary
|
||||
disk space.
|
||||
:::
|
||||
|
||||
## Control of OCR options
|
||||
|
||||
OCRmyPDF provides many features to control the behavior of the OCR
|
||||
engine, Tesseract.
|
||||
|
||||
### OCR processing mode
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
The `--mode` (`-m`) argument consolidates OCR processing options.
|
||||
:::
|
||||
|
||||
OCRmyPDF provides a unified `--mode` argument to control how pages with
|
||||
existing text are handled:
|
||||
|
||||
| Mode | Behavior | Legacy equivalent |
|
||||
|------|----------|-------------------|
|
||||
| `default` | Error if text is found | (no flag) |
|
||||
| `force` | Rasterize all content and run OCR | `--force-ocr` |
|
||||
| `skip` | Skip pages with existing text | `--skip-text` |
|
||||
| `redo` | Re-OCR pages, stripping old OCR layer | `--redo-ocr` |
|
||||
|
||||
```bash
|
||||
# Skip pages that already have text
|
||||
ocrmypdf --mode skip input.pdf output.pdf
|
||||
# or equivalently:
|
||||
ocrmypdf -m skip input.pdf output.pdf
|
||||
|
||||
# Force OCR on all pages (rasterizes everything)
|
||||
ocrmypdf --mode force input.pdf output.pdf
|
||||
|
||||
# Re-do OCR, replacing old invisible text
|
||||
ocrmypdf --mode redo input.pdf output.pdf
|
||||
```
|
||||
|
||||
The legacy flags (`--force-ocr`, `--skip-text`, `--redo-ocr`) remain as
|
||||
silent aliases for backward compatibility.
|
||||
|
||||
### When OCR is skipped
|
||||
|
||||
If a page in a PDF seems to have text, by default OCRmyPDF will exit
|
||||
without modifying the PDF. This is to ensure that PDFs that were
|
||||
previously OCRed or were "born digital" rather than scanned are not
|
||||
processed.
|
||||
|
||||
If `--mode skip` (or `--skip-text`) is issued, then no image processing or OCR will be
|
||||
performed on pages that already have text. The page will be copied to
|
||||
the output. This may be useful for documents that contain both "born
|
||||
digital" and scanned content, or to use OCRmyPDF to normalize and
|
||||
convert to PDF/A regardless of their contents.
|
||||
|
||||
If `--mode redo` (or `--redo-ocr`) is issued, then a detailed text analysis is performed.
|
||||
Text is categorized as either visible or invisible. Invisible text (OCR)
|
||||
is stripped out. Then an image of each page is created with visible text
|
||||
masked out. The page image is sent for OCR, and any additional text is
|
||||
inserted as OCR. If a file contains a mix of text and bitmap images that
|
||||
contain text, OCRmyPDF will locate the additional text in images without
|
||||
disrupting the existing text. Some PDF OCR solutions render text as
|
||||
technically printable or visible in some way, perhaps by drawing it and
|
||||
then painting over it. OCRmyPDF cannot distinguish this type of OCR
|
||||
text from real text, so it will not be "redone".
|
||||
|
||||
If `--mode force` (or `--force-ocr`) is issued, then all pages will be rasterized to
|
||||
images, discarding any hidden OCR text, rasterizing any printable
|
||||
text, and flattening form fields or interactive objects into their visual
|
||||
representation. This is useful for redoing OCR, for fixing OCR text
|
||||
with a damaged character map (text is selectable but not searchable),
|
||||
and destroying redacted information.
|
||||
|
||||
### Time and image size limits
|
||||
|
||||
By default, OCRmyPDF permits tesseract to run for three minutes (180
|
||||
seconds) per page. This is usually more than enough time to find all
|
||||
text on a reasonably sized page with modern hardware.
|
||||
|
||||
If a page is skipped, it will be inserted without OCR. If preprocessing
|
||||
was requested, the preprocessed image layer will be inserted.
|
||||
|
||||
If you want to adjust the amount of time spent on OCR, change
|
||||
`--tesseract-timeout`. You can also automatically skip images that
|
||||
exceed a certain number of megapixels with `--skip-big`. (A 300 DPI,
|
||||
8.5×11" page image is 8.4 megapixels.)
|
||||
|
||||
```bash
|
||||
# Allow 300 seconds for OCR; skip any page larger than 50 megapixels
|
||||
ocrmypdf --tesseract-timeout 300 --skip-big 50 bigfile.pdf output.pdf
|
||||
```
|
||||
|
||||
### OCR for huge images
|
||||
|
||||
Tesseract has internal limits on the size
|
||||
of images it will process. By default,
|
||||
`--tesseract-downsample-large-images` is enabled, and OCRmyPDF will
|
||||
downsample images to fit Tesseract limits. (The limits are usually encountered
|
||||
only for scanned images of oversized media, such as large maps or blueprints exceeding
|
||||
110 cm or 43 inches in either dimension, and at high DPI.) This feature can disabled
|
||||
using `--no-tesseract-downsample-large-images`.
|
||||
|
||||
`--tesseract-downsample-above Npixels` adjusts the threshold at which images
|
||||
will be downsampled. By default, only images that exceed any of Tesseract's
|
||||
internal limits are downsampled (32767 pixels on either dimension).
|
||||
|
||||
You will also need to set `--tesseract-timeout` high enough to allow
|
||||
for processing.
|
||||
|
||||
Only the image sent for OCR is downsampled. The original image is
|
||||
preserved.
|
||||
|
||||
```bash
|
||||
# Allow 600 seconds for OCR on huge images
|
||||
ocrmypdf --tesseract-timeout 600 \
|
||||
--tesseract-downsample-large-images \
|
||||
bigfile.pdf output.pdf
|
||||
|
||||
# Downsample images above 5000 pixels on the longest dimension to
|
||||
# 5000 pixels
|
||||
ocrmypdf --tesseract-timeout 120 \
|
||||
--tesseract-downsample-large-images \
|
||||
--tesseract-downsample-above 5000 \
|
||||
bigfile.pdf output_downsampled_ocr.pdf
|
||||
```
|
||||
|
||||
### Overriding default tesseract
|
||||
|
||||
OCRmyPDF checks the system `PATH` for the `tesseract` binary.
|
||||
|
||||
Some relevant environment variables that influence Tesseract's behavior
|
||||
include:
|
||||
|
||||
```{eval-rst}
|
||||
.. envvar:: TESSDATA_PREFIX
|
||||
|
||||
Overrides the path to Tesseract's data files. This can allow
|
||||
simultaneous installation of the "best" and "fast" training data
|
||||
sets. OCRmyPDF does not manage this environment variable.
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. envvar:: OMP_THREAD_LIMIT
|
||||
|
||||
Controls the number of threads Tesseract will use. OCRmyPDF will
|
||||
manage this environment variable if it is not already set.
|
||||
```
|
||||
|
||||
For example, if you have a development build of Tesseract don't wish to
|
||||
use the system installation, you can launch OCRmyPDF as follows:
|
||||
|
||||
```bash
|
||||
env \
|
||||
PATH=/home/user/src/tesseract/api:$PATH \
|
||||
TESSDATA_PREFIX=/home/user/src/tesseract \
|
||||
ocrmypdf input.pdf output.pdf
|
||||
```
|
||||
|
||||
In this example `TESSDATA_PREFIX` is required to redirect Tesseract to
|
||||
an alternate folder for its "tessdata" files.
|
||||
|
||||
### Overriding other support programs
|
||||
|
||||
In addition to tesseract, OCRmyPDF uses the following external binaries:
|
||||
|
||||
- `gs` (Ghostscript)
|
||||
- `unpaper`
|
||||
- `pngquant`
|
||||
- `jbig2`
|
||||
|
||||
In each case OCRmyPDF will search the `PATH` environment variable to
|
||||
locate the binaries. By modifying the `PATH` environment variable, you
|
||||
can override the binaries that OCRmyPDF uses.
|
||||
|
||||
### Changing Tesseract configuration variables
|
||||
|
||||
You can override Tesseract's default [control
|
||||
parameters](https://tesseract-ocr.github.io/tessdoc/tess3/ControlParams.html)
|
||||
with a configuration file.
|
||||
|
||||
As an example, this configuration will disable Tesseract's dictionary
|
||||
for current language. Normally the dictionary is helpful for
|
||||
interpolating words that are unclear, but it may interfere with OCR if
|
||||
the document does not contain many words (for example, a list of part
|
||||
numbers).
|
||||
|
||||
Create a file named "no-dict.cfg" with these contents:
|
||||
|
||||
```
|
||||
load_system_dawg 0
|
||||
language_model_penalty_non_dict_word 0
|
||||
language_model_penalty_non_freq_dict_word 0
|
||||
```
|
||||
|
||||
then run ocrmypdf as follows (along with any other desired arguments):
|
||||
|
||||
```bash
|
||||
ocrmypdf --tesseract-config no-dict.cfg input.pdf output.pdf
|
||||
```
|
||||
|
||||
:::{warning}
|
||||
Some combinations of control parameters will break Tesseract or break
|
||||
assumptions that OCRmyPDF makes about Tesseract's output.
|
||||
:::
|
||||
|
||||
### Changing page segmentation mode
|
||||
|
||||
The directive `--tesseract-pagesegmode Nmode` forwards the desired page segmentation
|
||||
mode to Tesseract OCR. The default is 3.
|
||||
|
||||
Page segmentation can improve OCR results when you know that a PDF ought to be
|
||||
analyzed a particular way, such as PDFs whose pages contain only a single line of
|
||||
text. For the vast majority of users, changing the page segmentation mode will only
|
||||
make things worse.
|
||||
|
||||
As of June 2024, the Tesseract page segmentation modes are:
|
||||
|
||||
| ID | Description |
|
||||
| --- | --------------------------------------------------------------------------------------------- |
|
||||
| 0 | Orientation and script detection (OSD) only. |
|
||||
| 1 | Automatic page segmentation with OSD. |
|
||||
| 2 | Automatic page segmentation, but no OSD, or OCR. (not implemented) |
|
||||
| 3 | Fully automatic page segmentation, but no OSD. (Default) |
|
||||
| 4 | Assume a single column of text of variable sizes. |
|
||||
| 5 | Assume a single uniform block of vertically aligned text. |
|
||||
| 6 | Assume a single uniform block of text. |
|
||||
| 7 | Treat the image as a single text line. |
|
||||
| 8 | Treat the image as a single word. |
|
||||
| 9 | Treat the image as a single word in a circle. |
|
||||
| 10 | Treat the image as a single character. |
|
||||
| 11 | Sparse text. Find as much text as possible in no particular order. |
|
||||
| 12 | Sparse text with OSD. |
|
||||
| 13 | Raw line. Treat the image as a single text line, bypassing hacks that are Tesseract-specific. |
|
||||
|
||||
Modes 0, 1, 2, and 12 (all of those that enable orientation and script detection)
|
||||
are not compatible with OCRmyPDF, which performs OSD in a separate step from OCR.
|
||||
Their use may interfere with `--rotate-pages` and other features.
|
||||
|
||||
It is currently not possible to use advanced Tesseract OCR features, such as creating
|
||||
OCR information, when using Tesseract through OCRmyPDF.
|
||||
|
||||
## Choosing a PDF rasterizer
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
rasterizing
|
||||
|
||||
: Converting a PDF page to an image for OCR processing.
|
||||
|
||||
OCRmyPDF supports two PDF rasterizers:
|
||||
|
||||
| Rasterizer | Package | Advantages | Disadvantages |
|
||||
|------------|---------|------------|---------------|
|
||||
| pypdfium2 | Python package | Faster, fewer version issues | Requires pypdfium2 package |
|
||||
| Ghostscript | System binary | More widely packaged | Version consistency issues, restrictive AGPLv3 |
|
||||
|
||||
The `--rasterizer` argument controls which rasterizer is used:
|
||||
|
||||
```bash
|
||||
# Automatic selection (default) - prefers pypdfium when available
|
||||
ocrmypdf --rasterizer auto input.pdf output.pdf
|
||||
|
||||
# Force pypdfium2
|
||||
ocrmypdf --rasterizer pypdfium input.pdf output.pdf
|
||||
|
||||
# Force Ghostscript
|
||||
ocrmypdf --rasterizer ghostscript input.pdf output.pdf
|
||||
```
|
||||
|
||||
pypdfium2 is a Python binding for pdfium, the PDF rendering library used
|
||||
by Google Chrome and Chromium. It generally produces output identical to
|
||||
Ghostscript but with better performance.
|
||||
|
||||
:::{note}
|
||||
If pypdfium2 is not installed and `--rasterizer pypdfium` is requested,
|
||||
OCRmyPDF will exit with an error. Install it with: `pip install pypdfium2`
|
||||
:::
|
||||
|
||||
## Changing the PDF renderer
|
||||
|
||||
rendering
|
||||
|
||||
: Creating a new PDF from other data (such as an existing PDF).
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
The fpdf2 renderer is now the default, replacing the legacy hOCR renderer.
|
||||
:::
|
||||
|
||||
OCRmyPDF uses PDF renderers to create the invisible text layer. The
|
||||
renderer may be selected using `--pdf-renderer`. The default is
|
||||
`auto` which selects `fpdf2`.
|
||||
|
||||
### The `fpdf2` renderer (default)
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
The fpdf2 renderer creates text layers using the fpdf2 library. It provides:
|
||||
|
||||
- Full multilingual support including RTL languages (Arabic, Hebrew, Persian)
|
||||
- Accurate text positioning aligned with OCR bounding boxes
|
||||
- Improved "Occulta" glyphless font handling:
|
||||
- Zero-width markers are properly handled
|
||||
- Double-width CJK characters are properly sized
|
||||
- Direct OcrElement tree input (no hOCR intermediate format required)
|
||||
|
||||
The fpdf2 renderer is the recommended choice for all installations.
|
||||
|
||||
:::{note}
|
||||
The fpdf2 renderer may be slightly slower than the legacy hocrtransform
|
||||
renderer for some workloads. This is an area of ongoing optimization.
|
||||
:::
|
||||
|
||||
In both renderers, a text-only layer is rendered and sandwiched (overlaid)
|
||||
on to either the original PDF page, or newly rasterized version of the
|
||||
original PDF page (when `--mode force` is used). In this way, loss
|
||||
of PDF information is generally avoided. (You may need to disable PDF/A
|
||||
conversion and optimization to eliminate all lossy transformations.)
|
||||
|
||||
### The `sandwich` renderer
|
||||
|
||||
The `sandwich` renderer uses Tesseract's text-only PDF feature,
|
||||
which produces a PDF page that lays out the OCR in invisible text.
|
||||
|
||||
Currently some problematic PDF viewers like Mozilla PDF.js and macOS
|
||||
Preview have problems with segmenting its text output, and
|
||||
mightrunseveralwordstogether. It also does not implement right to left
|
||||
fonts (Arabic, Hebrew, Persian). The output of this renderer cannot
|
||||
be edited. The sandwich renderer is retained for testing.
|
||||
|
||||
When image preprocessing features like `--deskew` are used, the
|
||||
original PDF will be rendered as a full page and the OCR layer will be
|
||||
placed on top.
|
||||
|
||||
### Legacy renderer options
|
||||
|
||||
The `hocr` and `hocrdebug` renderer options are deprecated and
|
||||
automatically redirect to `fpdf2`. They will be removed in a future version.
|
||||
|
||||
## Rendering and rasterizing options
|
||||
|
||||
:::{versionadded} 14.3.0
|
||||
:::
|
||||
|
||||
The `--continue-on-soft-render-error` option allows OCRmyPDF to
|
||||
proceed if a page cannot be rasterized/rendered. This is useful if you are
|
||||
trying to get the best possible OCR from a PDF that is not well-formed,
|
||||
and you are willing to accept some pages that may not visually match the
|
||||
input, and that may not OCR well.
|
||||
|
||||
## Color conversion strategy
|
||||
|
||||
:::{versionadded} 15.0.0
|
||||
:::
|
||||
|
||||
OCRmyPDF uses Ghostscript to convert PDF to PDF/A. In some cases, this
|
||||
conversion requires color conversion. The default strategy is to convert
|
||||
using the `LeaveColorUnchanged` strategy, which preserves the original
|
||||
color space wherever possible (some rare color spaces might still be
|
||||
converted).
|
||||
|
||||
Usually document scanners produce PDFs in the sRGB color space, and do
|
||||
not need to be converted, so the default strategy is appropriate.
|
||||
|
||||
Suppose that you have a document that was prepared for professional
|
||||
printing in a Separation or CMYK color space, and text was converted to
|
||||
curves. In this case, you may want to use a different color conversion
|
||||
strategy. The `--color-conversion-strategy` option allows you to select a
|
||||
different strategy, such as `RGB`.
|
||||
|
||||
## PDF/A output modes
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
The default `--output-type` is now `auto` instead of `pdfa`.
|
||||
:::
|
||||
|
||||
OCRmyPDF can produce PDF/A compliant output for long-term archival. The
|
||||
`--output-type` argument controls PDF/A conversion:
|
||||
|
||||
| Output type | Behavior |
|
||||
|-------------|----------|
|
||||
| `auto` | Best-effort PDF/A without requiring Ghostscript (default) |
|
||||
| `pdfa` | PDF/A-2b via Ghostscript |
|
||||
| `pdfa-1` | PDF/A-1b via Ghostscript |
|
||||
| `pdfa-2` | PDF/A-2b via Ghostscript (same as `pdfa`) |
|
||||
| `pdfa-3` | PDF/A-3b via Ghostscript |
|
||||
| `pdf` | Standard PDF, no PDF/A conversion |
|
||||
| `none` | No output file (useful with `--sidecar`) |
|
||||
|
||||
### Speculative PDF/A conversion
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
When `--output-type auto` is used (the default), OCRmyPDF attempts a
|
||||
fast "speculative" PDF/A conversion that avoids Ghostscript when possible:
|
||||
|
||||
1. OCRmyPDF adds an sRGB ICC profile and PDF/A XMP metadata using pikepdf
|
||||
2. If verapdf is available, it validates the result
|
||||
3. If validation passes, Ghostscript is skipped entirely
|
||||
4. If validation fails or verapdf is unavailable, falls back to Ghostscript
|
||||
|
||||
This approach is faster and avoids some Ghostscript limitations (such as
|
||||
image transcoding), but only works for PDFs that are already "mostly"
|
||||
PDF/A compliant.
|
||||
|
||||
### PDF/A conversion flow
|
||||
|
||||
The following diagram illustrates the PDF/A conversion decision tree:
|
||||
|
||||
```{mermaid}
|
||||
flowchart TD
|
||||
A[Start] --> B{--output-type?}
|
||||
B -->|pdf| C[Output standard PDF]
|
||||
B -->|pdfa/pdfa-N| D[Use Ghostscript]
|
||||
B -->|auto| E[Attempt speculative conversion]
|
||||
|
||||
E --> F["Add sRGB ICC + XMP metadata (pikepdf)"]
|
||||
F --> G{verapdf available?}
|
||||
|
||||
G -->|No| H{Ghostscript available?}
|
||||
G -->|Yes| I[Validate with verapdf]
|
||||
|
||||
I --> J{Validation passed?}
|
||||
J -->|Yes| K[Output PDF/A - Ghostscript skipped]
|
||||
J -->|No| H
|
||||
|
||||
H -->|Yes| D
|
||||
H -->|No| L[Output standard PDF + WARNING]
|
||||
|
||||
D --> M[Ghostscript PDF/A conversion]
|
||||
M --> N[Output PDF/A]
|
||||
|
||||
style K fill:#90EE90
|
||||
style N fill:#90EE90
|
||||
style L fill:#FFB6C1
|
||||
```
|
||||
|
||||
:::{warning}
|
||||
**Breaking change:** If neither Ghostscript nor verapdf is installed,
|
||||
`--output-type auto` will produce a standard PDF instead of PDF/A.
|
||||
This is a change from previous versions where Ghostscript was required
|
||||
and PDF/A was always produced.
|
||||
:::
|
||||
|
||||
## Return code policy
|
||||
|
||||
OCRmyPDF writes all messages to `stderr`. `stdout` is reserved for
|
||||
piping output files. `stdin` is reserved for piping input files.
|
||||
|
||||
The return codes generated by the OCRmyPDF are considered part of the
|
||||
stable user interface. They may be imported from
|
||||
`ocrmypdf.exceptions`.
|
||||
|
||||
```{eval-rst}
|
||||
.. list-table:: Return codes
|
||||
:widths: 5 35 60
|
||||
:header-rows: 1
|
||||
|
||||
* - Code
|
||||
- Name
|
||||
- Interpretation
|
||||
* - 0
|
||||
- ``ExitCode.ok``
|
||||
- Everything worked as expected.
|
||||
* - 1
|
||||
- ``ExitCode.bad_args``
|
||||
- Invalid arguments, exited with an error.
|
||||
* - 2
|
||||
- ``ExitCode.input_file``
|
||||
- The input file does not seem to be a valid PDF.
|
||||
* - 3
|
||||
- ``ExitCode.missing_dependency``
|
||||
- An external program required by OCRmyPDF is missing.
|
||||
* - 4
|
||||
- ``ExitCode.invalid_output_pdf``
|
||||
- An output file was created, but it does not seem to be a valid PDF. The file will be available.
|
||||
* - 5
|
||||
- ``ExitCode.file_access_error``
|
||||
- The user running OCRmyPDF does not have sufficient permissions to read the input file and write the output file.
|
||||
* - 6
|
||||
- ``ExitCode.already_done_ocr``
|
||||
- The file already appears to contain text so it may not need OCR. See output message.
|
||||
* - 7
|
||||
- ``ExitCode.child_process_error``
|
||||
- An error occurred in an external program (child process) and OCRmyPDF cannot continue.
|
||||
* - 8
|
||||
- ``ExitCode.encrypted_pdf``
|
||||
- The input PDF is encrypted. OCRmyPDF does not read encrypted PDFs. Use another program such as ``qpdf`` to remove encryption.
|
||||
* - 9
|
||||
- ``ExitCode.invalid_config``
|
||||
- A custom configuration file was forwarded to Tesseract using ``--tesseract-config``, and Tesseract rejected this file.
|
||||
* - 10
|
||||
- ``ExitCode.pdfa_conversion_failed``
|
||||
- A valid PDF was created, PDF/A conversion failed. The file will be available.
|
||||
* - 15
|
||||
- ``ExitCode.other_error``
|
||||
- Some other error occurred.
|
||||
* - 130
|
||||
- ``ExitCode.ctrl_c``
|
||||
- The program was interrupted by pressing Ctrl+C.
|
||||
|
||||
```
|
||||
|
||||
(tmpdir)=
|
||||
## Changing temporary storage location
|
||||
|
||||
OCRmyPDF generates many temporary files during processing.
|
||||
|
||||
To change where temporary files are stored, change the `TMPDIR`
|
||||
environment variable for ocrmypdf's environment. (Python's
|
||||
`tempfile.gettempdir()` returns the root directory in which temporary
|
||||
files will be stored.) For example, one could redirect `TMPDIR` to a
|
||||
large RAM disk to avoid wear on HDD/SSD and potentially improve
|
||||
performance.
|
||||
|
||||
On Windows, the `TEMP` environment variable is used instead.
|
||||
|
||||
## Debugging the intermediate files
|
||||
|
||||
OCRmyPDF normally saves its intermediate results to a temporary folder
|
||||
and deletes this folder when it exits, whether it succeeded or failed.
|
||||
|
||||
If the `--keep-temporary-files` (`-k`) argument is issued on the
|
||||
command line, OCRmyPDF will keep the temporary folder and print the location,
|
||||
whether it succeeded or failed. An example message is:
|
||||
|
||||
```none
|
||||
Temporary working files retained at:
|
||||
/tmp/ocrmypdf.io.u20wpz07
|
||||
```
|
||||
|
||||
When OCRmyPDF is launched as a snap, this corresponds to the snap filesystem, for instance:
|
||||
|
||||
> /tmp/snap-private-tmp/snap.ocrmypdf/tmp/ocrmypdf.io.u20wpz07
|
||||
|
||||
The organization of this folder is an implementation detail and subject
|
||||
to change between releases. However the general organization is that
|
||||
working files on a per page basis have the page number as a prefix
|
||||
(starting with page 1), an infix indicates the processing stage, and a
|
||||
suffix indicates the file type. Some important files include:
|
||||
|
||||
- `_rasterize.png` - what the input page looks like
|
||||
- `_ocr.png` - the file that is sent to Tesseract for OCR; depending
|
||||
on arguments this may differ from the presentation image
|
||||
- `_pp_deskew.png` - the image, after deskewing
|
||||
- `_pp_clean.png` - the image, after cleaning with unpaper
|
||||
- `_ocr_hocr.pdf` - the OCR file; appears as a blank page with invisible
|
||||
text embedded
|
||||
- `_ocr_hocr.txt` - the OCR text (not necessarily all text on the page,
|
||||
if the page is mixed format)
|
||||
- `fix_docinfo.pdf` - a temporary file created to fix the PDF DocumentInfo
|
||||
data structure
|
||||
- `graft_layers.pdf` - the rendered PDF with OCR layers grafted on
|
||||
- `pdfa.pdf` - `graft_layers.pdf` after conversion to PDF/A
|
||||
- `pdfa.ps` - a PostScript file used by Ghostscript for PDF/A conversion
|
||||
- `optimize.pdf` - the PDF generated before optimization
|
||||
- `optimize.out.pdf` - the PDF generated by optimization
|
||||
- `origin` - the input file
|
||||
- `origin.pdf` - the input file or the input image converted to PDF
|
||||
- `images/*` - images extracted during the optimization process; here
|
||||
the prefix indicates a PDF object ID not a page number
|
||||
@@ -1,343 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
=================
|
||||
Advanced features
|
||||
=================
|
||||
|
||||
Control of unpaper
|
||||
==================
|
||||
|
||||
OCRmyPDF uses ``unpaper`` to provide the implementation of the
|
||||
``--clean`` and ``--clean-final`` arguments.
|
||||
`unpaper <https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md>`__
|
||||
provides a variety of image processing filters to improve images.
|
||||
|
||||
By default, OCRmyPDF uses only ``unpaper`` arguments that were found to
|
||||
be safe to use on almost all files without having to inspect every page
|
||||
of the file afterwards. This is particularly true when only ``--clean``
|
||||
is used, since that instructs OCRmyPDF to only clean the image before
|
||||
OCR and not the final image.
|
||||
|
||||
However, if you wish to use the more aggressive options in ``unpaper``,
|
||||
you may use ``--unpaper-args '...'`` to override the OCRmyPDF's defaults
|
||||
and forward other arguments to unpaper. This option will forward
|
||||
arguments to ``unpaper`` without any knowledge of what that program
|
||||
considers to be valid arguments. The string of arguments must be quoted
|
||||
as shown in the examples below. No filename arguments may be included.
|
||||
OCRmyPDF will assume it can append input and output filename of
|
||||
intermediate images to the ``--unpaper-args`` string.
|
||||
|
||||
In this example, we tell ``unpaper`` to expect two pages of text on a
|
||||
sheet (image), such as occurs when two facing pages of a book are
|
||||
scanned. ``unpaper`` uses this information to deskew each independently
|
||||
and clean up the margins of both.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --clean --clean-final --unpaper-args '--layout double' input.pdf output.pdf
|
||||
ocrmypdf --clean --clean-final --unpaper-args '--layout double --no-noisefilter' input.pdf output.pdf
|
||||
|
||||
.. warning::
|
||||
|
||||
Some ``unpaper`` features will reposition text within the image.
|
||||
``--clean-final`` is recommended to avoid this issue.
|
||||
|
||||
.. warning::
|
||||
|
||||
Some ``unpaper`` features cause multiple input or output files to be
|
||||
consumed or produced. OCRmyPDF requires ``unpaper`` to consume one
|
||||
file and produce one file. An deviation from that condition will
|
||||
result in errors.
|
||||
|
||||
.. note::
|
||||
|
||||
``unpaper`` uses uncompressed PBM/PGM/PPM files for its intermediate
|
||||
files. For large images or documents, it can take a lot of temporary
|
||||
disk space.
|
||||
|
||||
Control of OCR options
|
||||
======================
|
||||
|
||||
OCRmyPDF provides many features to control the behavior of the OCR
|
||||
engine, Tesseract.
|
||||
|
||||
When OCR is skipped
|
||||
-------------------
|
||||
|
||||
If a page in a PDF seems to have text, by default OCRmyPDF will exit
|
||||
without modifying the PDF. This is to ensure that PDFs that were
|
||||
previously OCRed or were "born digital" rather than scanned are not
|
||||
processed.
|
||||
|
||||
If ``--skip-text`` is issued, then no image processing or OCR will be
|
||||
performed on pages that already have text. The page will be copied to
|
||||
the output. This may be useful for documents that contain both "born
|
||||
digital" and scanned content, or to use OCRmyPDF to normalize and
|
||||
convert to PDF/A regardless of their contents.
|
||||
|
||||
If ``--redo-ocr`` is issued, then a detailed text analysis is performed.
|
||||
Text is categorized as either visible or invisible. Invisible text (OCR)
|
||||
is stripped out. Then an image of each page is created with visible text
|
||||
masked out. The page image is sent for OCR, and any additional text is
|
||||
inserted as OCR. If a file contains a mix of text and bitmap images that
|
||||
contain text, OCRmyPDF will locate the additional text in images without
|
||||
disrupting the existing text.
|
||||
|
||||
If ``--force-ocr`` is issued, then all pages will be rasterized to
|
||||
images, discarding any hidden OCR text, and rasterizing any printable
|
||||
text. This is useful for redoing OCR, for fixing OCR text with a damaged
|
||||
character map (text is selectable but not searchable), and destroying
|
||||
redacted information. Any forms and vector graphics will be rasterized
|
||||
as well.
|
||||
|
||||
Time and image size limits
|
||||
--------------------------
|
||||
|
||||
By default, OCRmyPDF permits tesseract to run for three minutes (180
|
||||
seconds) per page. This is usually more than enough time to find all
|
||||
text on a reasonably sized page with modern hardware.
|
||||
|
||||
If a page is skipped, it will be inserted without OCR. If preprocessing
|
||||
was requested, the preprocessed image layer will be inserted.
|
||||
|
||||
If you want to adjust the amount of time spent on OCR, change
|
||||
``--tesseract-timeout``. You can also automatically skip images that
|
||||
exceed a certain number of megapixels with ``--skip-big``. (A 300 DPI,
|
||||
8.5×11" page is 8.4 megapixels.)
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# Allow 300 seconds for OCR; skip any page larger than 50 megapixels
|
||||
ocrmypdf --tesseract-timeout 300 --skip-big 50 bigfile.pdf output.pdf
|
||||
|
||||
Overriding default tesseract
|
||||
----------------------------
|
||||
|
||||
OCRmyPDF checks the system ``PATH`` for the ``tesseract`` binary.
|
||||
|
||||
Some relevant environment variables that influence Tesseract's behavior
|
||||
include:
|
||||
|
||||
.. envvar:: TESSDATA_PREFIX
|
||||
|
||||
Overrides the path to Tesseract's data files. This can allow
|
||||
simultaneous installation of the "best" and "fast" training data
|
||||
sets. OCRmyPDF does not manage this environment variable.
|
||||
|
||||
.. envvar:: OMP_THREAD_LIMIT
|
||||
|
||||
Controls the number of threads Tesseract will use. OCRmyPDF will
|
||||
manage this environment variable if it is not already set.
|
||||
|
||||
For example, if you have a development build of Tesseract don't wish to
|
||||
use the system installation, you can launch OCRmyPDF as follows:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
env \
|
||||
PATH=/home/user/src/tesseract/api:$PATH \
|
||||
TESSDATA_PREFIX=/home/user/src/tesseract \
|
||||
ocrmypdf input.pdf output.pdf
|
||||
|
||||
In this example ``TESSDATA_PREFIX`` is required to redirect Tesseract to
|
||||
an alternate folder for its "tessdata" files.
|
||||
|
||||
Overriding other support programs
|
||||
---------------------------------
|
||||
|
||||
In addition to tesseract, OCRmyPDF uses the following external binaries:
|
||||
|
||||
- ``gs`` (Ghostscript)
|
||||
- ``unpaper``
|
||||
- ``pngquant``
|
||||
- ``jbig2``
|
||||
|
||||
In each case OCRmyPDF will search the ``PATH`` environment variable to
|
||||
locate the binaries.
|
||||
|
||||
Changing tesseract configuration variables
|
||||
------------------------------------------
|
||||
|
||||
You can override tesseract's default `control
|
||||
parameters <https://tesseract-ocr.github.io/tessdoc/tess3/ControlParams.html>`__
|
||||
with a configuration file.
|
||||
|
||||
As an example, this configuration will disable Tesseract's dictionary
|
||||
for current language. Normally the dictionary is helpful for
|
||||
interpolating words that are unclear, but it may interfere with OCR if
|
||||
the document does not contain many words (for example, a list of part
|
||||
numbers).
|
||||
|
||||
Create a file named "no-dict.cfg" with these contents:
|
||||
|
||||
::
|
||||
|
||||
load_system_dawg 0
|
||||
language_model_penalty_non_dict_word 0
|
||||
language_model_penalty_non_freq_dict_word 0
|
||||
|
||||
then run ocrmypdf as follows (along with any other desired arguments):
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --tesseract-config no-dict.cfg input.pdf output.pdf
|
||||
|
||||
.. warning::
|
||||
|
||||
Some combinations of control parameters will break Tesseract or break
|
||||
assumptions that OCRmyPDF makes about Tesseract's output.
|
||||
|
||||
Changing the PDF renderer
|
||||
=========================
|
||||
|
||||
rasterizing
|
||||
Converting a PDF to an image for display.
|
||||
|
||||
rendering
|
||||
Creating a new PDF from other data (such as an existing PDF).
|
||||
|
||||
OCRmyPDF has these PDF renderers: ``sandwich`` and ``hocr``. The
|
||||
renderer may be selected using ``--pdf-renderer``. The default is
|
||||
``auto`` which lets OCRmyPDF select the renderer to use. Currently,
|
||||
``auto`` always selects ``sandwich``.
|
||||
|
||||
The ``sandwich`` renderer
|
||||
-------------------------
|
||||
|
||||
The ``sandwich`` renderer uses Tesseract's new text-only PDF feature,
|
||||
which produces a PDF page that lays out the OCR in invisible text. This
|
||||
page is then "sandwiched" onto the original PDF page, allowing lossless
|
||||
application of OCR even to PDF pages that contain other vector objects.
|
||||
|
||||
Currently this is the best renderer for most uses, however it is
|
||||
implemented in Tesseract so OCRmyPDF cannot influence it. Currently some
|
||||
problematic PDF viewers like Mozilla PDF.js and macOS Preview have
|
||||
problems with segmenting its text output, and
|
||||
mightrunseveralwordstogether.
|
||||
|
||||
When image preprocessing features like ``--deskew`` are used, the
|
||||
original PDF will be rendered as a full page and the OCR layer will be
|
||||
placed on top.
|
||||
|
||||
The ``hocr`` renderer
|
||||
---------------------
|
||||
|
||||
The ``hocr`` renderer works with older versions of Tesseract. The image
|
||||
layer is copied from the original PDF page if possible, avoiding
|
||||
potentially lossy transcoding or loss of other PDF information. If
|
||||
preprocessing is specified, then the image layer is a new PDF. (You may
|
||||
need to disable PDF/A conversion nad optimization to eliminate all
|
||||
lossy transformations.)
|
||||
|
||||
Unlike ``sandwich`` this renderer is implemented within OCRmyPDF; anyone
|
||||
looking to customize how OCR is presented should look here. A major
|
||||
disadvantage of this renderer is it not capable of correctly handling
|
||||
text outside the Latin alphabet (specifically, it supports the ISO 8859-1
|
||||
character). Pull requests to improve the situation are welcome.
|
||||
|
||||
Currently, this renderer has the best compatibility with Mozilla's
|
||||
PDF.js viewer.
|
||||
|
||||
This works in all versions of Tesseract.
|
||||
|
||||
Return code policy
|
||||
==================
|
||||
|
||||
OCRmyPDF writes all messages to ``stderr``. ``stdout`` is reserved for
|
||||
piping output files. ``stdin`` is reserved for piping input files.
|
||||
|
||||
The return codes generated by the OCRmyPDF are considered part of the
|
||||
stable user interface. They may be imported from
|
||||
``ocrmypdf.exceptions``.
|
||||
|
||||
.. list-table:: Return codes
|
||||
:widths: 5 35 60
|
||||
:header-rows: 1
|
||||
|
||||
* - Code
|
||||
- Name
|
||||
- Interpretation
|
||||
* - 0
|
||||
- ``ExitCode.ok``
|
||||
- Everything worked as expected.
|
||||
* - 1
|
||||
- ``ExitCode.bad_args``
|
||||
- Invalid arguments, exited with an error.
|
||||
* - 2
|
||||
- ``ExitCode.input_file``
|
||||
- The input file does not seem to be a valid PDF.
|
||||
* - 3
|
||||
- ``ExitCode.missing_dependency``
|
||||
- An external program required by OCRmyPDF is missing.
|
||||
* - 4
|
||||
- ``ExitCode.invalid_output_pdf``
|
||||
- An output file was created, but it does not seem to be a valid PDF. The file will be available.
|
||||
* - 5
|
||||
- ``ExitCode.file_access_error``
|
||||
- The user running OCRmyPDF does not have sufficient permissions to read the input file and write the output file.
|
||||
* - 6
|
||||
- ``ExitCode.already_done_ocr``
|
||||
- The file already appears to contain text so it may not need OCR. See output message.
|
||||
* - 7
|
||||
- ``ExitCode.child_process_error``
|
||||
- An error occurred in an external program (child process) and OCRmyPDF cannot continue.
|
||||
* - 8
|
||||
- ``ExitCode.encrypted_pdf``
|
||||
- The input PDF is encrypted. OCRmyPDF does not read encrypted PDFs. Use another program such as ``qpdf`` to remove encryption.
|
||||
* - 9
|
||||
- ``ExitCode.invalid_config``
|
||||
- A custom configuration file was forwarded to Tesseract using ``--tesseract-config``, and Tesseract rejected this file.
|
||||
* - 10
|
||||
- ``ExitCode.pdfa_conversion_failed``
|
||||
- A valid PDF was created, PDF/A conversion failed. The file will be available.
|
||||
* - 15
|
||||
- ``ExitCode.other_error``
|
||||
- Some other error occurred.
|
||||
* - 130
|
||||
- ``ExitCode.ctrl_c``
|
||||
- The program was interrupted by pressing Ctrl+C.
|
||||
|
||||
|
||||
Debugging the intermediate files
|
||||
================================
|
||||
|
||||
OCRmyPDF normally saves its intermediate results to a temporary folder
|
||||
and deletes this folder when it exits, whether it succeeded or failed.
|
||||
|
||||
If the ``-k`` argument is issued on the command line, OCRmyPDF will keep
|
||||
the temporary folder and print the location, whether it succeeded or
|
||||
failed (provided the Python interpreter did not crash). An example
|
||||
message is:
|
||||
|
||||
.. code-block:: none
|
||||
|
||||
Temporary working files retained at:
|
||||
/tmp/ocrmypdf.io.u20wpz07
|
||||
|
||||
The organization of this folder is an implementation detail and subject
|
||||
to change between releases. However the general organization is that
|
||||
working files on a per page basis have the page number as a prefix
|
||||
(starting with page 1), an infix indicates the processing stage, and a
|
||||
suffix indicates the file type. Some important files include:
|
||||
|
||||
- ``_rasterize.png`` - what the input page looks like
|
||||
- ``_ocr.png`` - the file that is sent to Tesseract for OCR; depending
|
||||
on arguments this may differ from the presentation image
|
||||
- ``_pp_deskew.png`` - the image, after deskewing
|
||||
- ``_pp_clean.png`` - the image, after cleaning with unpaper
|
||||
- ``_ocr_tess.pdf`` - the OCR file; appears as a blank page with invisible
|
||||
text embedded
|
||||
- ``_ocr_tess.txt`` - the OCR text (not necessarily all text on the page,
|
||||
if the page is mixed format)
|
||||
- ``fix_docinfo.pdf`` - a temporary file created to fix the PDF DocumentInfo
|
||||
data structure
|
||||
- ``graft_layers.pdf`` - the rendered PDF with OCR layers grafted on
|
||||
- ``pdfa.pdf`` - ``graft_layers.pdf`` after conversion to PDF/A
|
||||
- ``pdfa.ps`` - a PostScript file used by Ghostscript for PDF/A conversion
|
||||
- ``optimize.pdf`` - the PDF generated before optimization
|
||||
- ``optimize.out.pdf`` - the PDF generated by optimization
|
||||
- ``origin`` - the input file
|
||||
- ``origin.pdf`` - the input file or the input image converted to PDF
|
||||
- ``images/*`` - images extracted during the optimization process; here
|
||||
the prefix indicates a PDF object ID not a page number
|
||||
+177
@@ -0,0 +1,177 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Using the OCRmyPDF API
|
||||
|
||||
OCRmyPDF originated as a command line program and continues to have this
|
||||
legacy, but parts of it can be imported and used in other Python
|
||||
applications.
|
||||
|
||||
Some applications may want to consider running ocrmypdf from a
|
||||
subprocess call anyway, as this provides isolation of its activities.
|
||||
|
||||
## Example
|
||||
|
||||
OCRmyPDF provides one high-level function to run its main engine from an
|
||||
application.
|
||||
|
||||
```{versionchanged} 17.0
|
||||
The {func}`ocrmypdf.ocr` function now accepts an {class}`~ocrmypdf.OcrOptions`
|
||||
object as its first argument, providing a cleaner API with full type hints
|
||||
and validation. The previous positional argument style remains supported.
|
||||
```
|
||||
|
||||
### Modern API (recommended)
|
||||
|
||||
The recommended way to call {func}`ocrmypdf.ocr` is to construct an
|
||||
{class}`~ocrmypdf.OcrOptions` object with all settings, then pass it
|
||||
as the sole argument:
|
||||
|
||||
```python
|
||||
import ocrmypdf
|
||||
from ocrmypdf import OcrOptions
|
||||
|
||||
if __name__ == '__main__': # To ensure correct behavior on Windows and macOS
|
||||
options = OcrOptions(
|
||||
input_file='input.pdf',
|
||||
output_file='output.pdf',
|
||||
deskew=True,
|
||||
languages=['eng'],
|
||||
)
|
||||
ocrmypdf.ocr(options)
|
||||
```
|
||||
|
||||
{class}`~ocrmypdf.OcrOptions` is a Pydantic model that provides:
|
||||
|
||||
- Full type hints and IDE autocompletion
|
||||
- Validation of option values at construction time
|
||||
- Clear documentation of all available options
|
||||
|
||||
```{versionadded} 17.0
|
||||
The {class}`~ocrmypdf.OcrOptions` class is now exported from the top-level
|
||||
`ocrmypdf` module.
|
||||
```
|
||||
|
||||
### Legacy API
|
||||
|
||||
For compatibility with OCRmyPDF < v17, the traditional calling style
|
||||
with positional arguments is still fully supported:
|
||||
|
||||
```python
|
||||
import ocrmypdf
|
||||
|
||||
if __name__ == '__main__': # To ensure correct behavior on Windows and macOS
|
||||
ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True)
|
||||
```
|
||||
|
||||
With this style, all of the command line arguments are available
|
||||
and may be passed as equivalent keywords.
|
||||
|
||||
A few differences are that `verbose` and `quiet` are not available.
|
||||
Instead, output should be managed by configuring logging.
|
||||
|
||||
### Parent process requirements
|
||||
|
||||
The {func}`ocrmypdf.ocr` function runs OCRmyPDF similar to command line
|
||||
execution. To do this, it will:
|
||||
|
||||
- create worker processes or threads
|
||||
- manage the signal flags of its worker processes
|
||||
- execute other subprocesses (forking and executing other programs)
|
||||
|
||||
The Python process that calls {func}`ocrmypdf.ocr()` must be sufficiently
|
||||
privileged to perform these actions.
|
||||
|
||||
There currently is no option to manage how jobs are scheduled other
|
||||
than the argument `jobs=` which will limit the number of worker
|
||||
processes.
|
||||
|
||||
Creating a child process to call {func}`ocrmypdf.ocr()` is suggested. That
|
||||
way your application will survive and remain interactive even if
|
||||
OCRmyPDF fails for any reason. For example:
|
||||
|
||||
```python
|
||||
from multiprocessing import Process
|
||||
import ocrmypdf
|
||||
from ocrmypdf import OcrOptions
|
||||
|
||||
def ocrmypdf_process():
|
||||
options = OcrOptions(input_file='input.pdf', output_file='output.pdf')
|
||||
ocrmypdf.ocr(options)
|
||||
|
||||
def call_ocrmypdf_from_my_app():
|
||||
p = Process(target=ocrmypdf_process)
|
||||
p.start()
|
||||
p.join()
|
||||
```
|
||||
|
||||
Programs that call {func}`ocrmypdf.ocr()` should also install a SIGBUS signal
|
||||
handler (except on Windows), to raise an exception if access to a memory
|
||||
mapped file fails. OCRmyPDF may use memory mapping.
|
||||
|
||||
{func}`ocrmypdf.ocr()` will take a threading lock to prevent multiple runs of itself
|
||||
in the same Python interpreter process. This is not thread-safe, because of how
|
||||
OCRmyPDF's plugins and Python's library import system work. If you need to parallelize
|
||||
OCRmyPDF, use processes.
|
||||
|
||||
:::{warning}
|
||||
On Windows and macOS, the script that calls {func}`ocrmypdf.ocr()` must be
|
||||
protected by an "ifmain" guard (`if __name__ == '__main__'`). If you do
|
||||
not take at least one of these steps, process semantics will prevent
|
||||
OCRmyPDF from working correctly.
|
||||
:::
|
||||
|
||||
### Logging
|
||||
|
||||
OCRmyPDF will log under loggers named `ocrmypdf`. In addition, it
|
||||
imports `pdfminer` and `PIL`, both of which post log messages under
|
||||
those logging namespaces.
|
||||
|
||||
You can configure the logging as desired for your application or call
|
||||
{func}`ocrmypdf.configure_logging` to configure logging the same way
|
||||
OCRmyPDF itself does. The command line parameters such as `--quiet`
|
||||
and `--verbose` have no equivalents in the API; you must use the
|
||||
provided configuration function or do configuration in a way that suits
|
||||
your use case.
|
||||
|
||||
### Progress monitoring
|
||||
|
||||
OCRmyPDF uses the `rich` package to implement its progress bars.
|
||||
{func}`ocrmypdf.configure_logging` will set up logging output to
|
||||
`sys.stderr` in a way that is compatible with the display of the
|
||||
progress bar. Use `ocrmypdf.ocr(...progress_bar=False)` to disable
|
||||
the progress bar.
|
||||
|
||||
### Standard output
|
||||
|
||||
OCRmyPDF is strict about not writing to standard output so that
|
||||
users can safely use it in a pipeline and produce a valid output
|
||||
file. A caller application will have to ensure it does not write to
|
||||
standard output either, if it wants to be compatible with this
|
||||
behavior and support piping to a file. Another benefit of running
|
||||
OCRmyPDF in a child process, as recommended above, is that it will
|
||||
not interfere with the parent process's standard output.
|
||||
|
||||
### Exceptions
|
||||
|
||||
OCRmyPDF may throw standard Python exceptions, `ocrmypdf.exceptions.*`
|
||||
exceptions, some exceptions related to multiprocessing, and
|
||||
{exc}`KeyboardInterrupt`. The parent process should provide an exception
|
||||
handler. OCRmyPDF will clean up its temporary files and worker processes
|
||||
automatically when an exception occurs.
|
||||
|
||||
When OCRmyPDF succeeds conditionally, it returns an integer exit code.
|
||||
|
||||
### Plugin Development Changes
|
||||
|
||||
```{versionchanged} 16.13
|
||||
Plugin hooks now receive {class}`~ocrmypdf.OcrOptions` objects instead of
|
||||
`argparse.Namespace`.
|
||||
```
|
||||
|
||||
- {class}`~ocrmypdf.OcrOptions` provides the same attribute access as `Namespace` (duck-typing compatible)
|
||||
- Plugin developers should update type hints: `from ocrmypdf import OcrOptions`
|
||||
- Built-in plugins no longer modify options in-place for better immutability
|
||||
|
||||
Most existing plugins will continue working without modification due to the
|
||||
duck-typing compatibility between {class}`~ocrmypdf.OcrOptions` and `Namespace`.
|
||||
-121
@@ -1,121 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
======================
|
||||
Using the OCRmyPDF API
|
||||
======================
|
||||
|
||||
OCRmyPDF originated as a command line program and continues to have this
|
||||
legacy, but parts of it can be imported and used in other Python
|
||||
applications.
|
||||
|
||||
Some applications may want to consider running ocrmypdf from a
|
||||
subprocess call anyway, as this provides isolation of its activities.
|
||||
|
||||
Example
|
||||
=======
|
||||
|
||||
OCRmyPDF provides one high-level function to run its main engine from an
|
||||
application. The parameters are symmetric to the command line arguments
|
||||
and largely have the same functions.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
if __name__ == '__main__': # To ensure correct behavior on Windows and macOS
|
||||
ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True)
|
||||
|
||||
With some exceptions, all of the command line arguments are available
|
||||
and may be passed as equivalent keywords.
|
||||
|
||||
A few differences are that ``verbose`` and ``quiet`` are not available.
|
||||
Instead, output should be managed by configuring logging.
|
||||
|
||||
Parent process requirements
|
||||
---------------------------
|
||||
|
||||
The :func:`ocrmypdf.ocr` function runs OCRmyPDF similar to command line
|
||||
execution. To do this, it will:
|
||||
|
||||
- create a monitoring thread
|
||||
- create worker processes (on Linux, forking itself; on Windows and macOS, by
|
||||
spawning)
|
||||
- manage the signal flags of its worker processes
|
||||
- execute other subprocesses (forking and executing other programs)
|
||||
|
||||
The Python process that calls :func:`ocrmypdf.ocr()` must be sufficiently
|
||||
privileged to perform these actions.
|
||||
|
||||
There currently is no option to manage how jobs are scheduled other
|
||||
than the argument ``jobs=`` which will limit the number of worker
|
||||
processes.
|
||||
|
||||
Creating a child process to call :func:`ocrmypdf.ocr()` is suggested. That
|
||||
way your application will survive and remain interactive even if
|
||||
OCRmyPDF fails for any reason.
|
||||
|
||||
Programs that call :func:`ocrmypdf.ocr()` should also install a SIGBUS signal
|
||||
handler (except on Windows), to raise an exception if access to a memory
|
||||
mapped file fails. OCRmyPDF may use memory mapping.
|
||||
|
||||
:func:`ocrmypdf.ocr()` will take a threading lock to prevent multiple runs of itself
|
||||
in the same Python interpreter process. This is not thread-safe, because of how
|
||||
OCRmyPDF's plugins and Python's library import system work. If you need to parallelize
|
||||
OCRmyPDF, use processes.
|
||||
|
||||
.. warning::
|
||||
|
||||
On Windows and macOS, the script that calls :func:`ocrmypdf.ocr()` must be
|
||||
protected by an "ifmain" guard (``if __name__ == '__main__'``). If you do
|
||||
not take at least one of these steps, process semantics will prevent
|
||||
OCRmyPDF from working correctly.
|
||||
|
||||
Logging
|
||||
-------
|
||||
|
||||
OCRmyPDF will log under loggers named ``ocrmypdf``. In addition, it
|
||||
imports ``pdfminer`` and ``PIL``, both of which post log messages under
|
||||
those logging namespaces.
|
||||
|
||||
You can configure the logging as desired for your application or call
|
||||
:func:`ocrmypdf.configure_logging` to configure logging the same way
|
||||
OCRmyPDF itself does. The command line parameters such as ``--quiet``
|
||||
and ``--verbose`` have no equivalents in the API; you must use the
|
||||
provided configuration function or do configuration in a way that suits
|
||||
your use case.
|
||||
|
||||
Progress monitoring
|
||||
-------------------
|
||||
|
||||
OCRmyPDF uses the ``tqdm`` package to implement its progress bars.
|
||||
:func:`ocrmypdf.configure_logging` will set up logging output to
|
||||
``sys.stderr`` in a way that is compatible with the display of the
|
||||
progress bar. Use ``ocrmypdf.ocr(...progress_bar=False)`` to disable
|
||||
the progress bar.
|
||||
|
||||
Exceptions
|
||||
----------
|
||||
|
||||
OCRmyPDF may throw standard Python exceptions, ``ocrmypdf.exceptions.*``
|
||||
exceptions, some exceptions related to multiprocessing, and
|
||||
:exc:`KeyboardInterrupt`. The parent process should provide an exception
|
||||
handler. OCRmyPDF will clean up its temporary files and worker processes
|
||||
automatically when an exception occurs.
|
||||
|
||||
Programs that call OCRmyPDF should consider trapping KeyboardInterrupt
|
||||
so that they allow OCR to terminate with the whole program terminating.
|
||||
|
||||
When OCRmyPDF succeeds conditionally, it returns an integer exit code.
|
||||
|
||||
Reference
|
||||
---------
|
||||
|
||||
.. autofunction:: ocrmypdf.ocr
|
||||
|
||||
.. autoclass:: ocrmypdf.Verbosity
|
||||
:members:
|
||||
:undoc-members:
|
||||
|
||||
.. autofunction:: ocrmypdf.configure_logging
|
||||
@@ -0,0 +1,67 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# API reference
|
||||
|
||||
This page summarizes the rest of the public API. Generally speaking this
|
||||
should be mainly of interest to plugin developers.
|
||||
|
||||
## ocrmypdf.api
|
||||
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf.api
|
||||
:members:
|
||||
```
|
||||
|
||||
## ocrmypdf._options
|
||||
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf._options
|
||||
:members: OcrOptions
|
||||
```
|
||||
|
||||
## ocrmypdf.exceptions
|
||||
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf.exceptions
|
||||
:members:
|
||||
:undoc-members:
|
||||
```
|
||||
|
||||
## ocrmypdf.helpers
|
||||
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf.helpers
|
||||
:members:
|
||||
:noindex: deprecated
|
||||
|
||||
.. autodecorator:: deprecated
|
||||
```
|
||||
|
||||
## ocrmypdf.hocrtransform
|
||||
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf.hocrtransform
|
||||
:members:
|
||||
```
|
||||
|
||||
## ocrmypdf.pdfa
|
||||
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf.pdfa
|
||||
:members:
|
||||
```
|
||||
|
||||
## ocrmypdf.quality
|
||||
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf.quality
|
||||
:members:
|
||||
```
|
||||
|
||||
## ocrmypdf.subprocess
|
||||
|
||||
```{eval-rst}
|
||||
.. automodule:: ocrmypdf.subprocess
|
||||
:members:
|
||||
```
|
||||
@@ -1,59 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
=============
|
||||
API Reference
|
||||
=============
|
||||
|
||||
This page summarizes the rest of the public API. Generally speaking this
|
||||
should mainly of interest to plugin developers.
|
||||
|
||||
ocrmypdf
|
||||
========
|
||||
|
||||
.. autoclass:: ocrmypdf.PageContext
|
||||
:members:
|
||||
|
||||
.. autoclass:: ocrmypdf.PdfContext
|
||||
:members:
|
||||
|
||||
ocrmypdf.exceptions
|
||||
===================
|
||||
|
||||
.. automodule:: ocrmypdf.exceptions
|
||||
:members:
|
||||
:undoc-members:
|
||||
|
||||
ocrmypdf.helpers
|
||||
================
|
||||
|
||||
.. automodule:: ocrmypdf.helpers
|
||||
:members:
|
||||
:noindex: deprecated
|
||||
|
||||
.. autodecorator:: deprecated
|
||||
|
||||
ocrmypdf.hocrtransform
|
||||
======================
|
||||
|
||||
.. automodule:: ocrmypdf.hocrtransform
|
||||
:members:
|
||||
|
||||
ocrmypdf.pdfa
|
||||
=============
|
||||
|
||||
.. automodule:: ocrmypdf.pdfa
|
||||
:members:
|
||||
|
||||
ocrmypdf.quality
|
||||
================
|
||||
|
||||
.. automodule:: ocrmypdf.quality
|
||||
:members:
|
||||
|
||||
ocrmypdf.subprocess
|
||||
===================
|
||||
|
||||
.. automodule:: ocrmypdf.subprocess
|
||||
:members:
|
||||
+256
@@ -0,0 +1,256 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
Batch processing
|
||||
================
|
||||
|
||||
This article provides information about running OCRmyPDF on multiple
|
||||
files or configuring it as a service triggered by file system events.
|
||||
|
||||
Batch jobs
|
||||
----------
|
||||
|
||||
Consider using the excellent [GNU
|
||||
Parallel](https://www.gnu.org/software/parallel/) to apply OCRmyPDF to
|
||||
multiple files at once.
|
||||
|
||||
Both `parallel` and `ocrmypdf` will try to use all available processors.
|
||||
To maximize parallelism without overloading your system with processes,
|
||||
consider using `parallel -j 2` to limit parallel to running two jobs at
|
||||
once.
|
||||
|
||||
This command will run `ocrmypdf` on all files named `*.pdf` in the
|
||||
current directory and write them to the previously created `output/`
|
||||
folder. It will not search subdirectories.
|
||||
|
||||
The `--tag` argument tells parallel to print the filename as a prefix
|
||||
whenever a message is printed, so that one can trace any errors to the
|
||||
file that produced them.
|
||||
|
||||
:::{code} bash
|
||||
parallel --tag -j 2 ocrmypdf '{}' 'output/{}' ::: *.pdf
|
||||
:::
|
||||
|
||||
OCRmyPDF automatically repairs PDFs before parsing and gathering
|
||||
information from them.
|
||||
|
||||
Directory trees
|
||||
---------------
|
||||
|
||||
This will walk through a directory tree and run OCR on all files in
|
||||
place, and printing each filename in between runs:
|
||||
|
||||
:::{code} bash
|
||||
find . -name '*.pdf' -printf '%p\n' -exec ocrmypdf '{}' '{}' \;
|
||||
:::
|
||||
|
||||
This only runs one `ocrmypdf` process at a time. This variation uses
|
||||
`find` to create a directory list and `parallel` to parallelize runs of
|
||||
`ocrmypdf`, again updating files in place.
|
||||
|
||||
:::{code} bash
|
||||
find . -name '*.pdf' | parallel --tag -j 2 ocrmypdf '{}' '{}'
|
||||
:::
|
||||
|
||||
In a Windows batch file, use
|
||||
|
||||
:::{code} bat
|
||||
for /r %%f in (*.pdf) do ocrmypdf %%f %%f
|
||||
:::
|
||||
|
||||
With a Docker container, you will need to stream through standard input
|
||||
and output:
|
||||
|
||||
:::{code} bash
|
||||
find . -name '*.pdf' -print0 | xargs -0 | while read pdf; do
|
||||
pdfout=$(mktemp)
|
||||
docker run --rm -i jbarlow83/ocrmypdf - - <$pdf >$pdfout && cp $pdfout $pdf
|
||||
done
|
||||
:::
|
||||
|
||||
### Sample script
|
||||
|
||||
This user contributed script also provides an example of batch
|
||||
processing.
|
||||
|
||||
:::{literalinclude} ../misc/batch.py
|
||||
---
|
||||
caption: misc/batch.py
|
||||
---
|
||||
:::
|
||||
|
||||
### Synology DiskStations
|
||||
|
||||
Synology DiskStations (Network Attached Storage devices) can run the
|
||||
Docker image of OCRmyPDF if the Synology [Docker
|
||||
package](https://www.synology.com/en-global/dsm/packages/Docker) is
|
||||
installed. Attached is a script to address particular quirks of using
|
||||
OCRmyPDF on one of these devices.
|
||||
|
||||
At the time this script was written, it only worked for x86-based
|
||||
Synology products. It is not known if it will work on ARM-based Synology
|
||||
products. Further adjustments might be needed to deal with the
|
||||
Synology\'s relatively limited CPU and RAM.
|
||||
|
||||
:::{literalinclude} ../misc/synology.py
|
||||
---
|
||||
caption: misc/synology.py - Sample script for Synology DiskStations
|
||||
---
|
||||
:::
|
||||
|
||||
### Huge batch jobs
|
||||
|
||||
If you have thousands of files to work with, contact the author.
|
||||
Consulting work related to OCRmyPDF helps fund this open source project
|
||||
and all inquiries are appreciated.
|
||||
|
||||
Hot (watched) folders
|
||||
---------------------
|
||||
|
||||
### Watched folders with watcher.py
|
||||
|
||||
OCRmyPDF has a folder watcher called watcher.py, which is currently
|
||||
included in source distributions but not part of the main program. It
|
||||
may be used natively or may run in a Docker container. Native instances
|
||||
tend to give better performance. watcher.py works on all platforms.
|
||||
|
||||
Users may need to customize the script to meet their requirements.
|
||||
|
||||
:::{code} bash
|
||||
# Using uv (recommended)
|
||||
uv sync --extra watcher
|
||||
|
||||
# Or using pip
|
||||
pip3 install ocrmypdf[watcher]
|
||||
|
||||
env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \
|
||||
OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \
|
||||
OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
||||
python3 watcher.py
|
||||
:::
|
||||
|
||||
:::{list-table} watcher.py environment variables
|
||||
---
|
||||
header-rows: 1
|
||||
---
|
||||
|
||||
* - Environment variable
|
||||
- Description
|
||||
* - OCR\_INPUT\_DIRECTORY
|
||||
- Set input directory to monitor (recursive)
|
||||
* - OCR\_OUTPUT\_DIRECTORY
|
||||
- Set output directory (should not be under input)
|
||||
* - OCR\_ARCHIVE\_DIRECTORY
|
||||
- Set archive directory for processed originals (should not be under input, requires `OCR_ON_SUCCESS_ARCHIVE` to be set)
|
||||
* - OCR\_ON\_SUCCESS\_DELETE
|
||||
- This will move the processed original file to `OCR_ARCHIVE_DIRECTORY` if the exit code is 0 (OK). Note that `OCR_ON_SUCCESS_DELETE` takes precedence over this option, i.e. if both options are set, the input file will be deleted.
|
||||
* - OCR\_OUTPUT\_DIRECTORY\_YEAR\_MONTH
|
||||
- This will place files in the output in `{output}/{year}/{month}/{filename}`
|
||||
* - OCR\_DESKEW
|
||||
- Apply deskew to crooked input PDFs
|
||||
* - OCR\_JSON\_SETTINGS
|
||||
- A JSON string specifying any other arguments for `ocrmypdf.ocr`, e.g. `'OCR_JSON_SETTINGS={"rotate_pages": true, "optimize": "3"}'`.
|
||||
* - OCR\_POLL\_NEW\_FILE\_SECONDS
|
||||
- Polling interval
|
||||
* - OCR\_LOGLEVEL
|
||||
- Level of log messages t
|
||||
:::
|
||||
|
||||
One could configure a networked scanner or scanning computer to drop
|
||||
files in the watched folder.
|
||||
|
||||
### Watched folders with Docker
|
||||
|
||||
The watcher service is included in the OCRmyPDF Docker image. To run it:
|
||||
|
||||
:::{code} bash
|
||||
docker run \
|
||||
--volume <path to files to convert>:/input \
|
||||
--volume <path to store results>:/output \
|
||||
--volume <path to store processed originals>:/processed \
|
||||
--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
||||
--env OCR_ON_SUCCESS_ARCHIVE=1 \
|
||||
--env OCR_DESKEW=1 \
|
||||
--env PYTHONUNBUFFERED=1 \
|
||||
--interactive --tty --entrypoint python3 \
|
||||
jbarlow83/ocrmypdf \
|
||||
watcher.py
|
||||
:::
|
||||
|
||||
This service will watch for a file that matches `/input/\*.pdf`, convert
|
||||
it to a OCRed PDF in `/output/`, and move the processed original to
|
||||
`/processed`. The parameters to this image are:
|
||||
|
||||
:::{list-table} Watcher Docker Parameters
|
||||
:header-rows: 1
|
||||
|
||||
* - Parameter
|
||||
- Description
|
||||
* - `--volume <path to files to convert>:/input`
|
||||
- Files placed in this location will be OCRed
|
||||
* - `--volume <path to store results>:/output`
|
||||
- This is where OCRed files will be stored
|
||||
* - `--volume <path to store processed originals>:/processed`
|
||||
- Archive processed originals here
|
||||
* - `--env OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`
|
||||
- Define environment variable `OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1` to place files in the output in `{output}/{year}/{month}/{filename}`
|
||||
* - `--env OCR_ON_SUCCESS_ARCHIVE=1`
|
||||
- Define environment variable `OCR_ON_SUCCESS_ARCHIVE` to move processed originals
|
||||
* - `--env OCR_DESKEW=1`
|
||||
- Define environment variable `OCR_DESKEW` to apply deskew to crooked input PDFs
|
||||
* - `--env PYTHONBUFFERED=1`
|
||||
- This will force `STDOUT` to be unbuffered and allow you to see messages in docker logs
|
||||
* - `--env OCR_LOGLEVEL='DEBUG'`
|
||||
- Level of log messages
|
||||
* - `--env OCR_JSON_SETTINGS={"language":"deu+eng", "rotate_pages": true}`
|
||||
- A JSON string specifying any other arguments for `ocrmypdf.ocr`
|
||||
:::
|
||||
|
||||
This service relies on polling to check for changes to the filesystem.
|
||||
It may not be suitable for some environments, such as filesystems shared
|
||||
on a slow network.
|
||||
|
||||
A configuration manager such as Docker Compose could be used to ensure
|
||||
that the service is always available.
|
||||
|
||||
:::{literalinclude} ../misc/docker-compose.example.yml
|
||||
---
|
||||
caption: misc/docker-compose.example.yml
|
||||
---
|
||||
:::
|
||||
|
||||
### Caveats
|
||||
|
||||
- `watchmedo` may not work properly on a networked file system,
|
||||
depending on the capabilities of the file system client and server.
|
||||
- This simple recipe does not filter for the type of file system
|
||||
event, so file copies, deletes and moves, and directory operations,
|
||||
will all be sent to ocrmypdf, producing errors in several cases.
|
||||
Disable your watched folder if you are doing anything other than
|
||||
copying files to it.
|
||||
- If the source and destination directory are the same, watchmedo may
|
||||
create an infinite loop.
|
||||
- On BSD, FreeBSD and older versions of macOS, you may need to
|
||||
increase the number of file descriptors to monitor more files, using
|
||||
`ulimit -n 1024` to watch a folder of up to 1024 files.
|
||||
|
||||
### Alternatives
|
||||
|
||||
- On Linux, [systemd user
|
||||
services](https://wiki.archlinux.org/index.php/Systemd/User) can be
|
||||
configured to automatically perform OCR on a collection of files.
|
||||
- [Watchman](https://facebook.github.io/watchman/) is a more powerful
|
||||
alternative to `watchmedo`.
|
||||
|
||||
macOS Automator
|
||||
---------------
|
||||
|
||||
You can use the Automator app with macOS, to create a Workflow or Quick
|
||||
Action. Use a *Run Shell Script* action in your workflow. In the context
|
||||
of Automator, the `PATH` may be set differently your Terminal\'s `PATH`;
|
||||
you may need to explicitly set the PATH to include `ocrmypdf`. The
|
||||
following example may serve as a starting point:
|
||||
|
||||

|
||||
|
||||
You may customize the command sent to ocrmypdf.
|
||||
-229
@@ -1,229 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
================
|
||||
Batch processing
|
||||
================
|
||||
|
||||
This article provides information about running OCRmyPDF on multiple
|
||||
files or configuring it as a service triggered by file system events.
|
||||
|
||||
Batch jobs
|
||||
==========
|
||||
|
||||
Consider using the excellent `GNU
|
||||
Parallel <https://www.gnu.org/software/parallel/>`__ to apply OCRmyPDF
|
||||
to multiple files at once.
|
||||
|
||||
Both ``parallel`` and ``ocrmypdf`` will try to use all available
|
||||
processors. To maximize parallelism without overloading your system with
|
||||
processes, consider using ``parallel -j 2`` to limit parallel to running
|
||||
two jobs at once.
|
||||
|
||||
This command will run ``ocrmypdf`` on all files named ``*.pdf`` in the
|
||||
current directory and write them to the previously created ``output/``
|
||||
folder. It will not search subdirectories.
|
||||
|
||||
The ``--tag`` argument tells parallel to print the filename as a prefix
|
||||
whenever a message is printed, so that one can trace any errors to the
|
||||
file that produced them.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
parallel --tag -j 2 ocrmypdf '{}' 'output/{}' ::: *.pdf
|
||||
|
||||
OCRmyPDF automatically repairs PDFs before parsing and gathering
|
||||
information from them.
|
||||
|
||||
Directory trees
|
||||
===============
|
||||
|
||||
This will walk through a directory tree and run OCR on all files in
|
||||
place, and printing each filename in between runs:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
find . -printf '%p\n' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
|
||||
|
||||
Alternatively, with a Docker container and streaming the file through
|
||||
standard input and output:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
find . -name '*.pdf' -print0 | xargs -0 | while read pdf; do
|
||||
pdfout=$(mktemp)
|
||||
docker run --rm -i jbarlow83/ocrmypdf - - <$pdf >$pdfout && cp $pdfout $pdf
|
||||
done
|
||||
|
||||
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
||||
``find`` to create a directory list and ``parallel`` to parallelize runs
|
||||
of ``ocrmypdf``, again updating files in place.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
find . -name '*.pdf' | parallel --tag -j 2 ocrmypdf '{}' '{}'
|
||||
|
||||
In a Windows batch file, use
|
||||
|
||||
.. code-block:: bat
|
||||
|
||||
for /r %%f in (*.pdf) do ocrmypdf %%f %%f
|
||||
|
||||
Sample script
|
||||
-------------
|
||||
|
||||
This user contributed script also provides an example of batch
|
||||
processing.
|
||||
|
||||
.. literalinclude:: ../misc/batch.py
|
||||
:caption: misc/batch.py
|
||||
|
||||
Synology DiskStations
|
||||
---------------------
|
||||
|
||||
Synology DiskStations (Network Attached Storage devices) can run the
|
||||
Docker image of OCRmyPDF if the Synology `Docker
|
||||
package <https://www.synology.com/en-global/dsm/packages/Docker>`__ is
|
||||
installed. Attached is a script to address particular quirks of using
|
||||
OCRmyPDF on one of these devices.
|
||||
|
||||
This is only possible for x86-based Synology products. Some Synology
|
||||
products use ARM or Power processors and do not support Docker. Further
|
||||
adjustments might be needed to deal with the Synology's relatively
|
||||
limited CPU and RAM.
|
||||
|
||||
.. literalinclude:: ../misc/synology.py
|
||||
:caption: misc/synology.py - Sample script for Synology DiskStations
|
||||
|
||||
Huge batch jobs
|
||||
---------------
|
||||
|
||||
If you have thousands of files to work with, contact the author.
|
||||
Consulting work related to OCRmyPDF helps fund this open source project
|
||||
and all inquiries are appreciated.
|
||||
|
||||
Hot (watched) folders
|
||||
=====================
|
||||
|
||||
Watched folders with watcher.py
|
||||
-------------------------------
|
||||
|
||||
OCRmyPDF has a folder watcher called watcher.py, which is currently included in source
|
||||
distributions but not part of the main program. It may be used natively or may run
|
||||
in a Docker container. Native instances tend to give better performance. watcher.py
|
||||
works on all platforms.
|
||||
|
||||
Users may need to customize the script to meet their requirements.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip3 install ocrmypdf[watcher]
|
||||
|
||||
env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \
|
||||
OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \
|
||||
OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
||||
python3 watcher.py
|
||||
|
||||
.. csv-table:: watcher.py environment variables
|
||||
:header: "Environment variable", "Description"
|
||||
:widths: 50, 50
|
||||
|
||||
"OCR_INPUT_DIRECTORY", "Set input directory to monitor (recursive)"
|
||||
"OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)"
|
||||
"OCR_ARCHIVE_DIRECTORY", "Set archive directory for processed originals (should not be under input, requires ``OCR_ON_SUCCESS_ARCHIVE`` to be set)"
|
||||
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
|
||||
"OCR_ON_SUCCESS_ARCHIVE", "This will move the processed orignal file to ``OCR_ARCHIVE_DIRECTORY`` if the exit code is 0 (OK). Note that ``OCR_ON_SUCCESS_DELETE`` takes precedence over this option, i.e. if both options are set, the input file will be deleted."
|
||||
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
||||
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
|
||||
"OCR_POLL_NEW_FILE_SECONDS", "Polling interval"
|
||||
"OCR_LOGLEVEL", "Level of log messages to report"
|
||||
|
||||
One could configure a networked scanner or scanning computer to drop files in the
|
||||
watched folder.
|
||||
|
||||
Watched folders with Docker
|
||||
---------------------------
|
||||
|
||||
The watcher service is included in the OCRmyPDF Docker image. To run it:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run \
|
||||
-v <path to files to convert>:/input \
|
||||
-v <path to store results>:/output \
|
||||
-v <path to store processed originals>:/archive \
|
||||
-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
||||
-e OCR_ON_SUCCESS_ARCHIVE=1 \
|
||||
-e OCR_DESKEW=1 \
|
||||
-e PYTHONUNBUFFERED=1 \
|
||||
-it --entrypoint python3 \
|
||||
jbarlow83/ocrmypdf \
|
||||
watcher.py
|
||||
|
||||
This service will watch for a file that matches ``/input/\*.pdf``,
|
||||
convert it to a OCRed PDF in ``/output/``, and move the processed
|
||||
original to ``/archive``. The parameters to this image are:
|
||||
|
||||
.. csv-table:: watcher.py parameters for Docker
|
||||
:header: "Parameter", "Description"
|
||||
:widths: 50, 50
|
||||
|
||||
"``-v <path to files to convert>:/input``", "Files placed in this location will be OCRed"
|
||||
"``-v <path to store results>:/output``", "This is where OCRed files will be stored"
|
||||
"``-v <path to store processed originals>:/archive``", "Archive processed originals here"
|
||||
"``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||
"``-e OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals"
|
||||
"``-e OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs"
|
||||
"``-e PYTHONBUFFERED=1``", "This will force ``STDOUT`` to be unbuffered and allow you to see messages in docker logs"
|
||||
|
||||
This service relies on polling to check for changes to the filesystem. It
|
||||
may not be suitable for some environments, such as filesystems shared on a
|
||||
slow network.
|
||||
|
||||
A configuration manager such as Docker Compose could be used to ensure that the
|
||||
service is always available.
|
||||
|
||||
.. literalinclude:: ../misc/docker-compose.example.yml
|
||||
:language: yaml
|
||||
:caption: misc/docker-compose.example.yml
|
||||
|
||||
Caveats
|
||||
-------
|
||||
|
||||
- ``watchmedo`` may not work properly on a networked file system,
|
||||
depending on the capabilities of the file system client and server.
|
||||
- This simple recipe does not filter for the type of file system event,
|
||||
so file copies, deletes and moves, and directory operations, will all
|
||||
be sent to ocrmypdf, producing errors in several cases. Disable your
|
||||
watched folder if you are doing anything other than copying files to
|
||||
it.
|
||||
- If the source and destination directory are the same, watchmedo may
|
||||
create an infinite loop.
|
||||
- On BSD, FreeBSD and older versions of macOS, you may need to increase
|
||||
the number of file descriptors to monitor more files, using
|
||||
``ulimit -n 1024`` to watch a folder of up to 1024 files.
|
||||
|
||||
Alternatives
|
||||
------------
|
||||
|
||||
- On Linux, `systemd user services <https://wiki.archlinux.org/index.php/Systemd/User>`__
|
||||
can be configured to automatically perform OCR on a collection of files.
|
||||
|
||||
- `Watchman <https://facebook.github.io/watchman/>`__ is a more
|
||||
powerful alternative to ``watchmedo``.
|
||||
|
||||
macOS Automator
|
||||
===============
|
||||
|
||||
You can use the Automator app with macOS, to create a Workflow or Quick
|
||||
Action. Use a *Run Shell Script* action in your workflow. In the context
|
||||
of Automator, the ``PATH`` may be set differently your Terminal's
|
||||
``PATH``; you may need to explicitly set the PATH to include
|
||||
``ocrmypdf``. The following example may serve as a starting point:
|
||||
|
||||
.. figure:: images/macos-workflow.png
|
||||
:alt: Example macOS Automator workflow
|
||||
|
||||
You may customize the command sent to ocrmypdf.
|
||||
@@ -0,0 +1,84 @@
|
||||
% SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
(ocr-service)=
|
||||
|
||||
# Online deployments
|
||||
|
||||
OCRmyPDF is designed to be used as a command line tool, but it can be
|
||||
used in a web service. This document describes some considerations for
|
||||
doing so.
|
||||
|
||||
A basic web service implementation is provided in the source code
|
||||
repository, as `misc/webservice.py`. It is only demonstration quality
|
||||
and is not intended for production use.
|
||||
|
||||
OCRmyPDF is not designed for use as a public web service where a
|
||||
malicious user could upload a chosen PDF. In particular, it is not
|
||||
necessarily secure against PDF malware or PDFs that cause denial of
|
||||
service. For further discussino of security, see
|
||||
[security](security).
|
||||
|
||||
OCRmyPDF relies on Ghostscript, and therefore, if deployed online one
|
||||
should be prepared to comply with Ghostscript\'s Affero GPL license, and
|
||||
any other licenses.
|
||||
|
||||
Setting aside these concerns, a side effect of OCRmyPDF is that it may
|
||||
incidentally sanitize PDFs containing certain types of malware. It
|
||||
repairs the PDF with pikepdf/libqpdf, which could correct malformed PDF
|
||||
structures that are part of an attack. When PDF/A output is selected
|
||||
(the default), the input PDF is partially reconstructed by Ghostscript.
|
||||
When `--force-ocr` is used, all pages are rasterized and reconverted to
|
||||
PDF, which could remove malware in embedded images.
|
||||
|
||||
## Limiting CPU usage
|
||||
|
||||
OCRmyPDF will attempt to use all available CPUs and storage, so
|
||||
executing `nice ocrmypdf` or limiting the number of jobs with the
|
||||
`--jobs` argument may ensure the server remains responsive. Another
|
||||
option would be to run OCRmyPDF jobs inside a Docker container, a
|
||||
virtual machine, or a cloud instance, which can impose its own limits on
|
||||
CPU usage and be terminated \"from orbit\" if it fails to complete.
|
||||
|
||||
## Temporary storage requirements
|
||||
|
||||
OCRmyPDF will use a large amount of temporary storage for its work,
|
||||
proportional to the total number of pixels needed to rasterize the PDF.
|
||||
The raster image of a 8.5×11\" color page at 300 DPI takes 25 MB
|
||||
uncompressed; OCRmyPDF saves its intermediates as PNG, but that still
|
||||
means it requires about 9 MB per intermediate based on average
|
||||
compression ratios. Multiple intermediates per page are also required,
|
||||
depending on the command line given. A rule of thumb would be to allow
|
||||
100 MB of temporary storage per page in a file -- meaning that a small
|
||||
cloud servers or small VM partitions should be provisioned with plenty
|
||||
of extra space, if say, a 500 page file might be sent.
|
||||
|
||||
To change the temporary directory, see [tmpdir](#tmpdir).
|
||||
|
||||
On Amazon Web Services or other cloud vendors, consider setting your
|
||||
temporary directory to [empheral
|
||||
storage](https://docs.aws.amazon.com/AWSEC2/latest/UserGuide/InstanceStorage.html).
|
||||
|
||||
## Timeouts
|
||||
|
||||
To prevent excessively long OCR jobs consider setting
|
||||
`--tesseract-timeout` and/or `--skip-big` arguments. `--skip-big` is
|
||||
particularly helpful if your PDFs include documents such as reports on
|
||||
standard page sizes with large images attached - often large images are
|
||||
not worth OCR\'ing anyway.
|
||||
|
||||
## Document management systems
|
||||
|
||||
If you are looking for a full document management system, consider
|
||||
[paperless-ngx](https://github.com/paperless-ngx/paperless-ngx), which
|
||||
is a web application that uses OCRmyPDF to automatically OCR and archive
|
||||
documents.
|
||||
|
||||
## Commercial OCR alternatives
|
||||
|
||||
The author also provides professional services that include OCR and
|
||||
building databases around PDFs, and is happy to provide consultation.
|
||||
|
||||
Abbyy Cloud OCR is viable commercial alternative with a web services
|
||||
API. Amazon Textract, Google Cloud Vision, and Microsoft Azure Computer
|
||||
Vision provide advanced OCR but have less PDF rendering capability.
|
||||
+22
-21
@@ -2,6 +2,8 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# ruff: noqa: E402
|
||||
|
||||
# ocrmypdf documentation build configuration file, created by
|
||||
# sphinx-quickstart on Sun Sep 4 14:29:43 2016.
|
||||
#
|
||||
@@ -22,27 +24,31 @@
|
||||
# import sys
|
||||
# sys.path.insert(0, os.path.abspath('.'))
|
||||
|
||||
"""isort:skip_file"""
|
||||
|
||||
# -- General configuration ------------------------------------------------
|
||||
from __future__ import annotations
|
||||
|
||||
# If your documentation needs a minimal Sphinx version, state it here.
|
||||
#
|
||||
# needs_sphinx = '1.0'
|
||||
needs_sphinx = '8'
|
||||
|
||||
import datetime as dt
|
||||
|
||||
# Add any Sphinx extension module names here, as strings. They can be
|
||||
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
||||
# ones.
|
||||
extensions = [
|
||||
'myst_parser',
|
||||
'sphinx.ext.autodoc',
|
||||
'sphinx.ext.intersphinx',
|
||||
'sphinx.ext.autosummary',
|
||||
'sphinx.ext.napoleon',
|
||||
'sphinx.ext.imgconverter', # PDF docs needs this for SVG to PNG conversion
|
||||
'sphinx_issues',
|
||||
'sphinxcontrib.mermaid',
|
||||
]
|
||||
|
||||
myst_enable_extensions = ['colon_fence', 'attrs_block', 'attrs_inline', 'substitution']
|
||||
|
||||
# Extension settings
|
||||
intersphinx_mapping = {'https://docs.python.org/': None}
|
||||
intersphinx_mapping = {'python': ('https://docs.python.org/3', None)}
|
||||
napoleon_use_rtype = False
|
||||
issues_github_path = "ocrmypdf/OCRmyPDF"
|
||||
|
||||
@@ -50,22 +56,18 @@ issues_github_path = "ocrmypdf/OCRmyPDF"
|
||||
templates_path = ['_templates']
|
||||
|
||||
# The suffix(es) of source filenames.
|
||||
# You can specify multiple suffix as a list of string:
|
||||
#
|
||||
# source_suffix = ['.rst', '.md']
|
||||
source_suffix = '.rst'
|
||||
|
||||
# The encoding of source files.
|
||||
#
|
||||
# source_encoding = 'utf-8-sig'
|
||||
source_suffix = {'.rst': 'restructuredtext', '.md': 'markdown', '.txt': 'markdown'}
|
||||
|
||||
# The master toctree document.
|
||||
master_doc = 'index'
|
||||
|
||||
# General information about the project.
|
||||
project = 'ocrmypdf'
|
||||
|
||||
year = str(dt.date.today().year)
|
||||
copyright = (
|
||||
'2022, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.'
|
||||
f'{year}, James R. Barlow. '
|
||||
+ 'Licensed under Creative Commons Attribution-ShareAlike 4.0'
|
||||
)
|
||||
author = 'James R. Barlow'
|
||||
|
||||
@@ -78,7 +80,6 @@ author = 'James R. Barlow'
|
||||
import os
|
||||
from importlib.metadata import version as package_version
|
||||
|
||||
|
||||
on_rtd = os.environ.get('READTHEDOCS') == 'True'
|
||||
|
||||
if on_rtd:
|
||||
@@ -93,6 +94,7 @@ if on_rtd:
|
||||
|
||||
MOCK_MODULES = [
|
||||
'pikepdf',
|
||||
'pikepdf.canvas',
|
||||
'pikepdf.models',
|
||||
'pikepdf.models.metadata',
|
||||
]
|
||||
@@ -109,7 +111,7 @@ version = '.'.join(release.split('.')[:2])
|
||||
#
|
||||
# This is also used if you do content translation via gettext catalogs.
|
||||
# Usually you set "language" from the command line for these cases.
|
||||
language = None
|
||||
language = 'en'
|
||||
|
||||
# There are two options for replacing |today|: either, you set today to some
|
||||
# non-false value, then it is used:
|
||||
@@ -159,19 +161,18 @@ todo_include_todos = False
|
||||
|
||||
# -- Options for HTML output ----------------------------------------------
|
||||
|
||||
import sphinx_rtd_theme
|
||||
import sphinx_rtd_theme # noqa: F401
|
||||
|
||||
# The theme to use for HTML and HTML Help pages. See the documentation for
|
||||
# a list of builtin themes.
|
||||
#
|
||||
html_theme = 'sphinx_rtd_theme'
|
||||
html_theme_path = [sphinx_rtd_theme.get_html_theme_path()]
|
||||
|
||||
# Theme options are theme-specific and customize the look and feel of a theme
|
||||
# further. For a list of options available for each theme, see the
|
||||
# documentation.
|
||||
#
|
||||
html_theme_options = {'display_version': False}
|
||||
html_theme_options = {}
|
||||
|
||||
# Add any paths that contain custom themes here, relative to this directory.
|
||||
# html_theme_path = []
|
||||
@@ -199,7 +200,7 @@ html_theme_options = {'display_version': False}
|
||||
# Add any paths that contain custom static files (such as style sheets) here,
|
||||
# relative to this directory. They are copied after the builtin static files,
|
||||
# so a file named "default.css" will overwrite the builtin "default.css".
|
||||
html_static_path = ['_static']
|
||||
# html_static_path = ['_static']
|
||||
|
||||
# Add any extra paths that contain custom files (such as robots.txt or
|
||||
# .htaccess) here, relative to this directory. These files are copied
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
% SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Contributing guidelines
|
||||
|
||||
Contributions are welcome!
|
||||
|
||||
## Big changes
|
||||
|
||||
Please open a new issue to discuss or propose a major change. Not only
|
||||
is it fun to discuss big ideas, but we might save each other\'s time
|
||||
too. Perhaps some of the work you\'re contemplating is already half-done
|
||||
in a development branch.
|
||||
|
||||
## Code style
|
||||
|
||||
We use `ruff` for code formatting.
|
||||
The settings for these programs are in `pyproject.toml`. Pull requests
|
||||
should follow the style guide. One difference we use from \"black\"
|
||||
style is that strings shown to the user are always in double quotes
|
||||
(`"`) and strings for internal uses are in single quotes (`'`).
|
||||
|
||||
## Tests
|
||||
|
||||
New features should come with tests that confirm their correctness.
|
||||
|
||||
## New dependencies
|
||||
|
||||
If you are proposing a change that will require a new dependency, we
|
||||
prefer dependencies that are already packaged by Debian or Red Hat. This
|
||||
makes life much easier for our downstream package maintainers. A package
|
||||
that is only available on PyPI or GitHub, and not more widely packaged,
|
||||
may not be accepted.
|
||||
|
||||
We are unlikely to accept a dependency on CUDA or other GPU-based
|
||||
libraries, because these are still difficult to package and install on
|
||||
many systems. We recommend implementing these changes as plugins.
|
||||
|
||||
Python dependencies must also be license-compatible. GPLv3 or AGPLv3 are
|
||||
likely incompatible with the project\'s license, but LGPLv3 is
|
||||
compatible.
|
||||
|
||||
## New non-Python dependencies
|
||||
|
||||
OCRmyPDF uses several external programs (Tesseract, Ghostscript and
|
||||
others) for its functionality. In general we prefer to avoid adding new
|
||||
external programs, and if we are to add external programs, we prefer
|
||||
those that are already packaged by Debian or Red Hat.
|
||||
|
||||
## Plugins
|
||||
|
||||
Some new features may be a good fit for a plugin. Plugins are a way to
|
||||
add features to OCRmyPDF without adding them to the core program.
|
||||
Plugins are installed separately from OCRmyPDF. They are written in
|
||||
Python and can be installed from PyPI. See the [plugin
|
||||
documentation](https://ocrmypdf.readthedocs.io/en/latest/plugins.html).
|
||||
|
||||
We are happy to link users to your plugin from the documentation.
|
||||
|
||||
## Style guide: Is it OCRmyPDF or ocrmypdf?
|
||||
|
||||
The program/project is OCRmyPDF and the name of the executable or
|
||||
library is ocrmypdf.
|
||||
|
||||
## Copyright and license
|
||||
|
||||
For contributions over 10 lines of code, please add your name to list of
|
||||
copyright holders for that file. The core program is licensed under
|
||||
MPL-2.0, test files and documentation under CC-BY-SA 4.0, and
|
||||
miscellaneous files under MIT, with a few minor exceptions. Please
|
||||
contribute only content that you own or have the right to contribute
|
||||
under these licenses.
|
||||
@@ -1,60 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
=======================
|
||||
Contributing guidelines
|
||||
=======================
|
||||
|
||||
Contributions are welcome!
|
||||
|
||||
Big changes
|
||||
===========
|
||||
|
||||
Please open a new issue to discuss or propose a major change. Not only is it fun
|
||||
to discuss big ideas, but we might save each other's time too. Perhaps some of the
|
||||
work you're contemplating is already half-done in a development branch.
|
||||
|
||||
Code style
|
||||
==========
|
||||
|
||||
We use PEP8, ``black`` for code formatting and ``isort`` for import sorting. The
|
||||
settings for these programs are in ``pyproject.toml`` and ``setup.cfg``. Pull
|
||||
requests should follow the style guide. One difference we use from "black" style
|
||||
is that strings shown to the user are always in double quotes (``"``) and strings
|
||||
for internal uses are in single quotes (``'``).
|
||||
|
||||
Tests
|
||||
=====
|
||||
|
||||
New features should come with tests that confirm their correctness.
|
||||
|
||||
New Python dependencies
|
||||
=======================
|
||||
|
||||
If you are proposing a change that will require a new Python dependency, we
|
||||
prefer dependencies that are already packaged by Debian or Red Hat. This makes
|
||||
life much easier for our downstream package maintainers.
|
||||
|
||||
Python dependencies must also be license-compatible. GPLv3 or AGPLv3 are likely
|
||||
incompatible with the project's license, but LGPLv3 is compatible.
|
||||
|
||||
New non-Python dependencies
|
||||
===========================
|
||||
|
||||
OCRmyPDF uses several external programs (Tesseract, Ghostscript and others) for
|
||||
its functionality. In general we prefer to avoid adding new external programs.
|
||||
|
||||
Style guide: Is it OCRmyPDF or ocrmypdf?
|
||||
========================================
|
||||
|
||||
The program/project is OCRmyPDF and the name of the executable or library is ocrmypdf.
|
||||
|
||||
Copyright and license
|
||||
=====================
|
||||
|
||||
For contributions over 10 lines of code, please include your name to list of
|
||||
copyright holders for that file. The core program is licensed under MPL-2.0,
|
||||
test files and documentation under CC-BY-SA 4.0, and miscellaneous files under
|
||||
MIT. Please contribute code only that you wrote and you have the permission to
|
||||
contribute or license to us.
|
||||
@@ -0,0 +1,436 @@
|
||||
% SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Cookbook
|
||||
|
||||
## Basic examples
|
||||
|
||||
### Help!
|
||||
|
||||
ocrmypdf has built-in help.
|
||||
|
||||
```bash
|
||||
ocrmypdf --help
|
||||
```
|
||||
|
||||
### Add an OCR layer and convert to PDF/A
|
||||
|
||||
```bash
|
||||
ocrmypdf input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Add an OCR layer and output a standard PDF
|
||||
|
||||
```bash
|
||||
ocrmypdf --output-type pdf input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Create a PDF/A with all color and grayscale images converted to JPEG
|
||||
|
||||
```bash
|
||||
ocrmypdf --output-type pdfa --pdfa-image-compression jpeg input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Modify a file in place
|
||||
|
||||
The file will only be overwritten if OCRmyPDF is successful.
|
||||
|
||||
```bash
|
||||
ocrmypdf myfile.pdf myfile.pdf
|
||||
```
|
||||
|
||||
### Correct page rotation
|
||||
|
||||
OCR will attempt to automatic correct the rotation of each page. This
|
||||
can help fix a scanning job that contains a mix of landscape and
|
||||
portrait pages.
|
||||
|
||||
```bash
|
||||
ocrmypdf --rotate-pages myfile.pdf myfile.pdf
|
||||
```
|
||||
|
||||
You can increase (decrease) the parameter `--rotate-pages-threshold` to
|
||||
make page rotation more (less) aggressive. The threshold number is the
|
||||
ratio of how confidence the OCR engine is that the document image should
|
||||
be changed, compared to kept the same. The default value is quite
|
||||
conservative; on some files it may not attempt rotations at all unless
|
||||
it is very confident that the current rotation is wrong. A lower value
|
||||
of `2.0` will produce more rotations, and more false positives. Run with
|
||||
`-v1` to see the confidence level for each page to see if there may be a
|
||||
better value for your files.
|
||||
|
||||
If the page is \"just a little off horizontal\", like a crooked picture,
|
||||
then you want `--deskew`. `--rotate-pages` is for when the cardinal
|
||||
angle is wrong.
|
||||
|
||||
### OCR languages other than English
|
||||
|
||||
OCRmyPDF assumes the document is in English unless told otherwise. OCR
|
||||
quality may be poor if the wrong language is used.
|
||||
|
||||
```bash
|
||||
ocrmypdf -l fra LeParisien.pdf LeParisien.pdf
|
||||
ocrmypdf -l eng+fra Bilingual-English-French.pdf Bilingual-English-French.pdf
|
||||
```
|
||||
|
||||
Language packs must be installed for all languages specified. See
|
||||
`Installing additional language packs <lang-packs>`{.interpreted-text
|
||||
role="ref"}.
|
||||
|
||||
Unfortunately, the Tesseract OCR engine has no ability to detect the
|
||||
language when it is unknown.
|
||||
|
||||
### Produce PDF and text file containing OCR text
|
||||
|
||||
This produces a file named \"output.pdf\" and a companion text file
|
||||
named \"output.txt\".
|
||||
|
||||
```bash
|
||||
ocrmypdf --sidecar output.txt input.pdf output.pdf
|
||||
```
|
||||
|
||||
:::{note}
|
||||
The sidecar file contains the **OCR text** found by OCRmyPDF. If the
|
||||
document contains pages that already have text, that text will not
|
||||
appear in the sidecar. If the option `--pages` is used, only those pages
|
||||
on which OCR was performed will be included in the sidecar. If certain
|
||||
pages were skipped because of options like `--skip-big` or
|
||||
`--tesseract-timeout`, those pages will not be in the sidecar.
|
||||
|
||||
If you don\'t want to generate the output PDF, use `--output-type=none`
|
||||
to avoid generating one. Set the output filename to `-` (i.e. redirect
|
||||
to stdout).
|
||||
|
||||
To extract all text from a PDF, whether generated from OCR or otherwise,
|
||||
use a program like Poppler\'s `pdftotext` or `pdfgrep`.
|
||||
:::
|
||||
|
||||
### OCR images, not PDFs
|
||||
|
||||
#### Option: use Tesseract
|
||||
|
||||
If you are starting with images, you can just use Tesseract directly to
|
||||
convert images to PDFs:
|
||||
|
||||
```bash
|
||||
tesseract my-image.jpg output-prefix pdf
|
||||
```
|
||||
|
||||
```bash
|
||||
# When there are multiple images
|
||||
tesseract text-file-containing-list-of-image-filenames.txt output-prefix pdf
|
||||
```
|
||||
|
||||
Tesseract\'s PDF output is quite good -- OCRmyPDF uses it internally, in
|
||||
some cases. However, OCRmyPDF has many features not available in
|
||||
Tesseract like image processing, metadata control, and PDF/A generation.
|
||||
|
||||
#### Option: use img2pdf
|
||||
|
||||
You can also use a program like
|
||||
[img2pdf](https://gitlab.mister-muffin.de/josch/img2pdf) to convert your
|
||||
images to PDFs, and then pipe the results to run ocrmypdf. The `-` tells
|
||||
ocrmypdf to read standard input.
|
||||
|
||||
```bash
|
||||
img2pdf my-images*.jpg | ocrmypdf - myfile.pdf
|
||||
```
|
||||
|
||||
`img2pdf` is recommended because it does an excellent job at generating
|
||||
PDFs without transcoding images.
|
||||
|
||||
#### Option: use OCRmyPDF (single images only)
|
||||
|
||||
For convenience, OCRmyPDF can also convert single images to PDFs on its
|
||||
own. If the resolution (dots per inch, DPI) of an image is not set or is
|
||||
incorrect, it can be overridden with `--image-dpi`. (As 1 inch is 2.54
|
||||
cm, 1 dpi = 0.39 dpcm).
|
||||
|
||||
```bash
|
||||
ocrmypdf --image-dpi 300 image.png myfile.pdf
|
||||
```
|
||||
|
||||
If you have multiple images, you must use `img2pdf` to convert the
|
||||
images to PDF.
|
||||
|
||||
#### Not recommended
|
||||
|
||||
We caution against using ImageMagick or Ghostscript to convert images to
|
||||
PDF, since they may transcode images or produce downsampled images,
|
||||
sometimes without warning.
|
||||
|
||||
(image-processing)=
|
||||
|
||||
## Image processing
|
||||
|
||||
OCRmyPDF perform some image processing on each page of a PDF, if
|
||||
desired. The same processing is applied to each page. It is suggested
|
||||
that the user review files after image processing as these commands
|
||||
might remove desirable content, especially from poor quality scans.
|
||||
|
||||
- `--rotate-pages` attempts to determine the correct orientation for
|
||||
each page and rotates the page if necessary.
|
||||
- `--remove-background` attempts to detect and remove a noisy
|
||||
background from grayscale or color images. Monochrome images are
|
||||
ignored. This should not be used on documents that contain color
|
||||
photos as it may remove them.
|
||||
- `--deskew` will correct pages that were scanned at a skewed angle by
|
||||
rotating them back into place.
|
||||
- `--clean` uses [unpaper](https://www.flameeyes.eu/projects/unpaper)
|
||||
to clean up pages before OCR, but does not alter the final output.
|
||||
This makes it less likely that OCR will try to find text in
|
||||
background noise.
|
||||
- `--clean-final` uses unpaper to clean up pages before OCR and
|
||||
inserts the page into the final output. You will want to review each
|
||||
page to ensure that unpaper did not remove something important.
|
||||
|
||||
:::{note}
|
||||
In many cases image processing will rasterize PDF pages as images,
|
||||
potentially losing quality.
|
||||
:::
|
||||
|
||||
:::{warning}
|
||||
`--clean-final` and `--remove-background` may leave undesirable visual
|
||||
artifacts in some images where their algorithms have shortcomings. Files
|
||||
should be visually reviewed after using these options.
|
||||
:::
|
||||
|
||||
### Example: OCR and correct document skew (crooked scan)
|
||||
|
||||
Deskew:
|
||||
|
||||
```bash
|
||||
ocrmypdf --deskew input.pdf output.pdf
|
||||
```
|
||||
|
||||
Image processing commands can be combined. The order in which options
|
||||
are given does not matter. OCRmyPDF always applies the steps of the
|
||||
image processing pipeline in the same order (rotate, remove background,
|
||||
deskew, clean).
|
||||
|
||||
```bash
|
||||
ocrmypdf --deskew --clean --rotate-pages input.pdf output.pdf
|
||||
```
|
||||
|
||||
Don\'t actually OCR my PDF
|
||||
--------------------------
|
||||
|
||||
If you set `--ocr-engine none` OCRmyPDF will apply its image processing without
|
||||
performing OCR. This works if all you want to is to apply image processing or PDF/A
|
||||
conversion.
|
||||
|
||||
```bash
|
||||
ocrmypdf --ocr-engine none --deskew --output-type pdfa input.pdf output.pdf
|
||||
```
|
||||
|
||||
:::{versionchanged} v17.0.0
|
||||
|
||||
Prior to this version, `--tesseract-timeout 0` was recommended as an idiom
|
||||
to turn off OCR. This is not longer recommended, as we move away from
|
||||
Tesseract OCR as the primary OCR engine.
|
||||
|
||||
:::
|
||||
|
||||
:::{versionchanged} v14.1.0
|
||||
|
||||
Prior to this version, `--tesseract-timeout 0` would prevent other uses
|
||||
of Tesseract, such as deskewing, from working. This is no longer the
|
||||
case. Use `--tesseract-non-ocr-timeout` to control the timeout for
|
||||
non-OCR operations, if needed.
|
||||
:::
|
||||
|
||||
### Remove all text or OCR from my PDF
|
||||
|
||||
This is getting ridiculous, but OCRmyPDF can complete strip all textual
|
||||
information from a PDF and reconstruct it as a \"bag of images\" PDF.
|
||||
|
||||
```bash
|
||||
ocrmypdf --ocr-engine none --force-ocr input.pdf output.pdf
|
||||
```
|
||||
|
||||
Why would you want to do this? Perhaps you have a PDF where OCR fails to
|
||||
produce useful results, and just want to get rid of all OCR information.
|
||||
This command also removes OCR generated by third party tools.
|
||||
|
||||
### Optimize images without performing OCR
|
||||
|
||||
You can also optimize all images without performing any OCR:
|
||||
|
||||
```bash
|
||||
ocrmypdf --ocr-engine none --optimize 3 --skip-text input.pdf output.pdf
|
||||
```
|
||||
|
||||
## Using v17 features
|
||||
|
||||
### Select a rasterizer
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
OCRmyPDF can use pypdfium2 or Ghostscript to rasterize PDF pages. pypdfium2
|
||||
is generally faster and is preferred when available.
|
||||
|
||||
```bash
|
||||
# Automatic selection (default) - prefers pypdfium when available
|
||||
ocrmypdf --rasterizer auto input.pdf output.pdf
|
||||
|
||||
# Explicitly use pypdfium2 (requires pip install pypdfium2)
|
||||
ocrmypdf --rasterizer pypdfium input.pdf output.pdf
|
||||
|
||||
# Explicitly use Ghostscript
|
||||
ocrmypdf --rasterizer ghostscript input.pdf output.pdf
|
||||
```
|
||||
|
||||
### PDF/A without Ghostscript
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
With verapdf installed, OCRmyPDF can produce PDF/A without using Ghostscript
|
||||
for conversion. This is faster and avoids some Ghostscript limitations.
|
||||
|
||||
```bash
|
||||
# Uses speculative conversion with verapdf validation (default)
|
||||
ocrmypdf --output-type auto input.pdf output.pdf
|
||||
|
||||
# Explicitly request Ghostscript-based PDF/A conversion
|
||||
ocrmypdf --output-type pdfa input.pdf output.pdf
|
||||
```
|
||||
|
||||
### Using --mode instead of legacy flags
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
The `--mode` (`-m`) flag consolidates OCR behavior options:
|
||||
|
||||
```bash
|
||||
# Instead of --skip-text
|
||||
ocrmypdf --mode skip input.pdf output.pdf
|
||||
|
||||
# Instead of --force-ocr
|
||||
ocrmypdf --mode force input.pdf output.pdf
|
||||
|
||||
# Instead of --redo-ocr
|
||||
ocrmypdf --mode redo input.pdf output.pdf
|
||||
|
||||
# Short form
|
||||
ocrmypdf -m skip input.pdf output.pdf
|
||||
```
|
||||
|
||||
The legacy flags continue to work as aliases.
|
||||
|
||||
### Process only certain pages
|
||||
|
||||
You can ask OCRmyPDF to only apply [image processing](#image-processing)
|
||||
and OCR to certain pages.
|
||||
|
||||
```bash
|
||||
ocrmypdf --pages 2,3,13-17 input.pdf output.pdf
|
||||
```
|
||||
|
||||
Hyphens denote a range of pages and commas separate page numbers. If you
|
||||
prefer to use spaces, quote all of the page numbers:
|
||||
`--pages '2, 3, 5, 7'`.
|
||||
|
||||
OCRmyPDF will warn if your list of page numbers contains duplicates or
|
||||
overlapping pages. OCRmyPDF does not currently account for document page
|
||||
numbers, such as an introduction section of a book that uses Roman
|
||||
numerals. It simply counts the number of virtual pieces of paper since
|
||||
the start. If your list of pages is out of numerical order, OCRmyPDF
|
||||
will sort it for you.
|
||||
|
||||
Regardless of the argument to `--pages`, OCRmyPDF will optimize all
|
||||
pages/images in the file and convert it to PDF/A, unless you disable
|
||||
those options. Both of these steps are \"whole file\" operations. In
|
||||
this example, we want to OCR only the title and otherwise change the PDF
|
||||
as little as possible:
|
||||
|
||||
```bash
|
||||
ocrmypdf --pages 1 --output-type pdf --optimize 0 input.pdf output.pdf
|
||||
```
|
||||
|
||||
## Redo existing OCR
|
||||
|
||||
To redo OCR on a file OCRed with other OCR software or a previous
|
||||
version of OCRmyPDF and/or Tesseract, you may use the `--redo-ocr`
|
||||
argument. (Normally, OCRmyPDF will exit with an error if asked to modify
|
||||
a file with OCR.)
|
||||
|
||||
This may be helpful for users who want to take advantage of accuracy
|
||||
improvements in Tesseract for files they previously OCRed with an
|
||||
earlier version of Tesseract and OCRmyPDF.
|
||||
|
||||
```bash
|
||||
ocrmypdf --redo-ocr input.pdf output.pdf
|
||||
```
|
||||
|
||||
This method will replace OCR without rasterizing, reducing quality or
|
||||
removing vector content. If a file contains a mix of pure digital text
|
||||
and OCR, digital text will be ignored and OCR will be replaced. As such
|
||||
this mode is incompatible with image processing options, since they
|
||||
alter the appearance of the file.
|
||||
|
||||
In some cases, existing OCR cannot be detected or replaced. Files
|
||||
produced by OCRmyPDF v2.2 or earlier, for example, are internally
|
||||
represented as having visible text with an opaque image drawn on top.
|
||||
This situation cannot be detected.
|
||||
|
||||
If `--redo-ocr` does not work, you can use `--force-ocr`, which will
|
||||
force rasterization of all pages, potentially reducing quality or losing
|
||||
vector content.
|
||||
|
||||
Improving OCR quality
|
||||
---------------------
|
||||
|
||||
The [Image processing](#image-processing) features can improve OCR
|
||||
quality.
|
||||
|
||||
Rotating pages and deskewing helps to ensure that the page orientation
|
||||
is correct before OCR begins. Removing the background and/or cleaning
|
||||
the page can also improve results. The `--oversample DPI` argument can
|
||||
be specified to resample images to higher resolution before attempting
|
||||
OCR; this can improve results as well.
|
||||
|
||||
OCR quality will suffer if the resolution of input images is not correct
|
||||
(since the range of pixel sizes that will be checked for possible fonts
|
||||
will also be incorrect).
|
||||
|
||||
## PDF optimization
|
||||
|
||||
By default OCRmyPDF will attempt to perform lossless optimizations on
|
||||
the images inside PDFs after OCR is complete. Optimization is performed
|
||||
even if no OCR text is found.
|
||||
|
||||
The `--optimize N` (short form `-O`) argument controls optimization,
|
||||
where `N` ranges from 0 to 3 inclusive, analogous to the optimization
|
||||
levels in the GCC compiler. `-O1` is the default.
|
||||
|
||||
For further details, see the section on [PDF optimization](optimizer).
|
||||
|
||||
```bash
|
||||
ocrmypdf --optimize 3 in.pdf out.pdf # Make it small
|
||||
```
|
||||
|
||||
Some users may consider enabling lossy JBIG2. See:
|
||||
`jbig2-lossy`{.interpreted-text role="ref"}.
|
||||
|
||||
:::{note}
|
||||
Image processing and PDF/A conversion can also introduce lossy
|
||||
transformations to your PDF images, even when `--optimize 1` is in use.
|
||||
:::
|
||||
|
||||
Digitally signed PDFs
|
||||
---------------------
|
||||
|
||||
OCRmyPDF cannot preserve digital signatures in PDFs and also add OCR to
|
||||
them. By default, it will refuse to modify a signed PDF regardless of
|
||||
other settings. You can override this behavior with
|
||||
`--invalidate-digital-signatures`; as the name suggests, any digital
|
||||
signatures will be invalidated.
|
||||
|
||||
OCRmyPDF cannot open documents that are encrypted with a digital
|
||||
certificate.
|
||||
|
||||
Versions of OCRmyPDF prior to 14.4.0 would invalidate existing digital
|
||||
signatures without warning.
|
||||
@@ -1,374 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
========
|
||||
Cookbook
|
||||
========
|
||||
|
||||
Basic examples
|
||||
==============
|
||||
|
||||
Help!
|
||||
-----
|
||||
|
||||
ocrmypdf has built-in help.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --help
|
||||
|
||||
Add an OCR layer and convert to PDF/A
|
||||
-------------------------------------
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf input.pdf output.pdf
|
||||
|
||||
Add an OCR layer and output a standard PDF
|
||||
------------------------------------------
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --output-type pdf input.pdf output.pdf
|
||||
|
||||
Create a PDF/A with all color and grayscale images converted to JPEG
|
||||
--------------------------------------------------------------------
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --output-type pdfa --pdfa-image-compression jpeg input.pdf output.pdf
|
||||
|
||||
Modify a file in place
|
||||
----------------------
|
||||
|
||||
The file will only be overwritten if OCRmyPDF is successful.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf myfile.pdf myfile.pdf
|
||||
|
||||
Correct page rotation
|
||||
---------------------
|
||||
|
||||
OCR will attempt to automatic correct the rotation of each page. This
|
||||
can help fix a scanning job that contains a mix of landscape and
|
||||
portrait pages.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --rotate-pages myfile.pdf myfile.pdf
|
||||
|
||||
You can increase (decrease) the parameter ``--rotate-pages-threshold``
|
||||
to make page rotation more (less) aggressive. The threshold number is the ratio
|
||||
of how confidence the OCR engine is that the document image should be changed,
|
||||
compared to kept the same. The default value is quite conservative; on some files
|
||||
it may not attempt rotations at all unless it is very confident that the current
|
||||
rotation is wrong. A lower value of ``2.0`` will produce more rotations, and
|
||||
more false positives. Run with ``-v1`` to see the confidence level for each
|
||||
page to see if there may be a better value for your files.
|
||||
|
||||
If the page is "just a little off horizontal", like a crooked picture,
|
||||
then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal
|
||||
angle is wrong.
|
||||
|
||||
OCR languages other than English
|
||||
--------------------------------
|
||||
|
||||
OCRmyPDF assumes the document is in English unless told otherwise. OCR
|
||||
quality may be poor if the wrong language is used.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf -l fra LeParisien.pdf LeParisien.pdf
|
||||
ocrmypdf -l eng+fra Bilingual-English-French.pdf Bilingual-English-French.pdf
|
||||
|
||||
Language packs must be installed for all languages specified. See
|
||||
:ref:`Installing additional language packs <lang-packs>`.
|
||||
|
||||
Unfortunately, the Tesseract OCR engine has no ability to detect the
|
||||
language when it is unknown.
|
||||
|
||||
Produce PDF and text file containing OCR text
|
||||
---------------------------------------------
|
||||
|
||||
This produces a file named "output.pdf" and a companion text file named
|
||||
"output.txt".
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --sidecar output.txt input.pdf output.pdf
|
||||
|
||||
.. note::
|
||||
|
||||
The sidecar file contains the **OCR text** found by OCRmyPDF. If the document
|
||||
contains pages that already have text, that text will not appear in the
|
||||
sidecar. If the option ``--pages`` is used, only those pages on which OCR
|
||||
was performed will be included in the sidecar. If certain pages were skipped
|
||||
because of options like ``--skip-big`` or ``--tesseract-timeout``, those pages
|
||||
will not be in the sidecar.
|
||||
|
||||
If you don't want to generate the output PDF, use ``--output-type=none`` to
|
||||
avoid generating one. Set the output filename to ``-`` (i.e. redirect to stdout).
|
||||
|
||||
To extract all text from a PDF, whether generated from OCR or otherwise,
|
||||
use a program like Poppler's ``pdftotext`` or ``pdfgrep``.
|
||||
|
||||
OCR images, not PDFs
|
||||
--------------------
|
||||
|
||||
Option: use Tesseract
|
||||
~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
If you are starting with images, you can just use Tesseract directly to
|
||||
convert images to PDFs:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
tesseract my-image.jpg output-prefix pdf
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# When there are multiple images
|
||||
tesseract text-file-containing-list-of-image-filenames.txt output-prefix pdf
|
||||
|
||||
Tesseract's PDF output is quite good – OCRmyPDF uses it internally, in
|
||||
some cases. However, OCRmyPDF has many features not available in
|
||||
Tesseract like image processing, metadata control, and PDF/A generation.
|
||||
|
||||
Option: use img2pdf
|
||||
~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
You can also use a program like
|
||||
`img2pdf <https://gitlab.mister-muffin.de/josch/img2pdf>`__ to convert
|
||||
your images to PDFs, and then pipe the results to run ocrmypdf. The
|
||||
``-`` tells ocrmypdf to read standard input.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
img2pdf my-images*.jpg | ocrmypdf - myfile.pdf
|
||||
|
||||
``img2pdf`` is recommended because it does an excellent job at
|
||||
generating PDFs without transcoding images.
|
||||
|
||||
Option: use OCRmyPDF (single images only)
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
For convenience, OCRmyPDF can also convert single images to PDFs on its
|
||||
own. If the resolution (dots per inch, DPI) of an image is not set or is
|
||||
incorrect, it can be overridden with ``--image-dpi``. (As 1 inch is 2.54
|
||||
cm, 1 dpi = 0.39 dpcm).
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --image-dpi 300 image.png myfile.pdf
|
||||
|
||||
If you have multiple images, you must use ``img2pdf`` to convert the
|
||||
images to PDF.
|
||||
|
||||
Not recommended
|
||||
~~~~~~~~~~~~~~~
|
||||
|
||||
We caution against using ImageMagick or Ghostscript to convert images to
|
||||
PDF, since they may transcode images or produce downsampled images,
|
||||
sometimes without warning.
|
||||
|
||||
Image processing
|
||||
================
|
||||
|
||||
OCRmyPDF perform some image processing on each page of a PDF, if
|
||||
desired. The same processing is applied to each page. It is suggested
|
||||
that the user review files after image processing as these commands
|
||||
might remove desirable content, especially from poor quality scans.
|
||||
|
||||
- ``--rotate-pages`` attempts to determine the correct orientation for
|
||||
each page and rotates the page if necessary.
|
||||
- ``--remove-background`` attempts to detect and remove a noisy
|
||||
background from grayscale or color images. Monochrome images are
|
||||
ignored. This should not be used on documents that contain color
|
||||
photos as it may remove them.
|
||||
- ``--deskew`` will correct pages were scanned at a skewed angle by
|
||||
rotating them back into place.
|
||||
- ``--clean`` uses
|
||||
`unpaper <https://www.flameeyes.eu/projects/unpaper>`__ to clean up
|
||||
pages before OCR, but does not alter the final output. This makes it
|
||||
less likely that OCR will try to find text in background noise.
|
||||
- ``--clean-final`` uses unpaper to clean up pages before OCR and
|
||||
inserts the page into the final output. You will want to review each
|
||||
page to ensure that unpaper did not remove something important.
|
||||
|
||||
.. note::
|
||||
|
||||
In many cases image processing will rasterize PDF pages as images,
|
||||
potentially losing quality.
|
||||
|
||||
.. warning::
|
||||
|
||||
``--clean-final`` and ``--remove-background`` may leave undesirable
|
||||
visual artifacts in some images where their algorithms have
|
||||
shortcomings. Files should be visually reviewed after using these
|
||||
options.
|
||||
|
||||
Example: OCR and correct document skew (crooked scan)
|
||||
-----------------------------------------------------
|
||||
|
||||
Deskew:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --deskew input.pdf output.pdf
|
||||
|
||||
Image processing commands can be combined. The order in which options
|
||||
are given does not matter. OCRmyPDF always applies the steps of the
|
||||
image processing pipeline in the same order (rotate, remove background,
|
||||
deskew, clean).
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --deskew --clean --rotate-pages input.pdf output.pdf
|
||||
|
||||
Don't actually OCR my PDF
|
||||
=========================
|
||||
|
||||
If you set ``--tesseract-timeout 0`` OCRmyPDF will apply its image
|
||||
processing without performing OCR, if all you want to is to apply image
|
||||
processing or PDF/A conversion.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --tesseract-timeout=0 --remove-background input.pdf output.pdf
|
||||
|
||||
Optimize images without performing OCR
|
||||
--------------------------------------
|
||||
|
||||
You can also optimize all images without performing any OCR:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --tesseract-timeout=0 --optimize 3 --skip-text input.pdf output.pdf
|
||||
|
||||
Process only certain pages
|
||||
--------------------------
|
||||
|
||||
You can ask OCRmyPDF to only apply `image processing <#image-processing>`__
|
||||
and OCR to certain pages.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --pages 2,3,13-17 input.pdf output.pdf
|
||||
|
||||
Hyphens denote a range of pages and commas separate page numbers. If you prefer
|
||||
to use spaces, quote all of the page numbers: ``--pages '2, 3, 5, 7'``.
|
||||
|
||||
OCRmyPDF will warn if your list of page numbers contains duplicates or
|
||||
overlap pages. OCRmyPDF does not currently account for document page numbers,
|
||||
such as an introduction section of a book that uses Roman numerals. It simply
|
||||
counts the number of virtual pieces of paper since the start.
|
||||
|
||||
Regardless of the argument to ``--pages``, OCRmyPDF will optimize all pages/images
|
||||
in the file and convert it to PDF/A, unless you disable those options. Both of these
|
||||
steps are "whole file" operations. In this example, we want to OCR only the title
|
||||
and otherwise change the PDF as little as possible:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --pages 1 --output-type pdf --optimize 0 input.pdf output.pdf
|
||||
|
||||
Redo existing OCR
|
||||
=================
|
||||
|
||||
To redo OCR on a file OCRed with other OCR software or a previous
|
||||
version of OCRmyPDF and/or Tesseract, you may use the ``--redo-ocr``
|
||||
argument. (Normally, OCRmyPDF will exit with an error if asked to modify
|
||||
a file with OCR.)
|
||||
|
||||
This may be helpful for users who want to take advantage of accuracy
|
||||
improvements in Tesseract for files they previously OCRed with an
|
||||
earlier version of Tesseract and OCRmyPDF.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --redo-ocr input.pdf output.pdf
|
||||
|
||||
This method will replace OCR without rasterizing, reducing quality or
|
||||
removing vector content. If a file contains a mix of pure digital text
|
||||
and OCR, digital text will be ignored and OCR will be replaced. As such
|
||||
this mode is incompatible with image processing options, since they
|
||||
alter the appearance of the file.
|
||||
|
||||
In some cases, existing OCR cannot be detected or replaced. Files
|
||||
produced by OCRmyPDF v2.2 or earlier, for example, are internally
|
||||
represented as having visible text with an opaque image drawn on top.
|
||||
This situation cannot be detected.
|
||||
|
||||
If ``--redo-ocr`` does not work, you can use ``--force-ocr``, which will
|
||||
force rasterization of all pages, potentially reducing quality or losing
|
||||
vector content.
|
||||
|
||||
Improving OCR quality
|
||||
=====================
|
||||
|
||||
The `Image processing <#image-processing>`__ features can improve OCR
|
||||
quality.
|
||||
|
||||
Rotating pages and deskewing helps to ensure that the page orientation
|
||||
is correct before OCR begins. Removing the background and/or cleaning
|
||||
the page can also improve results. The ``--oversample DPI`` argument can
|
||||
be specified to resample images to higher resolution before attempting
|
||||
OCR; this can improve results as well.
|
||||
|
||||
OCR quality will suffer if the resolution of input images is not correct
|
||||
(since the range of pixel sizes that will be checked for possible fonts
|
||||
will also be incorrect).
|
||||
|
||||
PDF optimization
|
||||
================
|
||||
|
||||
By default OCRmyPDF will attempt to perform lossless optimizations on
|
||||
the images inside PDFs after OCR is complete. Optimization is performed
|
||||
even if no OCR text is found.
|
||||
|
||||
The ``--optimize N`` (short form ``-O``) argument controls optimization,
|
||||
where ``N`` ranges from 0 to 3 inclusive, analogous to the optimization
|
||||
levels in the GCC compiler.
|
||||
|
||||
.. list-table::
|
||||
:widths: auto
|
||||
:header-rows: 1
|
||||
|
||||
* - Level
|
||||
- Comments
|
||||
* - ``--optimize 0``
|
||||
- Disables optimization.
|
||||
* - ``--optimize 1``
|
||||
- Enables lossless optimizations, such as transcoding images to more
|
||||
efficient formats. Also compress other uncompressed objects in the
|
||||
PDF and enables the more efficient "object streams" within the PDF.
|
||||
(If ``--jbig2-lossy`` is issued, then lossy JBIG2 optimization is used.
|
||||
The decision to use lossy JBIG2 is separate from standard optimization
|
||||
settings.)
|
||||
* - ``--optimize 2``
|
||||
- All of the above, and enables lossy optimizations and color quantization.
|
||||
* - ``--optimize 3``
|
||||
- All of the above, and enables more aggressive optimizations and targets lower image quality.
|
||||
|
||||
Optimization is improved when a JBIG2 encoder is available and when
|
||||
``pngquant`` is installed. If either of these components are missing,
|
||||
then some types of images cannot be optimized.
|
||||
|
||||
The types of optimization available may expand over time. By default,
|
||||
OCRmyPDF compresses data streams inside PDFs, and will change
|
||||
inefficient compression modes to more modern versions. A program like
|
||||
``qpdf`` can be used to change encodings, e.g. to inspect the internals
|
||||
fo a PDF.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --optimize 3 in.pdf out.pdf # Make it small
|
||||
|
||||
Some users may consider enabling lossy JBIG2. See: :ref:`jbig2-lossy`.
|
||||
|
||||
.. note::
|
||||
|
||||
Image processing and PDF/A conversion can also introduce lossy transformations
|
||||
to your PDF images, even when ``--optimize 1`` is in use.
|
||||
@@ -0,0 +1,30 @@
|
||||
% SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Design notes
|
||||
|
||||
## Why doesn\'t OCRmyPDF use PyTesseract?
|
||||
|
||||
PyTesseract is a Python wrapper around the Tesseract OCR engine. When
|
||||
OCRmyPDF was first written, PyTesseract used ABI bindings to call the
|
||||
Tesseract library. This was not a good fit for OCRmyPDF because ABI
|
||||
bindings can be fragile.
|
||||
|
||||
PyTesseract has since evolved calling the Tesseract executable,
|
||||
abandoning the ABI approach and using the CLI instead, just like
|
||||
OCRmyPDF does. If it were written from scratch today, OCRmyPDF might use
|
||||
PyTesseract.
|
||||
|
||||
PyTesseract has more features don\'t particularly need PDF output, but
|
||||
less features than OCRmyPDF\'s API for creating PDFs.
|
||||
|
||||
## What is `executor()`?
|
||||
|
||||
OCRmyPDF uses a custom concurrent executor which can support either
|
||||
threads or processes with the same interface. This is useful because
|
||||
OCRmyPDF can use either threads or processes to parallelize work,
|
||||
whichever is more appropriate for the task at hand.
|
||||
|
||||
The interface is currently private and subject to change. In particular,
|
||||
if experiments with asyncio and anyio are successful, the interface will
|
||||
change.
|
||||
+251
@@ -0,0 +1,251 @@
|
||||
# OCRmyPDF Docker image {#docker}
|
||||
|
||||
OCRmyPDF is also available in Docker images that packages recent
|
||||
versions of all dependencies.
|
||||
|
||||
For users who already have Docker installed this may be an easy and
|
||||
convenient option.
|
||||
|
||||
On platforms other than Linux, Docker runs in a virtual machine, and so
|
||||
may be less performant. You may also want to adjust the Docker virtual
|
||||
machine\'s memory and CPU allocation. On Linux, the Docker image runs
|
||||
natively and performance is comparable to a system installation.
|
||||
|
||||
{#docker-install}
|
||||
## Installing the Docker image
|
||||
|
||||
If you have [Docker](https://docs.docker.com/) installed on your system,
|
||||
you can install a Docker image of the latest release.
|
||||
|
||||
If you can run this command successfully, your system is ready to
|
||||
download and execute the image:
|
||||
|
||||
:::{code} bash
|
||||
docker run hello-world
|
||||
:::
|
||||
|
||||
:::{list-table} Docker Images
|
||||
:header-rows: 1
|
||||
|
||||
* - Image
|
||||
- Architecture
|
||||
- Description
|
||||
* - `jbarlow83/ocrmypdf-alpine`
|
||||
- x86_64 and arm64
|
||||
- Recommended image, based on Alpine Linux.
|
||||
* - `jbarlow83/ocrmypdf-ubuntu`
|
||||
- x86_64 and arm64
|
||||
- Alternate image, based on Ubuntu. When the Alpine image is considered stable and available for arm64, this image will be deprecated.
|
||||
* - `jbarlow83/ocrmypdf`
|
||||
- x86_64 and arm64
|
||||
- Currently an alias for ocrmypdf-ubuntu. When the Alpine image is considered stable and available for arm64, this name will point to the Alpine image. If you don\'t know about the difference between Alpine and Ubuntu, use this image.
|
||||
:::
|
||||
|
||||
To install:
|
||||
|
||||
:::{code} bash
|
||||
docker pull jbarlow83/ocrmypdf-alpine
|
||||
:::
|
||||
|
||||
The `ocrmypdf` image is also available, but is deprecated and will be
|
||||
removed in the future.
|
||||
|
||||
OCRmyPDF will use all available CPU cores. See the Docker documentation
|
||||
for [adjusting memory and CPU on other
|
||||
platforms](https://docs.docker.com/config/containers/resource_constraints/)
|
||||
if you are using Docker on macOS or Windows, where you may need to
|
||||
manually assign more resources. On Linux, all resources will be
|
||||
available automatically.
|
||||
|
||||
The underlying operating system and other details in Docker images are
|
||||
considered implementation details and **subject to change at minor
|
||||
releases**. If you are modifying the image, you should pin the version
|
||||
you intend to use.
|
||||
|
||||
## Using the Docker image on the command line
|
||||
|
||||
**Unlike typical Docker containers**, in this section the OCRmyPDF
|
||||
Docker container is ephemeral -- it runs for one OCR job and terminates,
|
||||
just like a command line program. We are using Docker to deliver an
|
||||
application (as opposed to the more conventional case, where a Docker
|
||||
container runs as a server). For that reason we usually use the `--rm`
|
||||
argument to delete the container when it exits.
|
||||
|
||||
To start a Docker container (instance of the image):
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm -i jbarlow83/ocrmypdf-alpine (... all other arguments here...) - -
|
||||
:::
|
||||
|
||||
For convenience, create a shell alias to hide the Docker command. It is
|
||||
easier to send the input file as stdin and read the output from stdout
|
||||
-- **this avoids the messy permission issues with Docker entirely**.
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf --version # runs docker version
|
||||
docker_ocrmypdf - - <input.pdf >output.pdf
|
||||
:::
|
||||
|
||||
Or in the wonderful [fish shell](https://fishshell.com/):
|
||||
|
||||
:::{code} fish
|
||||
alias docker_ocrmypdf 'docker run --rm jbarlow83/ocrmypdf-alpine'
|
||||
funcsave docker_ocrmypdf
|
||||
:::
|
||||
|
||||
Alternately, you could mount the local current working directory as a
|
||||
Docker volume:
|
||||
|
||||
:::{code} bash
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" --workdir /data -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
docker_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
:::
|
||||
|
||||
## Podman
|
||||
|
||||
Especially if you use [Podman](https://podman.io/) (or use Docker in
|
||||
rootless mode), you may need to add `--userns keep-id` there,
|
||||
otherwise you may get access errors, because the user ID is otherwise not
|
||||
mapped to the same UID as on the host:
|
||||
|
||||
:::{code} bash
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id --workdir /data -v "$PWD:/data" jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
:::
|
||||
|
||||
If you have SELinux enabled, you may additionally need to add the `:Z` [suffix to
|
||||
the
|
||||
volume](https://docs.podman.io/en/stable/markdown/podman-run.1.html#volume-v-source-volume-host-dir-container-dir-options)
|
||||
or disable SELinux for the container using
|
||||
`--security-opt label=disable`, which is suggested for system files as
|
||||
they should not be re-labelled. Please refer to the „Note" section at
|
||||
the end of the linked podman documentation for details. This results in
|
||||
the following full command:
|
||||
|
||||
:::{code} bash
|
||||
alias podman_ocrmypdf='podman run --rm -i --user "$(id -u):$(id -g)" --userns keep-id --workdir /data -v "$PWD:/data" --security-opt label=disable jbarlow83/ocrmypdf-alpine'
|
||||
podman_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
:::
|
||||
|
||||
{#docker-lang-packs}
|
||||
## Adding languages to the Docker image
|
||||
|
||||
By default the Docker image includes English, German, Simplified
|
||||
Chinese, French, Portuguese and Spanish, the most popular languages for
|
||||
OCRmyPDF users based on feedback. You may add other languages by
|
||||
creating a new Dockerfile based on the public one.
|
||||
|
||||
:::{code} dockerfile
|
||||
FROM jbarlow83/ocrmypdf
|
||||
|
||||
# Example: add Italian
|
||||
RUN apt install tesseract-ocr-ita
|
||||
:::
|
||||
|
||||
To install language packs (training data) such as the
|
||||
[tessdata\_best](https://github.com/tesseract-ocr/tessdata_best) suite
|
||||
or custom data, you first need to determine the version of Tesseract
|
||||
data files, which may differ from the Tesseract program version. Use
|
||||
this command to determine the data file version:
|
||||
|
||||
:::{code} bash
|
||||
docker run -i --rm --entrypoint /bin/ls jbarlow83/ocrmypdf /usr/share/tesseract-ocr
|
||||
:::
|
||||
|
||||
As of 2021, the data file version is probably `4.00`.
|
||||
|
||||
You can then add new data with either a Dockerfile:
|
||||
|
||||
:::{code} dockerfile
|
||||
FROM jbarlow83/ocrmypdf:{TAG}
|
||||
|
||||
# Example: add a tessdata_best file
|
||||
COPY chi_tra_vert.traineddata /usr/share/tesseract-ocr/<data version>/tessdata/
|
||||
:::
|
||||
|
||||
When creating your own image, you should always pin a specific version
|
||||
of the OCRmyPDF Docker image. This ensures that your image will not
|
||||
break when a new version of OCRmyPDF is released.
|
||||
|
||||
Alternately, you can copy training data into a Docker container as
|
||||
follows:
|
||||
|
||||
:::{code} bash
|
||||
docker cp mycustomtraining.traineddata name_of_container:/usr/share/tesseract-ocr/<tesseract version>/tessdata/
|
||||
:::
|
||||
|
||||
Extending the Docker image
|
||||
--------------------------
|
||||
|
||||
You can extend the Docker image with your own customizations, similar to
|
||||
the way it is extended to add language packs.
|
||||
|
||||
Note that the Docker image is subject to change at any time. For
|
||||
example, the base image may be updated to a newer version of Ubuntu or
|
||||
Debian. Such changes will be noted in the release notes but might occur
|
||||
at minor versions releases, unless the way a \"casual\" user of the
|
||||
Docker image is affected.
|
||||
|
||||
If you extend the Docker image, you should pin a specific version of the
|
||||
OCRmyPDF Docker image.
|
||||
|
||||
Executing the test suite
|
||||
------------------------
|
||||
|
||||
The OCRmyPDF test suite is installed with image. To run it:
|
||||
|
||||
:::{code} bash
|
||||
docker run --rm --entrypoint python jbarlow83/ocrmypdf -m pytest
|
||||
:::
|
||||
|
||||
Accessing the shell
|
||||
-------------------
|
||||
|
||||
To use the shell in the Docker image:
|
||||
|
||||
:::{code} bash
|
||||
docker run -it --entrypoint sh jbarlow83/ocrmypdf
|
||||
:::
|
||||
|
||||
Using the OCRmyPDF web service wrapper
|
||||
--------------------------------------
|
||||
|
||||
The OCRmyPDF Docker image includes an example, barebones HTTP web
|
||||
service. The webservice may be launched as follows:
|
||||
|
||||
:::{code} bash
|
||||
docker run --entrypoint python -p 5000:5000 jbarlow83/ocrmypdf webservice.py
|
||||
:::
|
||||
|
||||
We omit the `--rm` parameter so that the container will not be
|
||||
automatically deleted when it exits.
|
||||
|
||||
This will configure the machine to listen on port 5000. On Linux
|
||||
machines this is port 5000 of localhost. On macOS or Windows machines
|
||||
running Docker, this is port 5000 of the virtual machine that runs your
|
||||
Docker images. You can find its IP address using the command
|
||||
`docker-machine ip`.
|
||||
|
||||
Unlike command line usage this program will open a socket and wait for
|
||||
connections.
|
||||
|
||||
:::{warning}
|
||||
The OCRmyPDF web service wrapper is intended for demonstration or
|
||||
development. It provides no security, no authentication, no protection
|
||||
against denial of service attacks, and no load balancing. The default
|
||||
Flask WSGI server is used, which is intended for development only. The
|
||||
server is single-threaded and so can respond to only one client at a
|
||||
time. While running OCR, it cannot respond to any other clients.
|
||||
:::
|
||||
|
||||
Clients must keep their open connection while waiting for OCR to
|
||||
complete. This may entail setting a long timeout; this interface is more
|
||||
useful for internal HTTP API calls.
|
||||
|
||||
Unlike the rest of OCRmyPDF, this web service is licensed under the
|
||||
Affero GPLv3 (AGPLv3) since Ghostscript is also licensed in this way.
|
||||
|
||||
In addition to the above, please read our
|
||||
`general remarks on using OCRmyPDF as a service <ocr-service>`{.interpreted-text
|
||||
role="ref"}.
|
||||
-200
@@ -1,200 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
.. _docker:
|
||||
|
||||
=====================
|
||||
OCRmyPDF Docker image
|
||||
=====================
|
||||
|
||||
OCRmyPDF is also available in a Docker image that packages recent
|
||||
versions of all dependencies.
|
||||
|
||||
For users who already have Docker installed this may be an easy and
|
||||
convenient option. However, it is less performant than a system
|
||||
installation and may require Docker engine configuration.
|
||||
|
||||
OCRmyPDF needs a generous amount of RAM, CPU cores, temporary storage
|
||||
space, whether running in a Docker container or on its own. It may be
|
||||
necessary to ensure the container is provisioned with additional
|
||||
resources.
|
||||
|
||||
.. _docker-install:
|
||||
|
||||
Installing the Docker image
|
||||
===========================
|
||||
|
||||
If you have `Docker <https://docs.docker.com/>`__ installed on your
|
||||
system, you can install a Docker image of the latest release.
|
||||
|
||||
If you can run this command successfully, your system is ready to download and
|
||||
execute the image:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run hello-world
|
||||
|
||||
The recommended OCRmyPDF Docker image is currently named ``ocrmypdf``:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker pull jbarlow83/ocrmypdf
|
||||
|
||||
|
||||
OCRmyPDF will use all available CPU cores. By default, the VirtualBox
|
||||
machine instance on Windows and macOS has only a single CPU core
|
||||
enabled. Use the VirtualBox Manager to determine the name of your Docker
|
||||
engine host, and then follow these optional steps to enable multiple
|
||||
CPUs:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# Optional step for Mac OS X users
|
||||
docker-machine stop "yourVM"
|
||||
VBoxManage modifyvm "yourVM" --cpus 2 # or whatever number of core is desired
|
||||
docker-machine start "yourVM"
|
||||
eval $(docker-machine env "yourVM")
|
||||
|
||||
See the Docker documentation for
|
||||
`adjusting memory and CPU on other platforms <https://docs.docker.com/config/containers/resource_constraints/>`__.
|
||||
|
||||
Using the Docker image on the command line
|
||||
==========================================
|
||||
|
||||
**Unlike typical Docker containers**, in this section the OCRmyPDF Docker
|
||||
container is ephemeral – it runs for one OCR job and terminates, just like a
|
||||
command line program. We are using Docker to deliver an application (as opposed
|
||||
to the more conventional case, where a Docker container runs as a server).
|
||||
|
||||
To start a Docker container (instance of the image):
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker tag jbarlow83/ocrmypdf ocrmypdf
|
||||
docker run --rm -i ocrmypdf (... all other arguments here...) - -
|
||||
|
||||
For convenience, create a shell alias to hide the Docker command. It is
|
||||
easier to send the input file as stdin and read the output from
|
||||
stdout – **this avoids the messy permission issues with Docker entirely**.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
alias docker_ocrmypdf='docker run --rm -i ocrmypdf'
|
||||
docker_ocrmypdf --version # runs docker version
|
||||
docker_ocrmypdf - - <input.pdf >output.pdf
|
||||
|
||||
Or in the wonderful `fish shell <https://fishshell.com/>`__:
|
||||
|
||||
.. code-block:: fish
|
||||
|
||||
alias docker_ocrmypdf 'docker run --rm ocrmypdf'
|
||||
funcsave docker_ocrmypdf
|
||||
|
||||
Alternately, you could mount the local current working directory as a
|
||||
Docker volume:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" --workdir /data -v "$PWD:/data" ocrmypdf'
|
||||
docker_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
|
||||
.. _docker-lang-packs:
|
||||
|
||||
Adding languages to the Docker image
|
||||
====================================
|
||||
|
||||
By default the Docker image includes English, German, Simplified Chinese,
|
||||
French, Portuguese and Spanish, the most popular languages for OCRmyPDF
|
||||
users based on feedback. You may add other languages by creating a new
|
||||
Dockerfile based on the public one.
|
||||
|
||||
.. code-block:: dockerfile
|
||||
|
||||
FROM jbarlow83/ocrmypdf
|
||||
|
||||
# Example: add Italian
|
||||
RUN apt install tesseract-ocr-ita
|
||||
|
||||
To install language packs (training data) such as the
|
||||
`tessdata_best <https://github.com/tesseract-ocr/tessdata_best>`_ suite or
|
||||
custom data, you first need to determine the version of Tesseract data files, which
|
||||
may differ from the Tesseract program version. Use this command to determine the data
|
||||
file version:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run -i --rm --entrypoint /bin/ls jbarlow83/ocrmypdf /usr/share/tesseract-ocr
|
||||
|
||||
As of 2021, the data file version is probably ``4.00``.
|
||||
|
||||
You can then add new data with either a Dockerfile:
|
||||
|
||||
.. code-block:: dockerfile
|
||||
|
||||
FROM jbarlow83/ocrmypdf
|
||||
|
||||
# Example: add a tessdata_best file
|
||||
COPY chi_tra_vert.traineddata /usr/share/tesseract-ocr/<data version>/tessdata/
|
||||
|
||||
Alternately, you can copy training data into a Docker container as follows:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker cp mycustomtraining.traineddata name_of_container:/usr/share/tesseract-ocr/<tesseract version>/tessdata/
|
||||
|
||||
Executing the test suite
|
||||
========================
|
||||
|
||||
The OCRmyPDF test suite is installed with image. To run it:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run --entrypoint python3 jbarlow83/ocrmypdf -m pytest
|
||||
|
||||
Accessing the shell
|
||||
===================
|
||||
|
||||
To use the bash shell in the Docker image:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run -it --entrypoint bash jbarlow83/ocrmypdf
|
||||
|
||||
Using the OCRmyPDF web service wrapper
|
||||
======================================
|
||||
|
||||
The OCRmyPDF Docker image includes an example, barebones HTTP web
|
||||
service. The webservice may be launched as follows:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run --entrypoint python3 -p 5000:5000 jbarlow83/ocrmypdf webservice.py
|
||||
|
||||
This will configure the machine to listen on port 5000. On Linux machines
|
||||
this is port 5000 of localhost. On macOS or Windows machines running
|
||||
Docker, this is port 5000 of the virtual machine that runs your Docker
|
||||
images. You can find its IP address using the command ``docker-machine ip``.
|
||||
|
||||
Unlike command line usage this program will open a socket and wait for
|
||||
connections.
|
||||
|
||||
.. warning::
|
||||
|
||||
The OCRmyPDF web service wrapper is intended for demonstration or
|
||||
development. It provides no security, no authentication, no
|
||||
protection against denial of service attacks, and no load balancing.
|
||||
The default Flask WSGI server is used, which is intended for
|
||||
development only. The server is single-threaded and so can respond to
|
||||
only one client at a time. While running OCR, it cannot respond to
|
||||
any other clients.
|
||||
|
||||
Clients must keep their open connection while waiting for OCR to
|
||||
complete. This may entail setting a long timeout; this interface is more
|
||||
useful for internal HTTP API calls.
|
||||
|
||||
Unlike the rest of OCRmyPDF, this web service is licensed under the
|
||||
Affero GPLv3 (AGPLv3) since Ghostscript is also licensed in this way.
|
||||
|
||||
In addition to the above, please read our
|
||||
:ref:`general remarks on using OCRmyPDF as a service <ocr-service>`.
|
||||
@@ -0,0 +1,51 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Common error messages
|
||||
|
||||
## Page already has text
|
||||
|
||||
:::{code}
|
||||
ERROR - 1: page already has text! – aborting (use --force-ocr to force OCR)
|
||||
:::
|
||||
|
||||
You ran ocrmypdf on a file that already contains printable text or a
|
||||
hidden OCR text layer (it can\'t quite tell the difference). You
|
||||
probably don\'t want to do this, because the file is already searchable.
|
||||
|
||||
As the error message suggests, your options are:
|
||||
|
||||
- `ocrmypdf --force-ocr` to
|
||||
`rasterize <raster-vector>`{.interpreted-text role="ref"} all vector
|
||||
content and run OCR on the images. This is useful if a previous OCR
|
||||
program failed, or if the document contains a text watermark.
|
||||
- `ocrmypdf --skip-text` to skip OCR and other processing on any pages
|
||||
that contain text. Text pages will be copied into the output PDF
|
||||
without modification.
|
||||
- `ocrmypdf --redo-ocr` to scan the file for any existing OCR
|
||||
(non-printing text), remove it, and do OCR again. This is one way to
|
||||
take advantage of improvements in OCR accuracy. Printable vector
|
||||
text is excluded from OCR, so this can be used on files that contain
|
||||
a mix of digital and scanned files.
|
||||
|
||||
## Input file \'filename\' is not a valid PDF
|
||||
|
||||
OCRmyPDF checks files with pikepdf, a library that in turn uses libqpdf
|
||||
to fixes errors in PDFs, before it tries to work on them. In most cases
|
||||
this happens because the PDF is corrupt and truncated (incomplete file
|
||||
copying) and not much can be done.
|
||||
|
||||
You can try rewriting the file with Ghostscript:
|
||||
|
||||
:::{code} bash
|
||||
gs -o output.pdf -dSAFER -sDEVICE=pdfwrite input.pdf
|
||||
:::
|
||||
|
||||
`pdftk` can also rewrite PDFs:
|
||||
|
||||
:::{code} bash
|
||||
pdftk input.pdf cat output output.pdf
|
||||
:::
|
||||
|
||||
Sometimes Acrobat can repair PDFs with its [Preflight
|
||||
tool](https://helpx.adobe.com/acrobat/using/correcting-problem-areas-preflight-tool.html).
|
||||
@@ -1,57 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
=====================
|
||||
Common error messages
|
||||
=====================
|
||||
|
||||
Page already has text
|
||||
=====================
|
||||
|
||||
.. code-block::
|
||||
|
||||
ERROR - 1: page already has text! – aborting (use --force-ocr to force OCR)
|
||||
|
||||
You ran ocrmypdf on a file that already contains printable text or a
|
||||
hidden OCR text layer (it can't quite tell the difference). You probably
|
||||
don't want to do this, because the file is already searchable.
|
||||
|
||||
As the error message suggests, your options are:
|
||||
|
||||
- ``ocrmypdf --force-ocr`` to :ref:`rasterize <raster-vector>` all
|
||||
vector content and run OCR on the images. This is useful if a
|
||||
previous OCR program failed, or if the document contains a text
|
||||
watermark.
|
||||
- ``ocrmypdf --skip-text`` to skip OCR and other processing on any
|
||||
pages that contain text. Text pages will be copied into the output
|
||||
PDF without modification.
|
||||
- ``ocrmypdf --redo-ocr`` to scan the file for any existing OCR
|
||||
(non-printing text), remove it, and do OCR again. This is one way
|
||||
to take advantage of improvements in OCR accuracy. Printable vector
|
||||
text is excluded from OCR, so this can be used on files that contain
|
||||
a mix of digital and scanned files.
|
||||
|
||||
|
||||
Input file 'filename' is not a valid PDF
|
||||
========================================
|
||||
|
||||
OCRmyPDF checks files with pikepdf, a library that in turn uses libqpdf to fixes
|
||||
errors in PDFs, before it tries to work on them. In most cases this happens
|
||||
because the PDF is corrupt and truncated (incomplete file copying) and not much
|
||||
can be done.
|
||||
|
||||
You can try rewriting the file with Ghostscript:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
gs -o output.pdf -dSAFER -sDEVICE=pdfwrite input.pdf
|
||||
|
||||
``pdftk`` can also rewrite PDFs:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pdftk input.pdf cat output output.pdf
|
||||
|
||||
Sometimes Acrobat can repair PDFs with its `Preflight
|
||||
tool <https://helpx.adobe.com/acrobat/using/correcting-problem-areas-preflight-tool.html>`__.
|
||||
@@ -0,0 +1,57 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# OCRmyPDF documentation
|
||||
|
||||
:::{figure} images/logo.svg
|
||||
:::
|
||||
|
||||
OCRmyPDF adds an optical character recognition (OCR) text layer to scanned PDF
|
||||
files, allowing them to be searched.
|
||||
|
||||
PDF is the best format for storing and exchanging scanned documents.
|
||||
Unfortunately, PDFs can be difficult to modify. OCRmyPDF makes it easy to apply
|
||||
image processing and OCR (recognized, searchable text) to existing PDFs.
|
||||
|
||||
```{toctree}
|
||||
:maxdepth: 1
|
||||
|
||||
introduction
|
||||
release_notes
|
||||
installation
|
||||
languages
|
||||
jbig2
|
||||
```
|
||||
|
||||
```{toctree}
|
||||
:caption: Usage
|
||||
:maxdepth: 2
|
||||
|
||||
cookbook
|
||||
optimizer
|
||||
docker
|
||||
advanced
|
||||
batch
|
||||
cloud
|
||||
performance
|
||||
pdfsecurity
|
||||
errors
|
||||
```
|
||||
|
||||
```{toctree}
|
||||
:caption: Developers
|
||||
:maxdepth: 2
|
||||
|
||||
api
|
||||
plugins
|
||||
apiref
|
||||
design_notes
|
||||
contributing
|
||||
maintainers
|
||||
```
|
||||
|
||||
# Indices and tables
|
||||
|
||||
- {ref}`genindex`
|
||||
- {ref}`modindex`
|
||||
- {ref}`search`
|
||||
@@ -1,54 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
OCRmyPDF documentation
|
||||
======================
|
||||
|
||||
.. figure:: images/logo.svg
|
||||
|
||||
OCRmyPDF adds an optical character recognition (OCR) text layer to scanned PDF
|
||||
files, allowing them to be searched.
|
||||
|
||||
PDF is the best format for storing and exchanging scanned documents.
|
||||
Unfortunately, PDFs can be difficult to modify. OCRmyPDF makes it easy to apply
|
||||
image processing and OCR to existing PDFs.
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 1
|
||||
|
||||
introduction
|
||||
release_notes
|
||||
installation
|
||||
optimizer
|
||||
languages
|
||||
jbig2
|
||||
|
||||
.. toctree::
|
||||
:caption: Usage
|
||||
:maxdepth: 2
|
||||
|
||||
cookbook
|
||||
docker
|
||||
advanced
|
||||
batch
|
||||
performance
|
||||
pdfsecurity
|
||||
errors
|
||||
|
||||
.. toctree::
|
||||
:caption: Developers
|
||||
:maxdepth: 2
|
||||
|
||||
api
|
||||
plugins
|
||||
apiref
|
||||
contributing
|
||||
maintainers
|
||||
|
||||
Indices and tables
|
||||
==================
|
||||
|
||||
* :ref:`genindex`
|
||||
* :ref:`modindex`
|
||||
* :ref:`search`
|
||||
@@ -0,0 +1,813 @@
|
||||
---
|
||||
myst:
|
||||
substitutions:
|
||||
deb_11: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/debian_11/ocrmypdf.svg
|
||||
:alt: Debian 11
|
||||
:::
|
||||
deb_12: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/debian_12/ocrmypdf.svg
|
||||
:alt: Debian 12
|
||||
:::
|
||||
deb_unstable: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/debian_unstable/ocrmypdf.svg
|
||||
:alt: Debian unstable
|
||||
:::
|
||||
fedora_38: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/fedora_38/ocrmypdf.svg
|
||||
:alt: Fedora 38
|
||||
:::
|
||||
fedora_39: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/fedora_39/ocrmypdf.svg
|
||||
:alt: Fedora 39
|
||||
:::
|
||||
fedora_rawhide: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
||||
:alt: Fedore Rawhide
|
||||
:::
|
||||
latest: |-
|
||||
:::{image} https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||
:alt: OCRmyPDF latest released version on PyPI
|
||||
:::
|
||||
ubu_2004: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
||||
:alt: Ubuntu 20.04 LTS
|
||||
:::
|
||||
ubu_2204: |-
|
||||
:::{image} https://repology.org/badge/version-for-repo/ubuntu_22_04/ocrmypdf.svg
|
||||
:alt: Ubuntu 22.04 LTS
|
||||
:::
|
||||
---
|
||||
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Installing OCRmyPDF
|
||||
|
||||
(latest)=
|
||||
|
||||
The easiest way to install OCRmyPDF is to follow the steps for your operating
|
||||
system/platform. This version may be out of date, however.
|
||||
|
||||
These platforms have one-liner installs:
|
||||
|
||||
:::{list-table}
|
||||
:header-rows: 0
|
||||
|
||||
* - Debian, Ubuntu
|
||||
- ``apt install ocrmypdf``
|
||||
* - Windows Subsystem for Linux
|
||||
- ``apt install ocrmypdf``
|
||||
* - Fedora
|
||||
- ``dnf install ocrmypdf tesseract-osd``
|
||||
* - macOS (Homebrew)
|
||||
- ``brew install ocrmypdf``
|
||||
* - macOS (MacPorts)
|
||||
- ``port install ocrmypdf``
|
||||
* - LinuxBrew
|
||||
- ``brew install ocrmypdf``
|
||||
* - FreeBSD
|
||||
- ``pkg install textproc/py-ocrmypdf``
|
||||
* - Snap (snapcraft packaging)
|
||||
- ``snap install ocrmypdf``
|
||||
:::
|
||||
|
||||
More detailed procedures are outlined below. If you want to do a manual
|
||||
install, or install a more recent version than your platform provides, read on.
|
||||
|
||||
:::{contents} Platform-specific steps
|
||||
:depth: 2
|
||||
:local: true
|
||||
:::
|
||||
|
||||
## Installing on Linux
|
||||
|
||||
### Debian and Ubuntu 20.04 or newer
|
||||
|
||||
:::{list-table}
|
||||
:header-rows: 1
|
||||
|
||||
* - OCRmyPDF versions in Debian & Ubuntu
|
||||
* - {{ latest }}
|
||||
* - {{ deb_11 }} {{ deb_12 }} {{ deb_unstable }}
|
||||
* - {{ ubu_2004 }} {{ ubu_2204 }}
|
||||
:::
|
||||
|
||||
Users of Debian or Ubuntu may simply
|
||||
|
||||
```bash
|
||||
apt install ocrmypdf
|
||||
```
|
||||
|
||||
As indicated in the table above, Debian and Ubuntu releases may lag
|
||||
behind the latest version. If the version available for your platform is
|
||||
out of date, you could opt to install the latest version from source.
|
||||
See [Installing HEAD revision from
|
||||
sources](#installing-head-revision-from-sources).
|
||||
|
||||
For full details on version availability for your platform, check the
|
||||
[Debian Package Tracker](https://tracker.debian.org/pkg/ocrmypdf) or
|
||||
[Ubuntu launchpad.net](https://launchpad.net/ocrmypdf).
|
||||
|
||||
:::{note}
|
||||
OCRmyPDF for Debian and Ubuntu currently omit the JBIG2 encoder.
|
||||
OCRmyPDF works fine without it but will produce larger output files.
|
||||
If you build jbig2enc from source, ocrmypdf will
|
||||
automatically detect it (specifically the `jbig2` binary) on the
|
||||
`PATH`. To add JBIG2 encoding, see {ref}`jbig2`.
|
||||
:::
|
||||
|
||||
### Fedora
|
||||
|
||||
:::{list-table}
|
||||
:header-rows: 1
|
||||
|
||||
* - OCRmyPDF version
|
||||
* - {{latest}}
|
||||
* - {{fedora_38}} {{fedora_39}} {{fedora_rawhide}}
|
||||
:::
|
||||
|
||||
Users of Fedora may simply
|
||||
|
||||
```bash
|
||||
dnf install ocrmypdf tesseract-osd
|
||||
```
|
||||
|
||||
For full details on version availability, check the [Fedora Package
|
||||
Tracker](https://packages.fedoraproject.org/pkgs/ocrmypdf/ocrmypdf/).
|
||||
|
||||
If the version available for your platform is out of date, you could opt
|
||||
to install the latest version from source. See [Installing HEAD revision
|
||||
from sources](#installing-head-revision-from-sources).
|
||||
|
||||
:::{note}
|
||||
OCRmyPDF for Fedora currently omits the JBIG2 encoder due to patent
|
||||
issues. OCRmyPDF works fine without it but will produce larger output
|
||||
files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later
|
||||
will automatically detect it on the `PATH`. To add JBIG2 encoding,
|
||||
see {ref}`Installing the JBIG2 encoder <jbig2>`.
|
||||
:::
|
||||
|
||||
(ubuntu-lts-latest)=
|
||||
|
||||
### RHEL 9
|
||||
|
||||
Prepare the environment by getting Python 3.11:
|
||||
|
||||
```bash
|
||||
dnf install python3.11 python3.11-pip
|
||||
```
|
||||
|
||||
Then, follow [Requirements for pip and HEAD install](#requirements-for-pip-and-head-install) to install dependencies:
|
||||
|
||||
```bash
|
||||
dnf install ghostscript tesseract
|
||||
```
|
||||
|
||||
and build ocrmypdf in virtual environment:
|
||||
|
||||
```bash
|
||||
python3.11 -m venv .venv
|
||||
```
|
||||
|
||||
To add JBIG2 encoding, see {ref}`Installing the JBIG2 encoder <jbig2>`.
|
||||
|
||||
Note Fedora packages for language data haven't been branched for RHEL/EPEL, but you can get traineddata files directly from [tesseract](https://github.com/tesseract-ocr/tessdata/) and place them in `/usr/share/tesseract/tessdata`.
|
||||
|
||||
### Installing the latest version on Ubuntu 22.04 LTS
|
||||
|
||||
Ubuntu 22.04 includes ocrmypdf 13.4.0 - you can install that with
|
||||
`apt install ocrmypdf`. To install a more recent version for the current
|
||||
user, follow these steps:
|
||||
|
||||
```bash
|
||||
sudo apt-get update
|
||||
sudo apt-get -y install ocrmypdf python3-pip
|
||||
|
||||
pip install --user --upgrade ocrmypdf
|
||||
```
|
||||
|
||||
If you get the message `WARNING: The script ocrmypdf is installed in
|
||||
'/home/$USER/.local/bin' which is not on PATH.`, you may need to re-login
|
||||
or open a new shell, or manually adjust your PATH.
|
||||
|
||||
To add JBIG2 encoding, see {ref}`jbig2`.
|
||||
|
||||
### Ubuntu 20.04 LTS
|
||||
|
||||
Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with `apt`. The
|
||||
most convenient way to install recent OCRmyPDF on older Ubuntu is to use
|
||||
Homebrew on Linux (Linuxbrew).
|
||||
|
||||
```bash
|
||||
brew install ocrmypdf
|
||||
```
|
||||
|
||||
### Arch Linux (AUR)
|
||||
|
||||
:::{image} https://repology.org/badge/version-for-repo/aur/ocrmypdf.svg
|
||||
:alt: ArchLinux
|
||||
:target: https://repology.org/metapackage/ocrmypdf
|
||||
:::
|
||||
|
||||
There is an [Arch User Repository (AUR) package for OCRmyPDF](https://aur.archlinux.org/packages/ocrmypdf/).
|
||||
|
||||
Installing AUR packages as root is not allowed, so you must first [setup a
|
||||
non-root user](https://wiki.archlinux.org/index.php/Users_and_groups#User_management) and
|
||||
[configure sudo](https://wiki.archlinux.org/index.php/Sudo#Configuration).
|
||||
The standard Docker image, `archlinux/base:latest`, does **not** have a
|
||||
non-root user configured, so users of that image must follow these guides. If
|
||||
you are using a VM image, such as [the official Vagrant image](https://app.vagrantup.com/archlinux/boxes/archlinux), this work may already
|
||||
be completed for you.
|
||||
|
||||
Next you should install the [base-devel package group](https://archlinux.org/packages/core/any/base-devel/). This includes the
|
||||
standard tooling needed to build packages, such as a compiler and binary tools.
|
||||
|
||||
```bash
|
||||
sudo pacman -S --needed base-devel
|
||||
```
|
||||
|
||||
Now you are ready to install the OCRmyPDF package.
|
||||
|
||||
```bash
|
||||
curl -O https://aur.archlinux.org/cgit/aur.git/snapshot/ocrmypdf.tar.gz
|
||||
tar xvzf ocrmypdf.tar.gz
|
||||
cd ocrmypdf
|
||||
makepkg -sri
|
||||
```
|
||||
|
||||
At this point you will have a working install of OCRmyPDF, but the Tesseract
|
||||
install won’t include any OCR language data. You can install [the
|
||||
tesseract-data package group](https://www.archlinux.org/groups/any/tesseract-data/) to add all supported
|
||||
languages, or use that package listing to identify the appropriate package for
|
||||
your desired language.
|
||||
|
||||
```bash
|
||||
sudo pacman -S tesseract-data-eng
|
||||
```
|
||||
|
||||
As an alternative to this manual procedure, consider using an [AUR helper](https://wiki.archlinux.org/index.php/AUR_helpers). Such a tool will
|
||||
automatically fetch, build and install the AUR package, resolve dependencies
|
||||
(including dependencies on AUR packages), and ease the upgrade procedure.
|
||||
|
||||
If you have any difficulties with installation, check the repository package
|
||||
page.
|
||||
|
||||
:::{note}
|
||||
The OCRmyPDF AUR package currently omits the JBIG2 encoder. OCRmyPDF works
|
||||
fine without it but will produce larger output files. The encoder is
|
||||
available from [the jbig2enc-git AUR package](https://aur.archlinux.org/packages/jbig2enc-git/) and may be installed
|
||||
using the same series of steps as for the installation OCRmyPDF AUR
|
||||
package. Alternatively, it may be built manually from source following the
|
||||
instructions in {ref}`Installing the JBIG2 encoder <jbig2>`. If JBIG2 is
|
||||
installed, OCRmyPDF 7.0.0 and later will automatically detect it.
|
||||
:::
|
||||
|
||||
### Alpine Linux
|
||||
|
||||
:::{image} https://repology.org/badge/version-for-repo/alpine_edge/ocrmypdf.svg
|
||||
:alt: Alpine Linux
|
||||
:target: https://repology.org/metapackage/ocrmypdf
|
||||
:::
|
||||
|
||||
To install OCRmyPDF for Alpine Linux:
|
||||
|
||||
```bash
|
||||
apk add ocrmypdf
|
||||
```
|
||||
|
||||
### Gentoo Linux
|
||||
|
||||
:::{image} https://repology.org/badge/version-for-repo/gentoo_ovl_guru/ocrmypdf.svg
|
||||
:alt: Gentoo Linux
|
||||
:target: https://repology.org/metapackage/ocrmypdf
|
||||
:::
|
||||
|
||||
To install OCRmyPDF on Gentoo Linux, use the following commands:
|
||||
|
||||
```bash
|
||||
eselect repository enable guru
|
||||
emaint sync --repo guru
|
||||
emerge --ask app-text/OCRmyPDF
|
||||
```
|
||||
|
||||
### Other Linux packages
|
||||
|
||||
See the
|
||||
[Repology](https://repology.org/metapackage/ocrmypdf/versions) page.
|
||||
|
||||
In general, first install the OCRmyPDF package for your system, then
|
||||
optionally use the procedure [Installing with Python
|
||||
pip](#installing-with-python-pip) to install a more recent version.
|
||||
|
||||
## Installing on macOS
|
||||
|
||||
### Homebrew
|
||||
|
||||
:::{image} https://img.shields.io/homebrew/v/ocrmypdf.svg
|
||||
:alt: homebrew
|
||||
:target: https://formulae.brew.sh/formula/ocrmypdf
|
||||
:::
|
||||
|
||||
OCRmyPDF is now a standard [Homebrew](https://brew.sh) formula. To
|
||||
install on macOS:
|
||||
|
||||
```bash
|
||||
brew install ocrmypdf
|
||||
```
|
||||
|
||||
This will include only the English language pack. If you need other
|
||||
languages you can optionally install them all:
|
||||
|
||||
```bash
|
||||
brew install tesseract-lang # Optional: Install all language packs
|
||||
```
|
||||
|
||||
### MacPorts
|
||||
|
||||
:::{image} https://img.shields.io/badge/dynamic/json?url=https%3A%2F%2Fports.macports.org%2Fapi%2Fv1%2Fports%2Focrmypdf%2F%3Fformat%3Djson&query=version&label=MacPorts
|
||||
:alt: Macports Version Information
|
||||
:target: https://ports.macports.org/port/ocrmypdf
|
||||
:::
|
||||
|
||||
OCRmyPDF is includes in MacPorts:
|
||||
|
||||
```bash
|
||||
sudo port install ocrmypdf
|
||||
```
|
||||
|
||||
Note that while this will install tesseract you will need to install
|
||||
the appropriate tesseract [language ports](https://ports.macports.org/search/?selected_facets=categories_exact%3Atextproc&installed_file=&q=tesseract&name=on).
|
||||
|
||||
### Manual installation on macOS
|
||||
|
||||
These instructions probably work on all macOS supported by Homebrew, and are
|
||||
for installing a more current version of OCRmyPDF than is available from
|
||||
Homebrew. Note that the Homebrew versions usually track the release versions
|
||||
fairly closely.
|
||||
|
||||
If it's not already present, [install Homebrew](http://brew.sh/).
|
||||
|
||||
Update Homebrew:
|
||||
|
||||
```bash
|
||||
brew update
|
||||
```
|
||||
|
||||
Install or upgrade the required Homebrew packages, if any are missing.
|
||||
To do this, use `brew edit ocrmypdf` to obtain a recent list of Homebrew
|
||||
dependencies. You could also check the `.workflows/build.yml`.
|
||||
|
||||
This will include the English, French, German and Spanish language
|
||||
packs. If you need other languages you can optionally install them all:
|
||||
|
||||
(macos-all-languages)=
|
||||
|
||||
> ```bash
|
||||
> brew install tesseract-lang # Option 2: for all language packs
|
||||
> ```
|
||||
|
||||
Update the homebrew pip:
|
||||
|
||||
```bash
|
||||
pip install --upgrade pip
|
||||
```
|
||||
|
||||
You can then install OCRmyPDF from PyPI for the current user:
|
||||
|
||||
```bash
|
||||
pip install --user ocrmypdf
|
||||
```
|
||||
|
||||
The command line program should now be available:
|
||||
|
||||
```bash
|
||||
ocrmypdf --help
|
||||
```
|
||||
|
||||
## Installing on Windows
|
||||
|
||||
### Native Windows
|
||||
|
||||
% If you have a Windows that is not the Home edition, you can use Windows Sandbox to test on a blank Windows instance.
|
||||
% https://learn.microsoft.com/en-us/windows/security/application-security/application-isolation/windows-sandbox/
|
||||
|
||||
:::{note}
|
||||
Administrator privileges will be required for some of these steps.
|
||||
:::
|
||||
|
||||
You must install the following for Windows:
|
||||
|
||||
- Python 64-bit
|
||||
- Tesseract 64-bit
|
||||
- Ghostscript 64-bit
|
||||
|
||||
Using the [winget](https://docs.microsoft.com/en-us/windows/package-manager/winget/)
|
||||
package manager:
|
||||
|
||||
- `winget install -e --id Python.Python.3.11`
|
||||
- `winget install -e --id UB-Mannheim.TesseractOCR`
|
||||
|
||||
You will need to install Ghostscript manually, [since it does not support automated
|
||||
installs anymore](https://artifex.com/news/ghostscript-10.01.0-disabling-silent-install-option).
|
||||
|
||||
- [Ghostscript download page](https://ghostscript.com/releases/gsdnld.html).\`
|
||||
|
||||
(Or alternately, using the [Chocolatey](https://chocolatey.org/) package manager, install
|
||||
the following when running in an Administrator command prompt):
|
||||
|
||||
- `choco install python3`
|
||||
- `choco install --pre tesseract`
|
||||
- `choco install pngquant` (optional)
|
||||
|
||||
Either set of commands will install the required software. At the moment there is no
|
||||
single command to install Windows.
|
||||
|
||||
You may then use `pip` to install ocrmypdf. (This can performed by a user or
|
||||
Administrator.):
|
||||
|
||||
- `python3 -m pip install ocrmypdf`
|
||||
|
||||
% The Windows Python versions do not place any python or python3 executable in the path.
|
||||
% They add the py launcher to the path:
|
||||
% https://docs.python.org/3/using/windows.html#python-launcher-for-windows
|
||||
|
||||
If you installed Python using WinGet, then use the following command instead:
|
||||
|
||||
- `py -m pip install ocrmypdf`
|
||||
|
||||
and use:
|
||||
|
||||
- `py -m ocrmypdf`
|
||||
|
||||
To start OCRmyPDF.
|
||||
|
||||
If you intend to use more Python software on your Windows machine, consider the use of
|
||||
[pipx](https://pipx.pypa.io/stable/) or a similar tool to create isolated Python
|
||||
environments for each Python software that you want to use.
|
||||
|
||||
OCRmyPDF will check the Windows Registry and standard locations in your Program Files
|
||||
for third party software it needs (specifically, Tesseract and Ghostscript). To
|
||||
override the versions OCRmyPDF selects, you can modify the `PATH` environment
|
||||
variable. [Follow these directions](https://www.computerhope.com/issues/ch000549.htm#dospath)
|
||||
to change the PATH.
|
||||
|
||||
:::{warning}
|
||||
As of early 2021, users have reported problems with the Microsoft Store version of
|
||||
Python and OCRmyPDF. These issues affect many other third party Python packages.
|
||||
Please download Python from Python.org or a package manager instead of the
|
||||
Microsoft Store version.
|
||||
:::
|
||||
|
||||
:::{warning}
|
||||
32-bit Windows is not supported.
|
||||
:::
|
||||
|
||||
### Windows Subsystem for Linux
|
||||
|
||||
1. Install Ubuntu 22.04 for Windows Subsystem for Linux, if not already installed.
|
||||
2. Follow the procedure to install {ref}`OCRmyPDF on Ubuntu 22.04 <ubuntu-lts-latest>`.
|
||||
3. Open the Windows command prompt and create a symlink:
|
||||
|
||||
```powershell
|
||||
wsl sudo ln -s /home/$USER/.local/bin/ocrmypdf /usr/local/bin/ocrmypdf
|
||||
```
|
||||
|
||||
Then confirm that the expected version from PyPI ({{ latest }}) is installed:
|
||||
|
||||
```powershell
|
||||
wsl ocrmypdf --version
|
||||
```
|
||||
|
||||
You can then run OCRmyPDF in the Windows command prompt or Powershell, prefixing
|
||||
`wsl`, and call it from Windows programs or batch files.
|
||||
|
||||
### Cygwin64
|
||||
|
||||
First install the the following prerequisite Cygwin packages using `setup-x86_64.exe`:
|
||||
|
||||
```
|
||||
python311 (or later)
|
||||
python3?-devel
|
||||
python3?-pip
|
||||
python3?-lxml
|
||||
python3?-imaging
|
||||
|
||||
(where 3? means match the version of python3 you installed)
|
||||
|
||||
gcc-g++
|
||||
ghostscript
|
||||
libexempi3
|
||||
libexempi-devel
|
||||
libffi6
|
||||
libffi-devel
|
||||
pngquant
|
||||
qpdf
|
||||
libqpdf-devel
|
||||
tesseract-ocr
|
||||
tesseract-ocr-devel
|
||||
```
|
||||
|
||||
Then open a Cygwin terminal (i.e. `mintty`), run the following commands. Note
|
||||
that if you are using the version of `pip` that was installed with the Cygwin
|
||||
Python package, the command name will be `pip3`. If you have since updated
|
||||
`pip` (with, for instance `pip3 install --upgrade pip`) the the command is
|
||||
likely just `pip` instead of `pip3`:
|
||||
|
||||
```bash
|
||||
pip3 install wheel
|
||||
pip3 install ocrmypdf
|
||||
```
|
||||
|
||||
The optional dependency "unpaper" that is currently not available under Cygwin.
|
||||
Without it, certain options such as `--clean` will produce an error message.
|
||||
However, the OCR-to-text-layer functionality is available.
|
||||
|
||||
### Docker
|
||||
|
||||
You can also [Install the Docker image](docker) on Windows. Ensure that
|
||||
your command prompt can run the docker "hello world" container.
|
||||
|
||||
## Installing on FreeBSD
|
||||
|
||||
:::{image} https://repology.org/badge/version-for-repo/freebsd/ocrmypdf.svg
|
||||
:alt: FreeBSD
|
||||
:target: https://repology.org/project/ocrmypdf/versions
|
||||
:::
|
||||
|
||||
```bash
|
||||
pkg install textproc/py-ocrmypdf
|
||||
```
|
||||
|
||||
To install a more recent version, you could attempt to first install the system
|
||||
version with `pkg`, then use `pip install --user ocrmypdf`.
|
||||
|
||||
## Installing the Docker image
|
||||
|
||||
For some users, installing the Docker image will be easier than
|
||||
installing all of OCRmyPDF's dependencies.
|
||||
|
||||
See [Installing the Docker image](docker) for more information.
|
||||
|
||||
(installing-with-python-pip)=
|
||||
|
||||
## Installing with Python pip
|
||||
|
||||
OCRmyPDF is delivered by PyPI because it is a convenient way to install
|
||||
the latest version. However, PyPI and `pip` cannot address the fact
|
||||
that `ocrmypdf` depends on certain non-Python system libraries and
|
||||
programs being installed.
|
||||
|
||||
For best results, first install [your platform's
|
||||
version](https://repology.org/metapackage/ocrmypdf/versions) of
|
||||
`ocrmypdf`, using the instructions elsewhere in this document. Then
|
||||
you can use `pip` to get the latest version if your platform version
|
||||
is out of date. Chances are that this will satisfy most dependencies.
|
||||
|
||||
Use `ocrmypdf --version` to confirm what version was installed.
|
||||
|
||||
Then you can install the latest OCRmyPDF from the Python wheels. First
|
||||
try:
|
||||
|
||||
```bash
|
||||
pip install --user ocrmypdf
|
||||
```
|
||||
|
||||
(If the message appears `Requirement already satisfied: ocrmypdf in...`,
|
||||
you will need to use `pip install --user --upgrade ocrmypdf`.)
|
||||
|
||||
You should then be able to run `ocrmypdf --version` and see that the
|
||||
latest version was located.
|
||||
|
||||
## Installing with pipx
|
||||
|
||||
Some users may prefer pipx. As with the method above, you will need to
|
||||
satisfy all non-Python dependencies. Then if pipx is installed, you
|
||||
can use
|
||||
|
||||
```bash
|
||||
pipx run ocrmypdf
|
||||
```
|
||||
|
||||
(If not installed, pipx will install first.)
|
||||
|
||||
(requirements-for-pip-and-head-install)=
|
||||
|
||||
### Requirements for pip and HEAD install
|
||||
|
||||
OCRmyPDF currently requires these external programs and libraries to be
|
||||
installed, and must be satisfied using the operating system package
|
||||
manager. `pip` cannot provide them.
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
Ghostscript is now optional. pypdfium2 can be used for PDF rasterization,
|
||||
and verapdf can validate speculative PDF/A conversion.
|
||||
:::
|
||||
|
||||
The following versions are required:
|
||||
|
||||
- Python 3.11 or newer
|
||||
- Tesseract 4.1.1 or newer
|
||||
- One of: Ghostscript 9.54+ **or** pypdfium2 (Python package)
|
||||
- One of: Ghostscript 9.54+ **or** verapdf (for PDF/A output)
|
||||
- fpdf2 2.8 or newer (Python package)
|
||||
- jbig2enc 0.29 or newer (optional)
|
||||
- pngquant 2.5 or newer (optional)
|
||||
- unpaper 6.1 (optional)
|
||||
|
||||
:::{note}
|
||||
For the best user experience, install both Ghostscript and pypdfium2.
|
||||
pypdfium2 is faster for rasterization, while Ghostscript provides
|
||||
broader compatibility and is required for certain PDF/A conversions.
|
||||
:::
|
||||
|
||||
We recommend 64-bit versions of all software. (32-bit versions are not
|
||||
supported, although on Linux, they may still work.)
|
||||
|
||||
**fpdf2** is a required dependency that provides the text layer
|
||||
rendering engine. It replaces the legacy hOCR-based renderer with improved
|
||||
multilingual support. Install with: `pip install fpdf2`
|
||||
|
||||
**pypdfium2**, if present, provides fast PDF page rasterization using
|
||||
the pdfium library (the same library used by Google Chrome). It is
|
||||
preferred over Ghostscript when available due to better performance.
|
||||
Install with: `pip install pypdfium2`
|
||||
|
||||
**verapdf**, if present, enables fast speculative PDF/A conversion.
|
||||
OCRmyPDF attempts to create PDF/A by adding metadata and ICC profiles
|
||||
using pikepdf, then validates with verapdf. If validation passes,
|
||||
Ghostscript is skipped entirely. See your distribution's package manager
|
||||
or visit [verapdf.org](https://verapdf.org/).
|
||||
|
||||
**jbig2enc**, if present, will be used to optimize the encoding of
|
||||
monochrome images. This can significantly reduce the file size of the
|
||||
output file. It is not required.
|
||||
[jbig2enc](https://github.com/agl/jbig2enc) is not generally
|
||||
available for Ubuntu or Debian due to lingering concerns about patent
|
||||
issues, but can easily be built from source. To add JBIG2 encoding, see
|
||||
{ref}`jbig2`.
|
||||
|
||||
:::{warning}
|
||||
Lossy JBIG2 encoding (`--jbig2-lossy`) has been removed in v17.0.0 due to
|
||||
well-documented risks of character substitution errors. Only lossless
|
||||
JBIG2 compression is now supported.
|
||||
:::
|
||||
|
||||
**pngquant**, if present, is optionally used to optimize the encoding of
|
||||
PNG-style images in PDFs (actually, any that are that losslessly
|
||||
encoded) by lossily quantizing to a smaller color palette. It is only
|
||||
activated then the `--optimize` argument is `2` or `3`.
|
||||
|
||||
**unpaper**, if present, enables the `--clean` and `--clean-final`
|
||||
command line options.
|
||||
|
||||
These are in addition to the Python packaging dependencies, meaning that
|
||||
unfortunately, the `pip install` command cannot satisfy all of them.
|
||||
|
||||
(installing-head-revision-from-sources)=
|
||||
|
||||
## Installing HEAD revision from sources
|
||||
|
||||
If you have `git` and Python 3.11 or newer installed, you can install
|
||||
from source. When the `pip` installer runs, it will alert you if
|
||||
dependencies are missing.
|
||||
|
||||
If you prefer to build every from source, you will need to [build
|
||||
pikepdf from
|
||||
source](https://pikepdf.readthedocs.io/en/latest/installation.html#building-from-source).
|
||||
First ensure you can build and install pikepdf.
|
||||
|
||||
To install the HEAD revision from sources in the current Python 3
|
||||
environment:
|
||||
|
||||
```bash
|
||||
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
```
|
||||
|
||||
Or, to install in editable mode
|
||||
allowing customization of OCRmyPDF, use the `-e` flag:
|
||||
|
||||
```bash
|
||||
pip install -e git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
```
|
||||
|
||||
You may find it easiest to install in a virtual environment, rather than
|
||||
system-wide:
|
||||
|
||||
```bash
|
||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
python3 -m venv .venv
|
||||
source .venv/bin/activate
|
||||
cd OCRmyPDF
|
||||
pip install .
|
||||
```
|
||||
|
||||
However, `ocrmypdf` will only be accessible on the system PATH when
|
||||
you activate the virtual environment.
|
||||
|
||||
To run the program:
|
||||
|
||||
```bash
|
||||
ocrmypdf --help
|
||||
```
|
||||
|
||||
If not yet installed, the script will notify you about dependencies that
|
||||
need to be installed. The script requires specific versions of the
|
||||
dependencies. Older version than the ones mentioned in the release notes
|
||||
are likely not to be compatible to OCRmyPDF.
|
||||
|
||||
## Optional Features
|
||||
|
||||
OCRmyPDF provides optional features and development tools. We recommend using `uv` as your package manager.
|
||||
|
||||
### Installing User Features
|
||||
|
||||
User features are available as optional dependencies. Install them with `uv` (recommended) or `pip`:
|
||||
|
||||
```bash
|
||||
# Using uv (recommended)
|
||||
uv sync --extra watcher # File watching service
|
||||
uv sync --extra webservice # Streamlit web UI
|
||||
uv sync --extra watcher --extra webservice # Multiple features
|
||||
|
||||
# Using pip (also works)
|
||||
pip install ocrmypdf[watcher]
|
||||
pip install ocrmypdf[webservice]
|
||||
pip install ocrmypdf[watcher,webservice]
|
||||
```
|
||||
|
||||
### Development Tools (uv only)
|
||||
|
||||
Development tools use dependency groups and require `uv`:
|
||||
|
||||
```bash
|
||||
# Testing infrastructure
|
||||
uv sync --group test
|
||||
|
||||
# Documentation building
|
||||
uv sync --group docs
|
||||
|
||||
# Enhanced Streamlit development
|
||||
uv sync --group streamlit-dev
|
||||
|
||||
# All development groups
|
||||
uv sync
|
||||
```
|
||||
|
||||
:::{note}
|
||||
**User features** (`watcher`, `webservice`) work with both `uv` and `pip`.
|
||||
**Developer tools** (`test`, `docs`, `streamlit-dev`) require `uv` and use dependency groups (PEP 735).
|
||||
:::
|
||||
|
||||
**Why use uv?**
|
||||
|
||||
- Modern, fast Python package manager
|
||||
- Required for development (testing, docs)
|
||||
- Better dependency resolution
|
||||
- Consistent across all platforms
|
||||
|
||||
Install uv: `pip install uv` or visit https://docs.astral.sh/uv/
|
||||
|
||||
### For development
|
||||
|
||||
To install all of the development and test requirements:
|
||||
|
||||
```bash
|
||||
git clone -b main https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
cd OCRmyPDF
|
||||
pip install uv # Install uv if not already installed
|
||||
uv sync --group test
|
||||
```
|
||||
|
||||
Note: Development requires `uv`. The old `pip install -e .[test]` method is no longer supported.
|
||||
|
||||
To add JBIG2 encoding, see {ref}`jbig2`.
|
||||
|
||||
## Shell completions
|
||||
|
||||
Completions for `bash` and `fish` are available in the project's
|
||||
`misc/completion` folder. The `bash` completions are likely `zsh`
|
||||
compatible but this has not been confirmed. Package maintainers, please
|
||||
install these at the appropriate locations for your system.
|
||||
|
||||
To manually install the `bash` completion, copy
|
||||
`misc/completion/ocrmypdf.bash` to `/etc/bash_completion.d/ocrmypdf`
|
||||
(rename the file).
|
||||
|
||||
To manually install the `fish` completion, copy
|
||||
`misc/completion/ocrmypdf.fish` to
|
||||
`~/.config/fish/completions/ocrmypdf.fish`.
|
||||
|
||||
## Note on 32-bit support
|
||||
|
||||
Many Python libraries no longer provide 32-bit binary wheels for Linux. This
|
||||
includes many of the libraries that OCRmyPDF depends on, such as
|
||||
Pillow. The easiest way to express this to end users is to say we don't
|
||||
support 32-bit Linux.
|
||||
|
||||
However, if your Linux distribution still supports 32-bit binaries, you
|
||||
can still install and use OCRmyPDF. A warning message will appear.
|
||||
In practice, OCRmyPDF may need more than 32-bit memory space to run when
|
||||
large documents are processed, so there are practical limitations to what
|
||||
users can accomplish with it. Still, for the common use case of an 32-bit
|
||||
ARM NAS or Raspberry Pi processing small documents, it should work.
|
||||
@@ -1,689 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
===================
|
||||
Installing OCRmyPDF
|
||||
===================
|
||||
|
||||
.. |latest| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||
:alt: OCRmyPDF latest released version on PyPI
|
||||
|
||||
|latest|
|
||||
|
||||
The easiest way to install OCRmyPDF is to follow the steps for your operating
|
||||
system/platform. This version may be out of date, however.
|
||||
|
||||
These platforms have one-liner installs:
|
||||
|
||||
+-------------------------------+-----------------------------------------+
|
||||
| Debian, Ubuntu | ``apt install ocrmypdf`` |
|
||||
+-------------------------------+-----------------------------------------+
|
||||
| Windows Subsystem for Linux | ``apt install ocrmypdf`` |
|
||||
+-------------------------------+-----------------------------------------+
|
||||
| Fedora | ``dnf install ocrmypdf`` |
|
||||
+-------------------------------+-----------------------------------------+
|
||||
| macOS | ``brew install ocrmypdf`` |
|
||||
+-------------------------------+-----------------------------------------+
|
||||
| LinuxBrew | ``brew install ocrmypdf`` |
|
||||
+-------------------------------+-----------------------------------------+
|
||||
| FreeBSD | ``pkg install textproc/py-ocrmypdf`` |
|
||||
+-------------------------------+-----------------------------------------+
|
||||
| Conda (WSL, macOS, Linux) | ``conda install ocrmypdf`` |
|
||||
+-------------------------------+-----------------------------------------+
|
||||
| Snap (snapcraft packaging) | ``snap install ocrmypdf`` |
|
||||
+-------------------------------+-----------------------------------------+
|
||||
|
||||
More detailed procedures are outlined below. If you want to do a manual
|
||||
install, or install a more recent version than your platform provides, read on.
|
||||
|
||||
.. contents:: Platform-specific steps
|
||||
:depth: 2
|
||||
:local:
|
||||
|
||||
Installing on Linux
|
||||
===================
|
||||
|
||||
Debian and Ubuntu 20.04 or newer
|
||||
--------------------------------
|
||||
|
||||
.. |deb-11| image:: https://repology.org/badge/version-for-repo/debian_11/ocrmypdf.svg
|
||||
:alt: Debian 11
|
||||
|
||||
.. |deb-12| image:: https://repology.org/badge/version-for-repo/debian_12/ocrmypdf.svg
|
||||
:alt: Debian 12
|
||||
|
||||
.. |deb-unstable| image:: https://repology.org/badge/version-for-repo/debian_unstable/ocrmypdf.svg
|
||||
:alt: Debian unstable
|
||||
|
||||
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
||||
:alt: Ubuntu 20.04 LTS
|
||||
|
||||
.. |ubu-2204| image:: https://repology.org/badge/version-for-repo/ubuntu_22_04/ocrmypdf.svg
|
||||
:alt: Ubuntu 22.04 LTS
|
||||
|
||||
+-----------------------------------------------+
|
||||
| **OCRmyPDF versions in Debian & Ubuntu** |
|
||||
+-----------------------------------------------+
|
||||
| |latest| |
|
||||
+-----------------------------------------------+
|
||||
| |deb-11| |deb-12| |deb-unstable| |
|
||||
+-----------------------------------------------+
|
||||
| |ubu-2004| |ubu-2204| |
|
||||
+-----------------------------------------------+
|
||||
|
||||
Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users
|
||||
of Windows Subsystem for Linux, may simply
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
apt install ocrmypdf
|
||||
|
||||
As indicated in the table above, Debian and Ubuntu releases may lag
|
||||
behind the latest version. If the version available for your platform is
|
||||
out of date, you could opt to install the latest version from source.
|
||||
See `Installing HEAD revision from
|
||||
sources <#installing-head-revision-from-sources>`__.
|
||||
|
||||
For full details on version availability for your platform, check the
|
||||
`Debian Package Tracker <https://tracker.debian.org/pkg/ocrmypdf>`__ or
|
||||
`Ubuntu launchpad.net <https://launchpad.net/ocrmypdf>`__.
|
||||
|
||||
.. note::
|
||||
|
||||
OCRmyPDF for Debian and Ubuntu currently omit the JBIG2 encoder.
|
||||
OCRmyPDF works fine without it but will produce larger output files.
|
||||
If you build jbig2enc from source, ocrmypdf will
|
||||
automatically detect it (specifically the ``jbig2`` binary) on the
|
||||
``PATH``. To add JBIG2 encoding, see :ref:`jbig2`.
|
||||
|
||||
Fedora
|
||||
------
|
||||
|
||||
.. |fedora-35| image:: https://repology.org/badge/version-for-repo/fedora_35/ocrmypdf.svg
|
||||
:alt: Fedora 35
|
||||
|
||||
.. |fedora-36| image:: https://repology.org/badge/version-for-repo/fedora_36/ocrmypdf.svg
|
||||
:alt: Fedora 36
|
||||
|
||||
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
|
||||
:alt: Fedore Rawhide
|
||||
|
||||
+-----------------------------------------------+
|
||||
| **OCRmyPDF version** |
|
||||
+-----------------------------------------------+
|
||||
| |latest| |
|
||||
+-----------------------------------------------+
|
||||
| |fedora-35| |fedora-36| |fedora-rawhide| |
|
||||
+-----------------------------------------------+
|
||||
|
||||
Users of Fedora 29 or later may simply
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
dnf install ocrmypdf
|
||||
|
||||
For full details on version availability, check the `Fedora Package
|
||||
Tracker <https://apps.fedoraproject.org/packages/ocrmypdf>`__.
|
||||
|
||||
If the version available for your platform is out of date, you could opt
|
||||
to install the latest version from source. See `Installing HEAD revision
|
||||
from sources <#installing-head-revision-from-sources>`__.
|
||||
|
||||
.. note::
|
||||
|
||||
OCRmyPDF for Fedora currently omits the JBIG2 encoder due to patent
|
||||
issues. OCRmyPDF works fine without it but will produce larger output
|
||||
files. If you build jbig2enc from source, ocrmypdf 7.0.0 and later
|
||||
will automatically detect it on the ``PATH``. To add JBIG2 encoding,
|
||||
see `Installing the JBIG2 encoder <jbig2>`__.
|
||||
|
||||
.. _ubuntu-lts-latest:
|
||||
|
||||
Installing the latest version on Ubuntu 22.04 LTS
|
||||
-------------------------------------------------
|
||||
|
||||
Ubuntu 22.04 includes ocrmypdf 13.4.0 - you can install that with
|
||||
``apt install ocrmypdf``. To install a more recent version for the current
|
||||
user, follow these steps:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo apt-get update
|
||||
sudo apt-get -y install ocrmypdf python3-pip
|
||||
|
||||
pip install --user --upgrade ocrmypdf
|
||||
|
||||
If you get the message ``WARNING: The script ocrmypdf is installed in
|
||||
'/home/$USER/.local/bin' which is not on PATH.``, you may need to re-login
|
||||
or open a new shell, or manually add this to your user's PATH.
|
||||
|
||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||
|
||||
Ubuntu 20.04 LTS
|
||||
----------------
|
||||
|
||||
Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with ``apt``. To
|
||||
install a more recent version, uninstall the system-provided version of
|
||||
ocrmypdf, and install the following dependencies:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo apt-get -y remove ocrmypdf # remove system ocrmypdf, if installed
|
||||
sudo apt-get -y update
|
||||
sudo apt-get -y install \
|
||||
ghostscript \
|
||||
icc-profiles-free \
|
||||
libxml2 \
|
||||
pngquant \
|
||||
python3-pip \
|
||||
tesseract-ocr \
|
||||
zlib1g
|
||||
|
||||
To install ocrmypdf for the system:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip3 install ocrmypdf
|
||||
|
||||
To install for the current user only:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
export PATH=$HOME/.local/bin:$PATH
|
||||
pip3 install --user ocrmypdf
|
||||
|
||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||
|
||||
Arch Linux (AUR)
|
||||
----------------
|
||||
|
||||
.. image:: https://repology.org/badge/version-for-repo/aur/ocrmypdf.svg
|
||||
:alt: ArchLinux
|
||||
:target: https://repology.org/metapackage/ocrmypdf
|
||||
|
||||
There is an `Arch User Repository (AUR) package for OCRmyPDF
|
||||
<https://aur.archlinux.org/packages/ocrmypdf/>`__.
|
||||
|
||||
Installing AUR packages as root is not allowed, so you must first `setup a
|
||||
non-root user
|
||||
<https://wiki.archlinux.org/index.php/Users_and_groups#User_management>`__ and
|
||||
`configure sudo <https://wiki.archlinux.org/index.php/Sudo#Configuration>`__.
|
||||
The standard Docker image, ``archlinux/base:latest``, does **not** have a
|
||||
non-root user configured, so users of that image must follow these guides. If
|
||||
you are using a VM image, such as `the official Vagrant image
|
||||
<https://app.vagrantup.com/archlinux/boxes/archlinux>`__, this work may already
|
||||
be completed for you.
|
||||
|
||||
Next you should install the `base-devel package group
|
||||
<https://www.archlinux.org/groups/x86_64/base-devel/>`__. This includes the
|
||||
standard tooling needed to build packages, such as a compiler and binary tools.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo pacman -S base-devel
|
||||
|
||||
Now you are ready to install the OCRmyPDF package.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
curl -O https://aur.archlinux.org/cgit/aur.git/snapshot/ocrmypdf.tar.gz
|
||||
tar xvzf ocrmypdf.tar.gz
|
||||
cd ocrmypdf
|
||||
makepkg -sri
|
||||
|
||||
At this point you will have a working install of OCRmyPDF, but the Tesseract
|
||||
install won’t include any OCR language data. You can install `the
|
||||
tesseract-data package group
|
||||
<https://www.archlinux.org/groups/any/tesseract-data/>`__ to add all supported
|
||||
languages, or use that package listing to identify the appropriate package for
|
||||
your desired language.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo pacman -S tesseract-data-eng
|
||||
|
||||
As an alternative to this manual procedure, consider using an `AUR helper
|
||||
<https://wiki.archlinux.org/index.php/AUR_helpers>`__. Such a tool will
|
||||
automatically fetch, build and install the AUR package, resolve dependencies
|
||||
(including dependencies on AUR packages), and ease the upgrade procedure.
|
||||
|
||||
If you have any difficulties with installation, check the repository package
|
||||
page.
|
||||
|
||||
.. note::
|
||||
|
||||
The OCRmyPDF AUR package currently omits the JBIG2 encoder. OCRmyPDF works
|
||||
fine without it but will produce larger output files. The encoder is
|
||||
available from `the jbig2enc-git AUR package
|
||||
<https://aur.archlinux.org/packages/jbig2enc-git/>`__ and may be installed
|
||||
using the same series of steps as for the installation OCRmyPDF AUR
|
||||
package. Alternatively, it may be built manually from source following the
|
||||
instructions in `Installing the JBIG2 encoder <jbig2>`__. If JBIG2 is
|
||||
installed, OCRmyPDF 7.0.0 and later will automatically detect it.
|
||||
|
||||
Alpine Linux
|
||||
------------
|
||||
|
||||
.. image:: https://repology.org/badge/version-for-repo/alpine_edge/ocrmypdf.svg
|
||||
:alt: Alpine Linux
|
||||
:target: https://repology.org/metapackage/ocrmypdf
|
||||
|
||||
To install OCRmyPDF for Alpine Linux:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
apk add ocrmypdf
|
||||
|
||||
Other Linux packages
|
||||
--------------------
|
||||
|
||||
See the
|
||||
`Repology <https://repology.org/metapackage/ocrmypdf/versions>`__ page.
|
||||
|
||||
In general, first install the OCRmyPDF package for your system, then
|
||||
optionally use the procedure `Installing with Python
|
||||
pip <#installing-with-python-pip>`__ to install a more recent version.
|
||||
|
||||
Installing on macOS
|
||||
===================
|
||||
|
||||
Homebrew
|
||||
--------
|
||||
|
||||
.. image:: https://img.shields.io/homebrew/v/ocrmypdf.svg
|
||||
:alt: homebrew
|
||||
:target: http://brewformulas.org/Ocrmypdf
|
||||
|
||||
OCRmyPDF is now a standard `Homebrew <https://brew.sh>`__ formula. To
|
||||
install on macOS:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
brew install ocrmypdf
|
||||
|
||||
This will include only the English language pack. If you need other
|
||||
languages you can optionally install them all:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
brew install tesseract-lang # Optional: Install all language packs
|
||||
|
||||
Manual installation on macOS
|
||||
----------------------------
|
||||
|
||||
These instructions probably work on all macOS supported by Homebrew, and are
|
||||
for installing a more current version of OCRmyPDF than is available from
|
||||
Homebrew. Note that the Homebrew versions usually track the release versions
|
||||
fairly closely.
|
||||
|
||||
If it's not already present, `install Homebrew <http://brew.sh/>`__.
|
||||
|
||||
Update Homebrew:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
brew update
|
||||
|
||||
Install or upgrade the required Homebrew packages, if any are missing.
|
||||
To do this, use ``brew edit ocrmypdf`` to obtain a recent list of Homebrew
|
||||
dependencies. You could also check the ``.workflows/build.yml``.
|
||||
|
||||
This will include the English, French, German and Spanish language
|
||||
packs. If you need other languages you can optionally install them all:
|
||||
|
||||
.. _macos-all-languages:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
brew install tesseract-lang # Option 2: for all language packs
|
||||
|
||||
Update the homebrew pip:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install --upgrade pip
|
||||
|
||||
You can then install OCRmyPDF from PyPI, for the current user:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install --user ocrmypdf
|
||||
|
||||
or system-wide:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install ocrmypdf
|
||||
|
||||
The command line program should now be available:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --help
|
||||
|
||||
Installing on Windows
|
||||
=====================
|
||||
|
||||
Native Windows
|
||||
--------------
|
||||
|
||||
.. note::
|
||||
|
||||
Administrator privileges will be required for some of these steps.
|
||||
|
||||
You must install the following for Windows:
|
||||
|
||||
* Python 3.8 (64-bit) or later
|
||||
* Tesseract 4.1.1 (64-bit) or later
|
||||
* Ghostscript 9.50 (64-bit) or later
|
||||
|
||||
Using the `Chocolatey <https://chocolatey.org/>`_ package manager, install the
|
||||
following when running in an Administrator command prompt:
|
||||
|
||||
* ``choco install python3``
|
||||
* ``choco install --pre tesseract``
|
||||
* ``choco install ghostscript``
|
||||
* ``choco install pngquant`` (optional)
|
||||
|
||||
The commands above will install Python 3.x (latest version), Tesseract, Ghostscript
|
||||
and pngquant. Chocolatey may also need to install the Windows Visual C++ Runtime
|
||||
DLLs or other Windows patches, and may require a reboot.
|
||||
|
||||
You may then use ``pip`` to install ocrmypdf. (This can performed by a user or
|
||||
Administrator.):
|
||||
|
||||
* ``pip install ocrmypdf``
|
||||
|
||||
Chocolatey automatically selects appropriate versions of these applications. Please make sure
|
||||
you are installing the 64-bit versions.
|
||||
|
||||
OCRmyPDF will check the Windows Registry and standard locations in your Program Files
|
||||
for third party software it needs (specifically, Tesseract and Ghostscript). To
|
||||
override the versions OCRmyPDF selects, you can modify the ``PATH`` environment
|
||||
variable. `Follow these directions <https://www.computerhope.com/issues/ch000549.htm#dospath>`_
|
||||
to change the PATH.
|
||||
|
||||
.. warning::
|
||||
|
||||
As of early 2021, users have reported problems with the Microsoft Store version of
|
||||
Python and OCRmyPDF. These issues affect many other third party Python packages.
|
||||
Please download Python from Python.org or Chocolatey instead, and do not use the
|
||||
Microsoft Store version.
|
||||
|
||||
.. warning::
|
||||
|
||||
32-bit Windows might work, but is not supported.
|
||||
|
||||
Windows Subsystem for Linux
|
||||
---------------------------
|
||||
|
||||
#. Install Ubuntu 22.04 for Windows Subsystem for Linux, if not already installed.
|
||||
#. Follow the procedure to install :ref:`OCRmyPDF on Ubuntu 22.04 <ubuntu-lts-latest>`.
|
||||
#. Open the Windows command prompt and create a symlink:
|
||||
|
||||
.. code-block:: powershell
|
||||
|
||||
wsl sudo ln -s /home/$USER/.local/bin/ocrmypdf /usr/local/bin/ocrmypdf
|
||||
|
||||
Then confirm that the expected version from PyPI (|latest|) is installed:
|
||||
|
||||
.. code-block:: powershell
|
||||
|
||||
wsl ocrmypdf --version
|
||||
|
||||
You can then run OCRmyPDF in the Windows command prompt or Powershell, prefixing
|
||||
``wsl``, and call it from Windows programs or batch files.
|
||||
|
||||
Cygwin64
|
||||
--------
|
||||
|
||||
First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``::
|
||||
|
||||
python38 (or later)
|
||||
python3?-devel
|
||||
python3?-pip
|
||||
python3?-lxml
|
||||
python3?-imaging
|
||||
|
||||
(where 3? means match the version of python3 you installed)
|
||||
|
||||
gcc-g++
|
||||
ghostscript (<=9.50 or >=9.52-2 see note below)
|
||||
libexempi3
|
||||
libexempi-devel
|
||||
libffi6
|
||||
libffi-devel
|
||||
pngquant
|
||||
qpdf
|
||||
libqpdf-devel
|
||||
tesseract-ocr
|
||||
tesseract-ocr-devel
|
||||
|
||||
.. note::
|
||||
|
||||
The Cygwin package for Ghostscript in versions 9.52 and
|
||||
9.52-1 contained a bug that caused an exception to occur when
|
||||
ocrmypdf invoked gs. Make sure you have either 9.50 (or earlier)
|
||||
or 9.52-2 (or later).
|
||||
|
||||
Then open a Cygwin terminal (i.e. ``mintty``), run the following commands. Note
|
||||
that if you are using the version of ``pip`` that was installed with the Cygwin
|
||||
Python package, the command name will be ``pip3``. If you have since updated
|
||||
``pip`` (with, for instance ``pip3 install --upgrade pip``) the the command is
|
||||
likely just ``pip`` instead of ``pip3``:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip3 install wheel
|
||||
pip3 install ocrmypdf
|
||||
|
||||
The optional dependency "unpaper" that is currently not available under Cygwin.
|
||||
Without it, certain options such as ``--clean`` will produce an error message.
|
||||
However, the OCR-to-text-layer functionality is available.
|
||||
|
||||
Docker
|
||||
------
|
||||
|
||||
You can also :ref:`Install the Docker <docker>` container on Windows. Ensure that
|
||||
your command prompt can run the docker "hello world" container.
|
||||
|
||||
Installing on FreeBSD
|
||||
=====================
|
||||
|
||||
.. image:: https://repology.org/badge/version-for-repo/freebsd/ocrmypdf.svg
|
||||
:alt: FreeBSD
|
||||
:target: https://repology.org/project/ocrmypdf/versions
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pkg install textproc/py-ocrmypdf
|
||||
|
||||
To install a more recent version, you could attempt to first install the system
|
||||
version with ``pkg``, then use ``pip install --user ocrmypdf``.
|
||||
|
||||
Installing the Docker image
|
||||
===========================
|
||||
|
||||
For some users, installing the Docker image will be easier than
|
||||
installing all of OCRmyPDF's dependencies.
|
||||
|
||||
See :ref:`docker` for more information.
|
||||
|
||||
Installing with Python pip
|
||||
==========================
|
||||
|
||||
OCRmyPDF is delivered by PyPI because it is a convenient way to install
|
||||
the latest version. However, PyPI and ``pip`` cannot address the fact
|
||||
that ``ocrmypdf`` depends on certain non-Python system libraries and
|
||||
programs being installed.
|
||||
|
||||
.. warning::
|
||||
|
||||
Debian and Ubuntu users: unfortunately, Debian and Ubuntu customize
|
||||
Python in non-standard ways, and the nature of these customizations
|
||||
varies from release to release. This can make for a frustrating
|
||||
user experience. The instructions below work on almost all platforms that
|
||||
have Python installed, except for Debian and Ubuntu, where you may need
|
||||
to take additional steps. For best results on Debian and Ubuntu, use the
|
||||
``apt`` packages; or if these are too old, run
|
||||
``apt install python3-pip python3-venv``, create a virtual environment,
|
||||
and install OCRmyPDF in that environment.
|
||||
|
||||
`See here for more inforation on Debian-Python issues
|
||||
<https://gist.github.com/tiran/2dec9e03c6f901814f6d1e8dad09528e>`__.
|
||||
|
||||
For best results, first install `your platform's
|
||||
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
|
||||
``ocrmypdf``, using the instructions elsewhere in this document. Then
|
||||
you can use ``pip`` to get the latest version if your platform version
|
||||
is out of date. Chances are that this will satisfy most dependencies.
|
||||
|
||||
Use ``ocrmypdf --version`` to confirm what version was installed.
|
||||
|
||||
Then you can install the latest OCRmyPDF from the Python wheels. First
|
||||
try:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install --user ocrmypdf
|
||||
|
||||
You should then be able to run ``ocrmypdf --version`` and see that the
|
||||
latest version was located.
|
||||
|
||||
Since ``pip install --user`` does not work correctly on some platforms,
|
||||
notably Ubuntu 16.04 and older, and the Homebrew version of Python,
|
||||
instead use this for a system wide installation:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install ocrmypdf
|
||||
|
||||
.. note::
|
||||
|
||||
AArch64 (ARM64) users: this process will be difficult because most
|
||||
Python packages are not available as binary wheels for your platform.
|
||||
You're probably better off using a platform install on Debian, Ubuntu,
|
||||
or Fedora.
|
||||
|
||||
Requirements for pip and HEAD install
|
||||
-------------------------------------
|
||||
|
||||
OCRmyPDF currently requires these external programs and libraries to be
|
||||
installed, and must be satisfied using the operating system package
|
||||
manager. ``pip`` cannot provide them.
|
||||
|
||||
The following versions are required:
|
||||
|
||||
- Python 3.8 or newer
|
||||
- Ghostscript 9.50 or newer
|
||||
- Tesseract 4.1.1 or newer
|
||||
- jbig2enc 0.29 or newer
|
||||
- pngquant 2.5 or newer
|
||||
- unpaper 6.1
|
||||
|
||||
jbig2enc, pngquant, and unpaper are optional. If missing certain
|
||||
features are disabled. OCRmyPDF will discover them as soon as they are
|
||||
available.
|
||||
|
||||
**jbig2enc**, if present, will be used to optimize the encoding of
|
||||
monochrome images. This can significantly reduce the file size of the
|
||||
output file. It is not required.
|
||||
`jbig2enc <https://github.com/agl/jbig2enc>`__ is not generally
|
||||
available for Ubuntu or Debian due to lingering concerns about patent
|
||||
issues, but can easily be built from source. To add JBIG2 encoding, see
|
||||
:ref:`jbig2`.
|
||||
|
||||
**pngquant**, if present, is optionally used to optimize the encoding of
|
||||
PNG-style images in PDFs (actually, any that are that losslessly
|
||||
encoded) by lossily quantizing to a smaller color palette. It is only
|
||||
activated then the ``--optimize`` argument is ``2`` or ``3``.
|
||||
|
||||
**unpaper**, if present, enables the ``--clean`` and ``--clean-final``
|
||||
command line options.
|
||||
|
||||
These are in addition to the Python packaging dependencies, meaning that
|
||||
unfortunately, the ``pip install`` command cannot satisfy all of them.
|
||||
|
||||
Installing HEAD revision from sources
|
||||
=====================================
|
||||
|
||||
If you have ``git`` and Python 3.8 or newer installed, you can install
|
||||
from source. When the ``pip`` installer runs, it will alert you if
|
||||
dependencies are missing.
|
||||
|
||||
If you prefer to build every from source, you will need to `build
|
||||
pikepdf from
|
||||
source <https://pikepdf.readthedocs.io/en/latest/installation.html#building-from-source>`__.
|
||||
First ensure you can build and install pikepdf.
|
||||
|
||||
To install the HEAD revision from sources in the current Python 3
|
||||
environment:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
|
||||
Or, to install in `development
|
||||
mode <https://pythonhosted.org/setuptools/setuptools.html#development-mode>`__,
|
||||
allowing customization of OCRmyPDF, use the ``-e`` flag:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install -e git+https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
|
||||
You may find it easiest to install in a virtual environment, rather than
|
||||
system-wide:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
git clone -b master https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
python3 -m venv venv
|
||||
source venv/bin/activate
|
||||
cd OCRmyPDF
|
||||
pip install .
|
||||
|
||||
However, ``ocrmypdf`` will only be accessible on the system PATH when
|
||||
you activate the virtual environment.
|
||||
|
||||
To run the program:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --help
|
||||
|
||||
If not yet installed, the script will notify you about dependencies that
|
||||
need to be installed. The script requires specific versions of the
|
||||
dependencies. Older version than the ones mentioned in the release notes
|
||||
are likely not to be compatible to OCRmyPDF.
|
||||
|
||||
For development
|
||||
---------------
|
||||
|
||||
To install all of the development and test requirements:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
git clone -b master https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
python -m venv
|
||||
source venv/bin/activate
|
||||
cd OCRmyPDF
|
||||
pip install -e .[test]
|
||||
|
||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||
|
||||
Shell completions
|
||||
=================
|
||||
|
||||
Completions for ``bash`` and ``fish`` are available in the project's
|
||||
``misc/completion`` folder. The ``bash`` completions are likely ``zsh``
|
||||
compatible but this has not been confirmed. Package maintainers, please
|
||||
install these at the appropriate locations for your system.
|
||||
|
||||
To manually install the ``bash`` completion, copy
|
||||
``misc/completion/ocrmypdf.bash`` to ``/etc/bash_completion.d/ocrmypdf``
|
||||
(rename the file).
|
||||
|
||||
To manually install the ``fish`` completion, copy
|
||||
``misc/completion/ocrmypdf.fish`` to
|
||||
``~/.config/fish/completions/ocrmypdf.fish``.
|
||||
@@ -0,0 +1,233 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Introduction
|
||||
|
||||
OCRmyPDF is a Python application and library that adds text "layers" to images in
|
||||
PDFs, making scanned image PDFs searchable. It uses OCR to guess the text
|
||||
contained in images. OCRmyPDF also supports plugins
|
||||
that enable customization of its processing steps, and it is highly tolerant
|
||||
of PDFs containing scanned images and "born digital" content that doesn't
|
||||
require text recognition.
|
||||
|
||||
## About OCR
|
||||
|
||||
[Optical character
|
||||
recognition](https://en.wikipedia.org/wiki/Optical_character_recognition)
|
||||
is a technology that converts images of typed or handwritten text, such as
|
||||
in a scanned document, into computer text that can be selected, searched and copied.
|
||||
|
||||
OCRmyPDF uses
|
||||
[Tesseract](https://github.com/tesseract-ocr/tesseract), a widely
|
||||
available open source OCR engine, to perform OCR.
|
||||
|
||||
(raster-vector)=
|
||||
|
||||
## About PDFs
|
||||
|
||||
PDFs are page description files that attempt to preserve a layout
|
||||
exactly. They contain [vector
|
||||
graphics](http://vector-conversions.com/vectorizing/raster_vs_vector.html)
|
||||
that can contain raster objects, such as scanned images. Because PDFs can
|
||||
contain multiple pages (unlike many image formats) and can contain fonts
|
||||
and text, they are a suitable format for exchanging scanned documents.
|
||||
|
||||
:::{image} images/bitmap_vs_svg.svg
|
||||
:::
|
||||
|
||||
A PDF page may contain multiple images, even if it appears to have only
|
||||
one image. Some scanners or scanning software may segment pages into
|
||||
monochromatic text and color regions, for example, to enhance the compression
|
||||
ratio and appearance of the page.
|
||||
|
||||
Rasterizing a PDF is the process of generating corresponding raster images.
|
||||
OCR engines like Tesseract work with images, not scalable vector graphics
|
||||
or mixed raster-vector-text graphics such as PDF.
|
||||
|
||||
## About PDF/A
|
||||
|
||||
[PDF/A](https://en.wikipedia.org/wiki/PDF/A) is an ISO-standardized
|
||||
subset of the full PDF specification that is designed for archiving (the
|
||||
'A' stands for Archive). PDF/A differs from PDF primarily by omitting
|
||||
features that could complicate future file readability,
|
||||
such as embedded Javascript, video, audio and references to external
|
||||
fonts. All fonts and resources needed to interpret the PDF must be
|
||||
contained within it. Because PDF/A disables Javascript and other types
|
||||
of embedded content, it is likely more secure.
|
||||
|
||||
There are various conformance levels and versions, such as "PDF/A-2b".
|
||||
|
||||
In general, the preferred format for scanned documents is PDF/A. Some
|
||||
governments and jurisdictions, US Courts in particular, [mandate the use
|
||||
of PDF/A](https://pdfblog.com/2012/02/13/what-is-pdfa/) for scanned
|
||||
documents.
|
||||
|
||||
Since most individuals scanning documents aim for long-term readability,
|
||||
OCRmyPDF defaults to generating PDF/A-2b.
|
||||
|
||||
PDF/A does have a few drawbacks. Some PDF viewers display an alert
|
||||
indicating that the file is in PDF/A format, which may confuse some users.
|
||||
Additionally, it tends to result in larger files than standard PDFs because
|
||||
it embeds certain resources, even if they are widely available. PDF/A
|
||||
files can be digitally signed but may not be encrypted to ensure future
|
||||
readability. Fortunately, converting from PDF/A to a regular PDF is
|
||||
straightforward, and any PDF viewer can handle PDF/A files.
|
||||
|
||||
## What OCRmyPDF does
|
||||
|
||||
OCRmyPDF analyzes each page of a PDF to determine the required colorspace
|
||||
and resolution (DPI) for capturing all the information on that page without
|
||||
losing content. It uses a PDF rasterizer (pypdfium2 or
|
||||
[Ghostscript](http://ghostscript.com/)) to convert each page to an image and
|
||||
subsequently performs OCR on the rasterized image to generate an OCR "layer."
|
||||
This layer is then integrated back into the original PDF.
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
OCRmyPDF now supports pypdfium2 as an alternative rasterizer to Ghostscript.
|
||||
pypdfium2 is a Python binding for pdfium, the PDF rendering library used by
|
||||
Google Chrome. The `--rasterizer auto` setting (default) prefers pypdfium2
|
||||
when available.
|
||||
:::
|
||||
|
||||
While it is possible to use a program like Ghostscript or ImageMagick to
|
||||
obtain an image and then run that image through Tesseract OCR, this process
|
||||
actually generates a new PDF, potentially resulting in the loss of various
|
||||
details (such as the document's metadata). In contrast, OCRmyPDF can produce
|
||||
a minimally altered PDF as the output.
|
||||
|
||||
OCRmyPDF also offers several image processing options, such as deskew, which
|
||||
enhances the visual quality of files and the accuracy of OCR. When these
|
||||
options are utilized, the OCR layer is integrated into the processed image.
|
||||
|
||||
By default, OCRmyPDF generates archival PDFs in the PDF/A format, which is
|
||||
a more rigid subset of PDF features designed for long-term archives. If you
|
||||
prefer regular PDFs, you can disable this feature using the
|
||||
`--output-type pdf` option.
|
||||
|
||||
## Why you shouldn't do this manually
|
||||
|
||||
A PDF is similar to an HTML file, in that it contains document structure
|
||||
along with images. While some PDFs may solely display a full-page image,
|
||||
they often contain additional content that would be forfeited if not preserved.
|
||||
|
||||
A manual process could take one of these approaches:
|
||||
|
||||
1. Rasterize each page as an image, perform OCR on the images, and then merge the
|
||||
output into a PDF. This method preserves the layout of each page, but
|
||||
resamples all images potentially leading to quality loss, increased file size,
|
||||
and the introduction of compression artifacts, among other issues.
|
||||
2. Extract each image, OCR, and combine the output into a PDF. This approach
|
||||
loses the context in which images are used in the PDF, potentially resulting
|
||||
in loss of information related to scaling and position of images. Some scanned
|
||||
PDFs contain multiple images segmented into black and white, grayscale
|
||||
and color regions, with stencil masks to prevent overlap, as this can
|
||||
enhance the appearance of a file while reducing file size.
|
||||
Reassembling these images can be challenging, and risks losing vector art
|
||||
or text that is not part of an image.
|
||||
|
||||
In cases where a PDF solely serves as a container for images without any
|
||||
rotation, scaling, or cropping, the second approach can be lossless.
|
||||
|
||||
OCRmyPDF uses various strategies depending on input options and the input PDF
|
||||
itself. Generally, it rasterizes a page for OCR and then integrates the OCR
|
||||
data back into the original PDF. This approach allows it to handle complex
|
||||
PDFs and preserve their content as much as possible.
|
||||
|
||||
Furthermore, OCRmyPDF supports a wide range of edge cases that have emerged
|
||||
during several years of development. It accommodates PDF features like
|
||||
images within Form XObjects and pages with UserUnit scaling. It also
|
||||
supports less common image formats like non-monochrome 1-bit images and
|
||||
provides warnings about files you may not want to OCR. Thanks to tools
|
||||
like pikepdf and QPDF, it can auto-repair damaged PDFs. You don't need to
|
||||
understand the intricacies of these issues; you should be able to use
|
||||
OCRmyPDF with any PDF file, and expect reasonable results.
|
||||
|
||||
## Limitations
|
||||
|
||||
OCRmyPDF is subject to limitations imposed by the Tesseract OCR engine.
|
||||
These limitations are inherent to any software relying on Tesseract:
|
||||
|
||||
- The OCR accuracy may not match that of commercial OCR solutions.
|
||||
- It is incapable of recognizing handwriting.
|
||||
- It may detect gibberish and report it as OCR output.
|
||||
- Results may be subpar when a document contains languages not specified
|
||||
in the `-l LANG` argument.
|
||||
- Tesseract may struggle to analyze the natural reading order of documents.
|
||||
For instance, it might fail to recognize two columns in a document and
|
||||
attempt to join text across columns.
|
||||
- Poor quality scans can result in subpar OCR quality. In other words, the
|
||||
quality of the OCR output depends on the quality of the input.
|
||||
- Tesseract does not provide information about the font family to which text
|
||||
belongs.
|
||||
- Tesseract does not divide text into paragraphs or headings. It only provides
|
||||
the text and its bounding box. As such, the generated PDF does not
|
||||
contain any information about the document's structure.
|
||||
|
||||
### Ghostscript considerations
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
Ghostscript is no longer strictly required. OCRmyPDF can use pypdfium2
|
||||
for rasterization and verapdf for PDF/A validation.
|
||||
:::
|
||||
|
||||
While Ghostscript remains a capable and feature-rich tool with a long history,
|
||||
recent releases have introduced some compatibility challenges that OCRmyPDF
|
||||
v17 addresses through alternative codepaths. When Ghostscript is used:
|
||||
|
||||
- PDFs containing JPEG 2000-encoded content may be converted to JPEG
|
||||
encoding, which may introduce compression artifacts, if Ghostscript
|
||||
PDF/A is enabled.
|
||||
- Ghostscript may transcode grayscale and color images, potentially
|
||||
lossily, based on an internal algorithm. This
|
||||
behavior can be suppressed by setting `--pdfa-image-compression` to
|
||||
`jpeg` or `lossless` to set all images to one type or the other.
|
||||
Ghostscript lacks an option to maintain the input image's format.
|
||||
(Modern Ghostscript can copy JPEG images without transcoding them.)
|
||||
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
||||
one of the standard XMP metadata namespaces for PDFs. In particular,
|
||||
PRISM Metadata is removed.
|
||||
- Ghostscript's PDF/A conversion may remove or deactivate
|
||||
hyperlinks and other active content.
|
||||
|
||||
When pypdfium2 and verapdf are available, many of these limitations can be
|
||||
avoided by using the speculative PDF/A conversion path (enabled by default
|
||||
with `--output-type auto`).
|
||||
|
||||
You can use `--output-type pdf` to disable PDF/A conversion and produce
|
||||
a standard, non-archival PDF.
|
||||
|
||||
Regarding OCRmyPDF itself:
|
||||
|
||||
- PDFs using transparency are not currently represented in the test
|
||||
suite
|
||||
|
||||
## Similar programs
|
||||
|
||||
To the author's knowledge, OCRmyPDF is the most feature-rich and
|
||||
thoroughly tested command line OCR PDF conversion tool. If it does not
|
||||
meet your needs, contributions and suggestions are welcome.
|
||||
|
||||
Ghostscript recently added three "pdfocr" output devices. They work by
|
||||
rasterizing all content and converting all pages to a single colour space.
|
||||
|
||||
## Web front-ends
|
||||
|
||||
The Docker image of OCRmyPDF provides a web service front-end
|
||||
that allows files to submitted over HTTP, and the results can be downloaded.
|
||||
This is an HTTP server intended to demonstrate how OCRmyPDF can be
|
||||
integrated into a web service. It is not intended to be deployed on the
|
||||
public internet and does not provide any security measures.
|
||||
|
||||
In addition, the following third-party integrations are available:
|
||||
|
||||
- [Paperless-ngx](https://docs.paperless-ngx.com/) is a free software
|
||||
document management system that uses OCRmyPDF to perform OCR on
|
||||
uploaded documents.
|
||||
- [Nextcloud OCR](https://github.com/janis91/ocr) is a free software
|
||||
plugin for the Nextcloud private cloud software.
|
||||
|
||||
OCRmyPDF is not designed to be secure against malware-bearing PDFs (see
|
||||
[Using OCRmyPDF online](ocr-service)). Users should ensure they
|
||||
comply with OCRmyPDF's licenses and the licenses of all dependencies. In
|
||||
particular, OCRmyPDF requires Ghostscript, which is licensed under
|
||||
AGPLv3.
|
||||
@@ -1,242 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
============
|
||||
Introduction
|
||||
============
|
||||
|
||||
OCRmyPDF is an application and library that adds text "layers" to images
|
||||
in PDFs, making scanned image PDFs searchable. It uses OCR to guess what text
|
||||
is contained in images. It is written in Python. OCRmyPDF supports plugins
|
||||
that allow customization of its processing steps, and is very tolerant of
|
||||
PDFs that contain scanned images and "born digital" content that needs no
|
||||
text recognition.
|
||||
|
||||
About OCR
|
||||
=========
|
||||
|
||||
`Optical character
|
||||
recognition <https://en.wikipedia.org/wiki/Optical_character_recognition>`__
|
||||
is technology that converts images of typed or handwritten text, such as
|
||||
in a scanned document, to computer text that can be selected, searched and copied.
|
||||
|
||||
OCRmyPDF uses
|
||||
`Tesseract <https://github.com/tesseract-ocr/tesseract>`__, the best
|
||||
available open source OCR engine, to perform OCR.
|
||||
|
||||
.. _raster-vector:
|
||||
|
||||
About PDFs
|
||||
==========
|
||||
|
||||
PDFs are page description files that attempts to preserve a layout
|
||||
exactly. They contain `vector
|
||||
graphics <http://vector-conversions.com/vectorizing/raster_vs_vector.html>`__
|
||||
that can contain raster objects such as scanned images. Because PDFs can
|
||||
contain multiple pages (unlike many image formats) and can contain fonts
|
||||
and text, it is a good format for exchanging scanned documents.
|
||||
|
||||
|image|
|
||||
|
||||
A PDF page might contain multiple images, even if it only appears to
|
||||
have one image. Some scanners or scanning software will segment pages
|
||||
into monochromatic text and color regions for example, to improve the
|
||||
compression ratio and appearance of the page.
|
||||
|
||||
Rasterizing a PDF is the process of generating corresponding raster images.
|
||||
OCR engines like Tesseract work with images, not scalable vector graphics
|
||||
or mixed raster-vector-text graphics such as PDF.
|
||||
|
||||
About PDF/A
|
||||
===========
|
||||
|
||||
`PDF/A <https://en.wikipedia.org/wiki/PDF/A>`__ is an ISO-standardized
|
||||
subset of the full PDF specification that is designed for archiving (the
|
||||
'A' stands for Archive). PDF/A differs from PDF primarily by omitting
|
||||
features that would make it difficult to read the file in the future,
|
||||
such as embedded Javascript, video, audio and references to external
|
||||
fonts. All fonts and resources needed to interpret the PDF must be
|
||||
contained within it. Because PDF/A disables Javascript and other types
|
||||
of embedded content, it is probably more secure.
|
||||
|
||||
There are various conformance levels and versions, such as "PDF/A-2b".
|
||||
|
||||
Generally speaking, the best format for scanned documents is PDF/A. Some
|
||||
governments and jurisdictions, US Courts in particular, `mandate the use
|
||||
of PDF/A <https://pdfblog.com/2012/02/13/what-is-pdfa/>`__ for scanned
|
||||
documents.
|
||||
|
||||
Since most people who scan documents are interested in reading them
|
||||
indefinitely into the future, OCRmyPDF generates PDF/A-2b by default.
|
||||
|
||||
PDF/A has a few drawbacks. Some PDF viewers include an alert that the
|
||||
file is a PDF/A, which may confuse some users. It also tends to produce
|
||||
larger files than PDF, because it embeds certain resources even if they
|
||||
are commonly available. PDF/A files can be digitally signed, but may not
|
||||
be encrypted, to ensure they can be read in the future. Fortunately,
|
||||
converting from PDF/A to a regular PDF is trivial, and any PDF viewer
|
||||
can view PDF/A.
|
||||
|
||||
What OCRmyPDF does
|
||||
==================
|
||||
|
||||
OCRmyPDF analyzes each page of a PDF to determine the colorspace and
|
||||
resolution (DPI) needed to capture all of the information on that page
|
||||
without losing content. It uses
|
||||
`Ghostscript <http://ghostscript.com/>`__ to rasterize the page, and
|
||||
then performs OCR on the rasterized image to create an OCR "layer".
|
||||
The layer is then grafted back onto the original PDF.
|
||||
|
||||
While one can use a program like Ghostscript or ImageMagick to get an
|
||||
image and put the image through Tesseract, that actually creates a new
|
||||
PDF and many details may be lost. OCRmyPDF can produce a minimally
|
||||
changed PDF as output.
|
||||
|
||||
OCRmyPDF also provides some image processing options, like deskew, which
|
||||
improves the appearance of files and quality of OCR. When these are used,
|
||||
the OCR layer is grafted onto the processed image instead.
|
||||
|
||||
By default, OCRmyPDF produces archival PDFs – PDF/A, which are a
|
||||
stricter subset of PDF features designed for long term archives. If
|
||||
regular PDFs are desired, this can be disabled with
|
||||
``--output-type pdf``.
|
||||
|
||||
Why you shouldn't do this manually
|
||||
==================================
|
||||
|
||||
A PDF is similar to an HTML file, in that it contains document structure
|
||||
along with images. Sometimes a PDF does nothing more than present a full
|
||||
page image, but often there is additional content that would be lost.
|
||||
|
||||
A manual process could work like either of these:
|
||||
|
||||
1. Rasterize each page as an image, OCR the images, and combine the
|
||||
output into a PDF. This preserves the layout of each page, but
|
||||
resamples all images (possibly losing quality, increasing file size,
|
||||
introducing compression artifacts, etc.).
|
||||
2. Extract each image, OCR, and combine the output into a PDF. This
|
||||
loses the context in which images are used in the PDF, meaning that
|
||||
cropping, rotation and scaling of pages may be lost. Some scanned
|
||||
PDFs use multiple images segmented into black and white, grayscale
|
||||
and color regions, with stencil masks to prevent overlap, as this can
|
||||
enhance the appearance of a file while reducing file size. Clearly,
|
||||
reassembling these images will be easy. This also loses and text or
|
||||
vector art on any pages in a PDF with both scanned and pure digital
|
||||
content.
|
||||
|
||||
In the case of a PDF that is nothing other than a container of images
|
||||
(no rotation, scaling, cropping, one image per page), the second
|
||||
approach can be lossless.
|
||||
|
||||
OCRmyPDF uses several strategies depending on input options and the
|
||||
input PDF itself, but generally speaking it rasterizes a page for OCR
|
||||
and then grafts the OCR back onto the original. As such it can handle
|
||||
complex PDFs and still preserve their contents as much as possible.
|
||||
|
||||
OCRmyPDF also supports a many, many edge cases that have cropped over
|
||||
several years of development. We support PDF features like images inside
|
||||
of Form XObjects, and pages with UserUnit scaling. We support rare image
|
||||
formats like non-monochrome 1-bit images. We warn about files you may
|
||||
not to OCR. Thanks to pikepdf and QPDF, we auto-repair PDFs that are
|
||||
damaged. (Not that you need to know what any of these are! You should be
|
||||
able to throw any PDF at it.)
|
||||
|
||||
Limitations
|
||||
===========
|
||||
|
||||
OCRmyPDF is limited by the Tesseract OCR engine. As such it experiences
|
||||
these limitations, as do any other programs that rely on Tesseract:
|
||||
|
||||
- The OCR is not as accurate as commercial OCR solutions.
|
||||
- It is not capable of recognizing handwriting.
|
||||
- It may find gibberish and report this as OCR output.
|
||||
- If a document contains languages outside of those given in the
|
||||
``-l LANG`` arguments, results may be poor.
|
||||
- It is not always good at analyzing the natural reading order of
|
||||
documents. For example, it may fail to recognize that a document
|
||||
contains two columns, and may try to join text across columns.
|
||||
- Poor quality scans may produce poor quality OCR. Garbage in, garbage
|
||||
out.
|
||||
- It does not expose information about what font family text belongs
|
||||
to.
|
||||
|
||||
OCRmyPDF is also limited by the PDF specification:
|
||||
|
||||
- PDF encodes the position of text glyphs but does not encode document
|
||||
structure. There is no markup that divides a document in sections,
|
||||
paragraphs, sentences, or even words (since blank spaces are not
|
||||
represented). As such all elements of document structure including
|
||||
the spaces between words must be derived heuristically. Some PDF
|
||||
viewers do a better job of this than others.
|
||||
- Because some popular open source PDF viewers have a particularly hard
|
||||
time with spaces between words, OCRmyPDF appends a space to each text
|
||||
element as a workaround (when using ``--pdf-renderer hocr``). While
|
||||
this mixes document structure with graphical information that ideally
|
||||
should be left to the PDF viewer to interpret, it improves
|
||||
compatibility with some viewers and does not cause problems for
|
||||
better ones.
|
||||
|
||||
Ghostscript also imposes some limitations:
|
||||
|
||||
- PDFs containing JBIG2-encoded content will be converted to CCITT
|
||||
Group4 encoding, which has lower compression ratios, if Ghostscript
|
||||
PDF/A is enabled.
|
||||
- PDFs containing JPEG 2000-encoded content will be converted to JPEG
|
||||
encoding, which may introduce compression artifacts, if Ghostscript
|
||||
PDF/A is enabled.
|
||||
- Ghostscript may transcode grayscale and color images, either lossy to
|
||||
lossless or lossless to lossy, based on an internal algorithm. This
|
||||
behavior can be suppressed by setting ``--pdfa-image-compression`` to
|
||||
``jpeg`` or ``lossless`` to set all images to one type or the other.
|
||||
Ghostscript has no option to maintain the input image's format.
|
||||
(Modern Ghostscript can copy JPEG images without transcoding them.)
|
||||
- Ghostscript's PDF/A conversion removes any XMP metadata that is not
|
||||
one of the standard XMP metadata namespaces for PDFs. In particular,
|
||||
PRISM Metdata is removed.
|
||||
- Ghostscript's PDF/A conversion seems to remove or deactivate
|
||||
hyperlinks and other active content.
|
||||
|
||||
You can use ``--output-type pdf`` to disable PDF/A conversion and produce
|
||||
a standard, non-archival PDF.
|
||||
|
||||
Regarding OCRmyPDF itself:
|
||||
|
||||
- PDFs that use transparency are not currently represented in the test
|
||||
suite
|
||||
|
||||
Similar programs
|
||||
================
|
||||
|
||||
To the author's knowledge, OCRmyPDF is the most feature-rich and
|
||||
thoroughly tested command line OCR PDF conversion tool. If it does not
|
||||
meet your needs, contributions and suggestions are welcome. If not,
|
||||
consider one of these similar open source programs:
|
||||
|
||||
- pdf2pdfocr
|
||||
- pdfsandwich
|
||||
|
||||
Ghostscript recently added three "pdfocr" output devices. They work by
|
||||
rasterizing all content and converting all pages to a single colour space.
|
||||
|
||||
Web front-ends
|
||||
==============
|
||||
|
||||
The Docker image ``ocrmypdf`` provides a web service front-end
|
||||
that allows files to submitted over HTTP and the results "downloaded".
|
||||
This is an HTTP server intended to simplify web services deployments; it
|
||||
is not intended to be deployed on the public internet and no real
|
||||
security measures to speak of.
|
||||
|
||||
In addition, the following third-party integrations are available:
|
||||
|
||||
- `Nextcloud OCR <https://github.com/janis91/ocr>`__ is a free software
|
||||
plugin for the Nextcloud private cloud software
|
||||
|
||||
OCRmyPDF is not designed to be secure against malware-bearing PDFs (see
|
||||
`Using OCRmyPDF online <ocr-service>`__). Users should ensure they
|
||||
comply with OCRmyPDF's licenses and the licenses of all dependencies. In
|
||||
particular, OCRmyPDF requires Ghostscript, which is licensed under
|
||||
AGPLv3.
|
||||
|
||||
.. |image| image:: images/bitmap_vs_svg.svg
|
||||
@@ -0,0 +1,63 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
{#jbig2}
|
||||
|
||||
# Installing the JBIG2 encoder
|
||||
|
||||
Most Linux distributions do not include a JBIG2 encoder since JBIG2
|
||||
encoding was patented for a long time. All known JBIG2 US patents have
|
||||
expired as of 2017, but it is possible that unknown patents exist.
|
||||
|
||||
JBIG2 encoding is recommended for OCRmyPDF and is used to losslessly
|
||||
create smaller PDFs. If JBIG2 encoding is not available, lower quality
|
||||
CCITT encoding will be used for monochrome images.
|
||||
|
||||
JBIG2 decoding is not patented and is performed automatically by most
|
||||
PDF viewers. It is widely supported and has been part of the PDF
|
||||
specification since 2001.
|
||||
|
||||
JBIG encoding is automatically provided by these OCRmyPDF packages: -
|
||||
Docker image (both Ubuntu and Alpine) - Snap package - ArchLinux AUR
|
||||
package - Alpine Linux package - Homebrew on macOS
|
||||
|
||||
For all other platforms, you would need to build the JBIG2 encoder from
|
||||
source:
|
||||
|
||||
:::{code} bash
|
||||
git clone https://github.com/agl/jbig2enc
|
||||
cd jbig2enc
|
||||
./autogen.sh
|
||||
./configure && make
|
||||
[sudo] make install
|
||||
:::
|
||||
|
||||
Dependencies include libtoolize and libleptonica, which on Ubuntu
|
||||
systems are packaged as libtool and libleptonica-dev. On Fedora (35)
|
||||
they are packaged as libtool and leptonica-devel. For this to work,
|
||||
please make sure to install `autotools`, `automake`, `libtool`, `pkg-config`
|
||||
and `leptonica` first if not already installed. Other dependencies might
|
||||
be required depending on your system.
|
||||
|
||||
:::{code} bash
|
||||
[sudo] apt install autotools-dev automake libtool libleptonica-dev pkg-config
|
||||
:::
|
||||
|
||||
## JBIG2 Compression
|
||||
|
||||
OCRmyPDF uses JBIG2 lossless compression for bitonal (black and white)
|
||||
images. This provides excellent compression ratios compared to the older
|
||||
CCITT G4 standard, while preserving the exact pixel content of the
|
||||
original image.
|
||||
|
||||
You can adjust the threshold for JBIG2 compression with
|
||||
`--jbig2-threshold`. The default is 0.85.
|
||||
|
||||
:::{note}
|
||||
Previous versions of OCRmyPDF supported a lossy JBIG2 mode
|
||||
(`--jbig2-lossy`). This feature has been removed due to the well-known
|
||||
risk of character substitution errors (e.g., 6/8 confusion). See
|
||||
[JBIG2 disadvantages](https://en.wikipedia.org/wiki/JBIG2#Disadvantages)
|
||||
for more information on why lossy JBIG2 is problematic. The `--jbig2-lossy`
|
||||
and `--jbig2-page-group-size` arguments are now ignored with a warning.
|
||||
:::
|
||||
@@ -1,63 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
.. _jbig2:
|
||||
|
||||
============================
|
||||
Installing the JBIG2 encoder
|
||||
============================
|
||||
|
||||
Most Linux distributions do not include a JBIG2 encoder since JBIG2
|
||||
encoding was patented for a long time. All known JBIG2 US patents have
|
||||
expired as of 2017, but it is possible that unknown patents exist.
|
||||
|
||||
JBIG2 encoding is recommended for OCRmyPDF and is used to losslessly
|
||||
create smaller PDFs. If JBIG2 encoding is not available, lower quality
|
||||
encodings will be used.
|
||||
|
||||
JBIG2 decoding is not patented and is performed automatically by most
|
||||
PDF viewers. It is widely supported and has been part of the PDF
|
||||
specification since 2001.
|
||||
|
||||
On macOS, Homebrew packages jbig2enc and OCRmyPDF includes it by
|
||||
default. The Docker image for OCRmyPDF also builds its own JBIG2 encoder
|
||||
from source.
|
||||
|
||||
For all other Linux, you must build a JBIG2 encoder from source:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
git clone https://github.com/agl/jbig2enc
|
||||
cd jbig2enc
|
||||
./autogen.sh
|
||||
./configure && make
|
||||
[sudo] make install
|
||||
|
||||
.. _jbig2-lossy:
|
||||
|
||||
Dependencies include libtoolize and libleptonica, which on Ubuntu systems
|
||||
are packaged as libtool and libleptonica-dev. On Fedora (35) they are packaged
|
||||
as libtool and leptonica-devel.
|
||||
|
||||
Lossy mode JBIG2
|
||||
================
|
||||
|
||||
OCRmyPDF provides lossy mode JBIG2 as an advanced feature. Users should
|
||||
`review the technical concerns with JBIG2 in lossy
|
||||
mode <https://en.wikipedia.org/wiki/JBIG2#Disadvantages>`__
|
||||
and decide if this feature is acceptable for their use case.
|
||||
|
||||
JBIG2 lossy mode does achieve higher compression ratios than any other
|
||||
monochrome (bitonal) compression technology; for large text documents
|
||||
the savings are considerable. JBIG2 lossless still gives great
|
||||
compression ratios and is a major improvement over the older CCITT G4
|
||||
standard. As explained above, there is some risk of substitution errors.
|
||||
|
||||
To turn on JBIG2 lossy mode, add the argument ``--jbig2-lossy``.
|
||||
``--optimize {1,2,3}`` are necessary for the argument to take effect
|
||||
also required. Also, a JBIG2 encoder must be installed as described in
|
||||
the previous section.
|
||||
|
||||
*Due to an oversight, ocrmypdf v7.0 and v7.1 used lossy mode by
|
||||
default.*
|
||||
@@ -0,0 +1,129 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
(lang-packs)=
|
||||
|
||||
# Installing additional language packs
|
||||
|
||||
OCRmyPDF uses Tesseract for OCR, and relies on its language packs for all languages.
|
||||
On most platforms, English is installed with Tesseract by default, but not always.
|
||||
|
||||
Tesseract supports [most
|
||||
languages](https://github.com/tesseract-ocr/tesseract/blob/main/doc/tesseract.1.asc#languages).
|
||||
Languages are identified by standardized three-letter codes (called ISO 639-2 Alpha-3).
|
||||
Tesseract's documentation also lists the three-letter code for your language.
|
||||
Some are anglicized, e.g. Spanish is `spa` rather than `esp`, while others
|
||||
are not, e.g. German is `deu` and French is `fra`.
|
||||
|
||||
Language packs (strictly speaking, Tesseract "traineddata" files) generally correspond
|
||||
to the language in question, but different language packs are used in certain
|
||||
situations. For German, the "Fraktur" language pack can assist with reading older
|
||||
materials in the Fraktur typeface family (`deu_frak`). Some communities have changed
|
||||
their script from Cyrillic to Latin; the Cyrillic version of Uzbek is available
|
||||
as `uzb_cyrl` and the Latin version is `uzb`.
|
||||
|
||||
After you have installed a language pack, you can use it with `ocrmypdf -l <language>`,
|
||||
for example `ocrmypdf -l spa`. For multilingual documents, you can specify
|
||||
all languages to be expected, e.g. `ocrmypdf -l eng+fra` for English and French.
|
||||
English is assumed by default unless other language(s) are specified.
|
||||
|
||||
For Linux users, you can often find packages that provide language
|
||||
packs.
|
||||
|
||||
## Platform install steps
|
||||
|
||||
### Debian and Ubuntu (apt)
|
||||
|
||||
```bash
|
||||
# Display a list of all Tesseract language packs
|
||||
apt-cache search tesseract-ocr
|
||||
|
||||
# Install Chinese Simplified language pack
|
||||
apt-get install tesseract-ocr-chi-sim
|
||||
```
|
||||
|
||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as
|
||||
to what languages it should search for. Multiple languages can be
|
||||
requested using either `-l eng+fra` (English and French) or
|
||||
`-l eng -l fra`.
|
||||
|
||||
### Fedora
|
||||
|
||||
```bash
|
||||
# Display a list of all Tesseract language packs
|
||||
dnf search tesseract
|
||||
|
||||
# Install Chinese Simplified language pack
|
||||
dnf install tesseract-langpack-chi_sim
|
||||
```
|
||||
|
||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as
|
||||
to what languages it should search for. Multiple languages can be
|
||||
requested using either `-l eng+fra` (English and French) or
|
||||
`-l eng -l fra`.
|
||||
|
||||
### Arch Linux
|
||||
|
||||
```bash
|
||||
# Display a list of all Tesseract language packs
|
||||
pacman -Ss tesseract-data
|
||||
|
||||
# Install German language pack
|
||||
pacman -S tesseract-data-deu
|
||||
```
|
||||
|
||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as
|
||||
to what languages it should search for. Multiple languages can be
|
||||
requested using either `-l eng+fra` (English and French) or
|
||||
`-l eng -l fra`.
|
||||
|
||||
### Gentoo
|
||||
|
||||
On Gentoo the package `app-text/tessdata_fast`, which `app-text/tesseract` depends on, handles Tesseract languages.
|
||||
It accepts USE flags to select what languages should be installed, these can be set in `/etc/portage/package.use`.
|
||||
Alternatively one can globally set the [L10N use extension](https://wiki.gentoo.org/wiki/Localization/Guide#L10N) in `/etc/portage/make.conf`.
|
||||
This enables these languages for all packages (e.g. including aspell).
|
||||
|
||||
```bash
|
||||
# Display a list of all Tesseract language packs
|
||||
equery uses app-text/tessdata_fast
|
||||
|
||||
# Add English and German language support for Tesseract only
|
||||
echo 'app-text/tessdata_fast l10n_de l10n_en' >> /etc/portage/package.use
|
||||
|
||||
# Add global English and German language support (the `l10n_` from equery has to be omitted)
|
||||
echo L10N="de en" >> /etc/portage/make.conf
|
||||
|
||||
# update system to reflect changed USE flags
|
||||
emerge --update --deep --newuse @world
|
||||
```
|
||||
|
||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as
|
||||
to what languages it should search for. Multiple languages can be
|
||||
requested using either `-l eng+fra` (English and French) or
|
||||
`-l eng -l fra`.
|
||||
|
||||
### macOS
|
||||
|
||||
You can install additional language packs by
|
||||
{ref}`installing Tesseract using Homebrew with all language packs <macos-all-languages>`.
|
||||
|
||||
### Docker
|
||||
|
||||
Users of the OCRmyPDF Docker image should install language packs into a
|
||||
derived Docker image as
|
||||
{ref}`described in that section <docker-lang-packs>`.
|
||||
|
||||
### Windows
|
||||
|
||||
The Tesseract installer provided by Chocolatey currently includes only English language.
|
||||
To install other languages, download the respective language pack (`.traineddata` file)
|
||||
from <https://github.com/tesseract-ocr/tessdata/> and place it in
|
||||
`C:\\Program Files\\Tesseract-OCR\\tessdata` (or wherever Tesseract OCR is installed).
|
||||
|
||||
## Custom language packs
|
||||
|
||||
If you have fine-tuned or trained Tesseract and generated custom trained data, you can
|
||||
copy your `customlang.traineddata` file into your Tesseract "tessdata" folder, and
|
||||
then use the `-l customlang` argument to tell OCRmyPDF to pass that language on to
|
||||
Tesseract.
|
||||
@@ -1,107 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
.. _lang-packs:
|
||||
|
||||
====================================
|
||||
Installing additional language packs
|
||||
====================================
|
||||
|
||||
OCRmyPDF uses Tesseract for OCR, and relies on its language packs for all languages.
|
||||
On most platforms, English is installed with Tesseract by default, but not always.
|
||||
|
||||
Tesseract supports `most
|
||||
languages <https://github.com/tesseract-ocr/tesseract/blob/master/doc/tesseract.1.asc#languages>`__.
|
||||
Languages are identified by standardized three-letter codes (called ISO 639-2 Alpha-3).
|
||||
Tesseract's documentation also lists the three-letter code for your language.
|
||||
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
||||
are not, e.g. German is ``deu`` and French is ``fra``.
|
||||
|
||||
After you have installed a language pack, you can use it with ``ocrmypdf -l <language>``,
|
||||
for example ``ocrmypdf -l spa``. For multilingual documents, you can specify
|
||||
all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French.
|
||||
English is assumed by default unless other language(s) are specified.
|
||||
|
||||
For Linux users, you can often find packages that provide language
|
||||
packs:
|
||||
|
||||
Debian and Ubuntu users
|
||||
=======================
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# Display a list of all Tesseract language packs
|
||||
apt-cache search tesseract-ocr
|
||||
|
||||
# Install Chinese Simplified language pack
|
||||
apt-get install tesseract-ocr-chi-sim
|
||||
|
||||
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||
to what languages it should search for. Multiple languages can be
|
||||
requested using either ``-l eng+fra`` (English and French) or
|
||||
``-l eng -l fra``.
|
||||
|
||||
Fedora users
|
||||
============
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# Display a list of all Tesseract language packs
|
||||
dnf search tesseract
|
||||
|
||||
# Install Chinese Simplified language pack
|
||||
dnf install tesseract-langpack-chi_sim
|
||||
|
||||
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||
to what languages it should search for. Multiple languages can be
|
||||
requested using either ``-l eng+fra`` (English and French) or
|
||||
``-l eng -l fra``.
|
||||
|
||||
Gentoo users
|
||||
============
|
||||
|
||||
On Gentoo the package ``app-text/tessdata_fast``, which ``app-text/tesseract`` depends on, handles Tesseract languages.
|
||||
It accepts USE flags to select what languages should be installed, these can be set in ``/etc/portage/package.use``.
|
||||
Alternatively one can globally set the `L10N use extension <https://wiki.gentoo.org/wiki/Localization/Guide#L10N>`__ in ``/etc/portage/make.conf``.
|
||||
This enables these languages for all packages (e.g. including aspell).
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
# Display a list of all Tesseract language packs
|
||||
equery uses app-text/tessdata_fast
|
||||
|
||||
# Add English and German language support for Tesseract only
|
||||
echo 'app-text/tessdata_fast l10n_de l10n_en' >> /etc/portage/package.use
|
||||
|
||||
# Add global English and German language support (the `l10n_` from equery has to be omited)
|
||||
echo L10N="de en" >> /etc/portage/make.conf
|
||||
|
||||
# update system to reflect changed USE flags
|
||||
emerge --update --deep --newuse @world
|
||||
|
||||
You can then pass the ``-l LANG`` argument to OCRmyPDF to give a hint as
|
||||
to what languages it should search for. Multiple languages can be
|
||||
requested using either ``-l eng+fra`` (English and French) or
|
||||
``-l eng -l fra``.
|
||||
|
||||
macOS users
|
||||
===========
|
||||
|
||||
You can install additional language packs by
|
||||
:ref:`installing Tesseract using Homebrew with all language packs <macos-all-languages>`.
|
||||
|
||||
Docker users
|
||||
============
|
||||
|
||||
Users of the OCRmyPDF Docker image should install language packs into a
|
||||
derived Docker image as
|
||||
:ref:`described in that section <docker-lang-packs>`.
|
||||
|
||||
Windows users
|
||||
=============
|
||||
|
||||
The Tesseract installer provided by Chocolatey currently includes only English language.
|
||||
To install other languages, download the respective language pack (``.traineddata`` file)
|
||||
from https://github.com/tesseract-ocr/tessdata/ and place it in
|
||||
``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed).
|
||||
@@ -0,0 +1,179 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Maintainer notes
|
||||
|
||||
This is for those who package OCRmyPDF for downstream use. (Thank you
|
||||
for your hard work.)
|
||||
|
||||
## Known ports/packagers
|
||||
|
||||
OCRmyPDF has been ported to many platforms already. If you are
|
||||
interesting in porting to a new platform, check with
|
||||
[Repology](https://repology.org/projects/?search=ocrmypdf) to see the
|
||||
status of that platform.
|
||||
|
||||
### Make sure you can package pikepdf
|
||||
|
||||
pikepdf, created by the same author, is a mixed Python and C++14 package
|
||||
with much stiffer build requirements. If you want to use OCRmyPDF on
|
||||
some novel platform or distribution, first make sure you can package
|
||||
pikepdf.
|
||||
|
||||
### Core dependencies
|
||||
|
||||
:::{versionchanged} 17.0.0
|
||||
Ghostscript is no longer strictly required. OCRmyPDF now supports alternative
|
||||
codepaths for both PDF rasterization and PDF/A conversion.
|
||||
:::
|
||||
|
||||
OCRmyPDF has the following runtime dependencies:
|
||||
|
||||
**For PDF rasterization** (converting PDF pages to images for OCR):
|
||||
|
||||
- `pypdfium2` (Python package) - OR -
|
||||
- `ghostscript` (system binary)
|
||||
- Recommendation: Install both for best compatibility
|
||||
|
||||
**For PDF/A conversion**:
|
||||
|
||||
- `verapdf` (system binary) with pikepdf's speculative conversion - OR -
|
||||
- `ghostscript` (system binary)
|
||||
- Recommendation: Install both for best compatibility
|
||||
|
||||
**For OCR**:
|
||||
- `tesseract-ocr` (system binary) - Required for MVP
|
||||
|
||||
**For text rendering** (expressing OCR results in PDF):
|
||||
- `fpdf2` (Python package) - Required for text layer rendering
|
||||
- `uharfbuzz` (Python package) - Required for text layer rendering
|
||||
- `font-noto` (system package) - Recommended for text layer rendering
|
||||
|
||||
**Other dependencies**:
|
||||
- `unpaper` (system binary) - Optional, enables `--clean` and `--clean-final`
|
||||
- `pngquant` (system binary) - Optional, enables `--optimize 2` and `--optimize 3`
|
||||
- `jbig2enc` (system binary) - Optional, improves compression of monochrome images
|
||||
|
||||
While Ghostscript remains a capable and feature-rich tool with a long history,
|
||||
recent releases have introduced some compatibility challenges that OCRmyPDF v17
|
||||
addresses through alternative codepaths. For the best user experience, packagers
|
||||
should install both Ghostscript and the alternative tools (pypdfium2, verapdf)
|
||||
when available.
|
||||
|
||||
On Windows, OCRmyPDF will also check the registry for Tesseract and Ghostscript
|
||||
locations.
|
||||
|
||||
Tesseract OCR relies on SIMD for performance and only has proper support
|
||||
for this on ARM and x86\_64. Performance may be poor on other processor
|
||||
architectures.
|
||||
|
||||
### Versioning scheme
|
||||
|
||||
OCRmyPDF uses hatch-vcs for versioning, which derives the version from
|
||||
Git as a single source of truth. This may be unsuitable for some
|
||||
distributions, e.g. to indicate that your distribution modifies OCRmyPDF
|
||||
in some way.
|
||||
|
||||
You can patch the `__version__` variable in `src/ocrmypdf/_version.py`
|
||||
if necessary, or set the environment variable
|
||||
`SETUPTOOLS_SCM_PRETEND_VERSION` to the required version, if you need to
|
||||
override versioning for some reason.
|
||||
|
||||
### jbig2enc
|
||||
|
||||
OCRmyPDF will use jbig2enc, a JBIG2 encoder, if one can be found. Some
|
||||
distributions have shied away from packaging JBIG2 because it contains
|
||||
patented algorithms, but all patents have expired since 2017. If
|
||||
possible, consider packaging it too to improve OCRmyPDF's compression.
|
||||
|
||||
:::{note}
|
||||
Lossy JBIG2 encoding has been removed in v17.0.0 due to well-documented
|
||||
risks of character substitution errors. Previously we provided this feature
|
||||
on a "caveat emptor" basis but in the interest of focusing and eliminating
|
||||
risks, we decided to remove this option. Now, only lossless JBIG2 compression
|
||||
is supported.
|
||||
:::
|
||||
|
||||
### Dependency matrix for packagers
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
The following table summarizes the dependency options introduced in v17.0.0:
|
||||
|
||||
| Feature | Option 1 | Option 2 | Notes |
|
||||
|---------|----------|----------|-------|
|
||||
| PDF rasterization | pypdfium2 (Python) | ghostscript (binary) | pypdfium2 preferred when available |
|
||||
| PDF/A conversion | verapdf + pikepdf | ghostscript | verapdf validates speculative conversion |
|
||||
| Text rendering | fpdf2 (Python) | - | Required, replaces legacy hOCR renderer |
|
||||
| OCR | tesseract-ocr | `--ocr-engine none` | Can be skipped entirely |
|
||||
|
||||
**Minimum viable installation:**
|
||||
|
||||
- tesseract-ocr + (pypdfium2 OR ghostscript) + fpdf2
|
||||
|
||||
**Recommended installation:**
|
||||
|
||||
- tesseract-ocr + pypdfium2 + ghostscript + verapdf + fpdf2 + unpaper + pngquant + jbig2enc
|
||||
|
||||
:::{warning}
|
||||
If Ghostscript is not installed and verapdf is not available, PDF/A output
|
||||
cannot be produced. The output will be a standard PDF instead. This is a
|
||||
breaking change for rare configurations that previously relied on PDF/A
|
||||
output without Ghostscript alternatives.
|
||||
:::
|
||||
|
||||
**Sample debian/control dependency specification**
|
||||
|
||||
```
|
||||
Depends:
|
||||
fonts-noto,
|
||||
fpdf2 (>= 2.8),
|
||||
ghostscript (>= 9.55), # Not strictly required, but best user experience
|
||||
icc-profiles-free,
|
||||
img2pdf,
|
||||
python3-coloredlogs,
|
||||
python3-deprecation,
|
||||
python3-pdfminer (>= 20181108+dfsg-3),
|
||||
python3-pikepdf (>= 8.14.0),
|
||||
python3-pil,
|
||||
python3-pluggy,
|
||||
python3-reportlab,
|
||||
python3-rich,
|
||||
python3-uharfbuzz, # Not currently in Debian
|
||||
tesseract-ocr (>= 5.0.0),
|
||||
zlib1g,
|
||||
${misc:Depends},
|
||||
${python3:Depends},
|
||||
Recommends:
|
||||
cyclopts, # Not currently in Debian
|
||||
jbig2
|
||||
paddleocr, # Not currently in Debian
|
||||
pngquant,
|
||||
pypdfium2, # Not currently in Debian
|
||||
unpaper,
|
||||
verapdf, # Not currently in Debian
|
||||
Suggests:
|
||||
ocrmypdf-doc,
|
||||
python-watchdog,
|
||||
```
|
||||
|
||||
### Command line completions
|
||||
|
||||
Please ensure that command line completions are installed, as described
|
||||
in the installation documentation.
|
||||
|
||||
### 32-bit Linux support
|
||||
|
||||
If you maintain a Linux distribution that supports 32-bit x86 or ARM,
|
||||
OCRmyPDF should continue to work as long as all of its dependencies
|
||||
continue to be available in 32-bit form. Please note we do not test on
|
||||
32-bit platforms.
|
||||
|
||||
### HEIF/HEIC
|
||||
|
||||
OCRmyPDF defaults to installing the pi-heif PyPI package, which supports
|
||||
converting HEIF (High Efficiency Image File Format) images to PDF from
|
||||
the command line. If your distribution does not have this library
|
||||
available, you can exclude it and OCRmyPDF will gracefully degrade
|
||||
automatically, losing only support for this feature.
|
||||
@@ -1,64 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
================
|
||||
Maintainer notes
|
||||
================
|
||||
|
||||
This is for those who package OCRmyPDF for downstream use. (Thank you
|
||||
for your hard work.)
|
||||
|
||||
Known ports/packagers
|
||||
=====================
|
||||
|
||||
OCRmyPDF has been ported to many platforms already. If you are interesting in
|
||||
porting to a new platform, check with
|
||||
`Repology <https://repology.org/projects/?search=ocrmypdf>`__ to see the status
|
||||
of that platform.
|
||||
|
||||
Make sure you can package pikepdf
|
||||
---------------------------------
|
||||
|
||||
pikepdf, created by the same author, is a mixed Python and C++14 package with
|
||||
much stiffer build requirements. If you want to use OCRmyPDF on some novel platform
|
||||
or distribution, first make sure you can package pikepdf.
|
||||
|
||||
Non-Python dependencies
|
||||
-----------------------
|
||||
|
||||
Note that we have non-Python dependencies. In particular, OCRmyPDF requires
|
||||
Ghostscript and Tesseract OCR to be installed and needs to be able to locate their
|
||||
binaries on the system PATH. On Windows, OCRmyPDF will also check the registry
|
||||
for their locations.
|
||||
|
||||
Tesseract OCR relies on SIMD for performance and only has proper support for this
|
||||
on ARM and x86_64. Performance may be poor on other processor architectures.
|
||||
|
||||
Versioning scheme
|
||||
-----------------
|
||||
|
||||
OCRmyPDF uses setuptools-scm for versioning, which derives the version from
|
||||
Git as a single source of truth. This may be unsuitable for some distributions, e.g.
|
||||
to indicate that your distribution modifies OCRmyPDF in some way.
|
||||
|
||||
You can patch the ``__version__`` variable in ``src/ocrmypdf/_version.py`` if
|
||||
necessary.
|
||||
|
||||
OCRmyPDF uses setuptools-scm-git-archive to ensure that tarballs downloaded from
|
||||
GitHub contain version information. Unfortunately, these tarballs are not always
|
||||
deterministic. See this
|
||||
`issue <https://github.com/ocrmypdf/OCRmyPDF/issues/841#issuecomment-936562696>`_.
|
||||
|
||||
jbig2enc
|
||||
--------
|
||||
|
||||
OCRmyPDF will use jbig2enc, a JBIG2 encoder, if one can be found. Some distributions
|
||||
have shied away from packaging JBIG2 because it contains patented algorithms, but
|
||||
all patents have expired since 2017. If possible, consider packaging it too to
|
||||
improve OCRmyPDF's compression.
|
||||
|
||||
Command line completions
|
||||
------------------------
|
||||
|
||||
Please ensure that command line completions are installed.
|
||||
@@ -0,0 +1,104 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# PDF optimization
|
||||
|
||||
OCRmyPDF includes an image-oriented PDF optimizer. By default, the
|
||||
optimizer runs with safe settings with the goal of improving compression
|
||||
at no loss of quality. At higher optimization levels, lossy
|
||||
optimizations may be applied and tuned. Optimization occurs after OCR,
|
||||
and only if OCR succeeded. It does not perform other possible
|
||||
optimizations such as deduplicating resources, consolidating fonts,
|
||||
simplifying vector drawings, or anything of that nature.
|
||||
|
||||
:::{list-table} OCRmyPDF optimization settings
|
||||
---
|
||||
widths: 33 6 60
|
||||
header-rows: 1
|
||||
---
|
||||
|
||||
* - Optimization level
|
||||
- Shorthand
|
||||
- Description
|
||||
* - ``--optimize 0``
|
||||
- ``-O0``
|
||||
- Disable most optimizations.
|
||||
* - ``--optimize 1`` (default)
|
||||
- ``-O1``
|
||||
- Enables lossless optimizations, such as transcoding images to more
|
||||
efficient formats. Also compress other uncompressed objects in the
|
||||
PDF and enables the more efficient "object streams" within the PDF.
|
||||
* - ``--optimize 2``
|
||||
- ``-O2``
|
||||
- All of the above, and enables lossy optimizations and color quantization.
|
||||
* - ``--optimize 3``
|
||||
- ``-O3``
|
||||
- All of the above, and enables more aggressive optimizations and targets lower
|
||||
image quality.
|
||||
:::
|
||||
|
||||
The exact type of optimizations performed will vary over time, and
|
||||
depend on what third party tools are installed.
|
||||
|
||||
Despite optimizations, OCRmyPDF might still increase the overall file
|
||||
size, since it must embed information about the recognized text, and
|
||||
depending on the settings chosen, may not be able to represent the
|
||||
output file as compactly as the input file.
|
||||
|
||||
## Optimizations that always occurs
|
||||
|
||||
OCRmyPDF will automatically replace obsolete or inferior compression
|
||||
schemes such as RLE or LZW with superior schemes such as Deflate, and
|
||||
convert monochrome images to CCITT G4. Since this is lossless, it always
|
||||
occurs and there is no way to disable it. Other non-image compressed
|
||||
objects are compressed as well.
|
||||
|
||||
## Fast web view
|
||||
|
||||
OCRmyPDF automatically optimizes PDFs for \"fast web view\" in Adobe
|
||||
Acrobat\'s parlance, or equivalently, linearizes PDFs so that the
|
||||
resources they reference are presented in the order a viewer needs them
|
||||
for sequential display. This reduces the latency of viewing a PDF both
|
||||
online and from local storage, in exchange for a slight increase in file
|
||||
size.
|
||||
|
||||
To disable this optimization and all others, use
|
||||
`ocrmypdf --optimize 0 ...` or the shorthand `-O0`.
|
||||
|
||||
Adobe Acrobat might not report the file as being \"fast web view\".
|
||||
|
||||
## Lossless optimizations
|
||||
|
||||
At optimization level `-O1` (the default), OCRmyPDF will also attempt
|
||||
lossless image optimization.
|
||||
|
||||
If a JBIG2 encoder is available, then monochrome images will be
|
||||
converted to JBIG2, with the potential for huge savings on large black
|
||||
and white images, since JBIG2 is far more efficient than any other
|
||||
monochrome (bi-level) compression. (All known US patents related to
|
||||
JBIG2 have probably expired, but it remains the responsibility of the
|
||||
user to supply a JBIG2 encoder such as
|
||||
[jbig2enc](https://github.com/agl/jbig2enc). OCRmyPDF does not implement
|
||||
JBIG2 encoding on its own.)
|
||||
|
||||
OCRmyPDF currently does not attempt to recompress losslessly compressed
|
||||
objects more aggressively.
|
||||
|
||||
## Lossy optimizations
|
||||
|
||||
At optimization level `-O1`, `-O2` and `-O3`, OCRmyPDF will some attempt
|
||||
loss image optimization.
|
||||
|
||||
If Ghostscript is used to create a PDF/A (the default), Ghostscript will
|
||||
optimize some images by converting them to JPEG, which are lossy. If
|
||||
`--output-type pdf` is used, there are no lossy optimizations. Ghostscript's
|
||||
JPEG conversion is quite safe.
|
||||
|
||||
If `pngquant` is installed, OCRmyPDF will use it to perform quantize
|
||||
paletted images to reduce their size.
|
||||
|
||||
The quality of JPEGs may be lowered, on the assumption that a lower
|
||||
quality image may be suitable for storage after OCR.
|
||||
|
||||
It is not possible to optimize all image types. Uncommon image types may
|
||||
be skipped by the optimizer.
|
||||
@@ -1,79 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
================
|
||||
PDF optimization
|
||||
================
|
||||
|
||||
OCRmyPDF includes an image-oriented PDF optimizer. By default, the optimizer
|
||||
runs with safe settings with the goal of improving compression at no loss of
|
||||
quality. At higher optimization levels, lossy optimizations may be applied and
|
||||
tuned. Optimization occurs after OCR, and only if OCR succeeded. It does not
|
||||
perform other possible optimizations such as deduplicating resources,
|
||||
consolidating fonts, simplifying vector drawings, or anything of that nature.
|
||||
|
||||
Optimization ranges from ``-O0`` through ``-O3``, where ``0`` disables
|
||||
optimization and ``3`` implements all options. ``1``, the default, performs only
|
||||
safe and lossless optimizations. (This is similar to GCC's optimization
|
||||
parameter.) The exact type of optimizations performed will vary over time.
|
||||
|
||||
PDF optimization requires third-party, optional tools for certain optimizations.
|
||||
If these are not installed or cannot be found by OCRmyPDF, optimization will not
|
||||
be as good.
|
||||
|
||||
Optimizations that always occurs
|
||||
================================
|
||||
|
||||
OCRmyPDF will automatically replace obsolete or inferior compression schemes
|
||||
such as RLE or LZW with superior schemes such as Deflate and converting
|
||||
monochrome images to CCITT G4. Since this is harmless it always occurs and there
|
||||
is no way to disable it. Other non-image compressed objects are compressed as
|
||||
well.
|
||||
|
||||
Fast web view
|
||||
=============
|
||||
|
||||
OCRmyPDF automatically optimizes PDFs for "fast web view" in Adobe Acrobat's
|
||||
parlance, or equivalently, linearizes PDFs so that the resources they reference
|
||||
are presented in the order a viewer needs them for sequential display. This
|
||||
reduces the latency of viewing a PDF both online and from local storage. This
|
||||
actually slightly increases the file size.
|
||||
|
||||
To disable this optimization and all others, use ``ocrmypdf --optimize 0 ...``
|
||||
or the shorthand ``-O0``.
|
||||
|
||||
Lossless optimizations
|
||||
======================
|
||||
|
||||
At optimization level ``-O1`` (the default), OCRmyPDF will also attempt lossless
|
||||
image optimization.
|
||||
|
||||
If a JBIG2 encoder is available, then monochrome images will be converted to
|
||||
JBIG2, with the potential for huge savings on large black and white images,
|
||||
since JBIG2 is far more efficient than any other monochrome (bi-level)
|
||||
compression. (All known US patents related to JBIG2 have probably expired, but
|
||||
it remains the responsibility of the user to supply a JBIG2 encoder such as
|
||||
`jbig2enc <https://github.com/agl/jbig2enc>`__. OCRmyPDF does not implement
|
||||
JBIG2 encoding on its own.)
|
||||
|
||||
OCRmyPDF currently does not attempt to recompress losslessly compressed objects
|
||||
more aggressively.
|
||||
|
||||
Lossy optimizations
|
||||
===================
|
||||
|
||||
At optimization level ``-O2`` and ``-O3``, OCRmyPDF will some attempt lossy
|
||||
image optimization.
|
||||
|
||||
If ``pngquant`` is installed, OCRmyPDF will use it to perform quantize paletted
|
||||
images to reduce their size.
|
||||
|
||||
The quality of JPEGs may be lowered, on the assumption that a lower quality
|
||||
image may be suitable for storage after OCR.
|
||||
|
||||
It is not possible to optimize all image types. Uncommon image types may be
|
||||
skipped by the optimizer.
|
||||
|
||||
OCRmyPDF provides :ref:`lossy mode JBIG2 <jbig2-lossy>` as an advanced feature
|
||||
that additional requires the argument ``--jbig2-lossy``.
|
||||
@@ -0,0 +1,115 @@
|
||||
(security)=
|
||||
|
||||
# PDF security issues
|
||||
|
||||
> OCRmyPDF should only be used on PDFs you trust. It is not designed to
|
||||
> protect you against malware.
|
||||
|
||||
Recognizing that many users have an interest in handling PDFs and
|
||||
applying OCR to PDFs they did not generate themselves, this article
|
||||
discusses the security implications of PDFs and how users can protect
|
||||
themselves.
|
||||
|
||||
The disclaimer applies: this software has no warranties of any kind.
|
||||
|
||||
## PDFs may contain malware
|
||||
|
||||
PDF is a rich, complex file format. The official PDF 1.7 specification,
|
||||
ISO 32000:2008, is hundreds of pages long and references several annexes
|
||||
each of which are similar in length. PDFs can contain video, audio, XML,
|
||||
JavaScript and other programming, and forms. In some cases, they can
|
||||
open internet connections to pre-selected URLs. All of these are
|
||||
possible attack vectors.
|
||||
|
||||
In short, PDFs [may contain
|
||||
viruses](https://security.stackexchange.com/questions/64052/can-a-pdf-file-contain-a-virus).
|
||||
|
||||
If you do not trust a PDF or its source, do not open it or use OCRmyPDF
|
||||
on it. Consider using a Docker container or virtual machine to isolate
|
||||
an untrusted PDF from your system.
|
||||
|
||||
## How OCRmyPDF processes PDFs
|
||||
|
||||
OCRmyPDF must open and interpret your PDF in order to insert an OCR
|
||||
layer. First, it runs all PDFs through
|
||||
[pikepdf](https://github.com/pikepdf/pikepdf), a library based on
|
||||
[QPDF](https://github.com/qpdf/qpdf), a program that repairs PDFs with
|
||||
syntax errors. This is done because, in the author\'s experience, a
|
||||
significant number of PDFs in the wild, especially those created by
|
||||
scanners, are not well-formed files. QPDF makes it more likely that
|
||||
OCRmyPDF will succeed, but offers no security guarantees. QPDF is also
|
||||
used to split the PDF into single page PDFs.
|
||||
|
||||
Finally, OCRmyPDF rasterizes each page of the PDF using
|
||||
[Ghostscript](http://ghostscript.com/) in `-dSAFER` mode.
|
||||
|
||||
Depending on the options specified, OCRmyPDF may graft the OCR layer
|
||||
into the existing PDF or it may essentially reconstruct (\"re-fry\") a
|
||||
visually identical PDF that may be quite different at the binary level.
|
||||
That said, OCRmyPDF is not a tool designed for sanitizing PDFs.
|
||||
|
||||
## Password protected PDFs
|
||||
|
||||
Password protected PDFs usually have two passwords, and owner and user
|
||||
password. When the user password is set to empty, PDF readers will open
|
||||
the file automatically and mark it as \"(SECURED)\". Password security
|
||||
can also request certain restrictions on the PDF, but anyone can remove
|
||||
these restrictions if they have either the owner *or* user password.
|
||||
Passwords mainly present a barrier for casual users.
|
||||
|
||||
OCRmyPDF cannot remove passwords from PDFs. If you want to remove a
|
||||
password from a PDF, you must use other software, such as `qpdf`.
|
||||
|
||||
If the owner and user password are set, a password is required for
|
||||
`qpdf`. If only the owner password is set, then the password can be
|
||||
stripped, even if one does not have the owner password. To remove the
|
||||
password from a using QPDF, use:
|
||||
|
||||
:::{code} bash
|
||||
qpdf --decrypt --password='abc123' input.pdf no_password.pdf
|
||||
:::
|
||||
|
||||
Then you can run OCRmyPDF on the file.
|
||||
|
||||
In its default mode, OCRmyPDF generates PDF/A. Passwords may not be set
|
||||
on PDF/A documents. If you want to set a password on the output PDF, you
|
||||
must specify `--output-type pdf`.
|
||||
|
||||
## Signature images
|
||||
|
||||
Many programs exist which are capable of inserting an image of
|
||||
someone\'s signature. On its own, this offers no security guarantees. It
|
||||
is trivial to remove the signature image and apply it to other files.
|
||||
This practice offers no real security.
|
||||
|
||||
## Digital signatures
|
||||
|
||||
Important documents can be digitally signed and certified to attest to
|
||||
their authorship, approval or execution of a legal agreement. OCRmyPDF
|
||||
will detect signed PDFs and will not modify them, unless the
|
||||
`--invalidate-digital-signatures` option is used, which will invalidate
|
||||
any signatures. (The signature may still be present in the PDF if
|
||||
opened, but PDF readers will not validate it.)
|
||||
|
||||
A digital signature adds a cryptographic hash of the document to the
|
||||
document, so tamper protection is provided. That also precludes OCRmyPDF
|
||||
from modifying the document and preserving the signature.
|
||||
|
||||
Digital signatures are not the same as a signature image. A digital
|
||||
signature is a cryptographic hash of the document that is encrypted with
|
||||
the author\'s private key. The signature is decrypted with the author\'s
|
||||
public key. The public key is usually distributed by a certificate
|
||||
authority. The signature is then verified by the PDF reader. If the
|
||||
document is modified, the signature will be invalidated.
|
||||
|
||||
## Certificate-encrypted PDFs
|
||||
|
||||
PDFs can be encrypted with a certificate. This is a more secure form of
|
||||
encryption than a password. The certificate is usually issued by a
|
||||
certificate authority. A certificate is used to encrypt the document
|
||||
using the public key for the benefit of a specific recipient who
|
||||
possesses the private key.
|
||||
|
||||
OCRmyPDF cannot open certificate-encrypted PDFs. If you have the
|
||||
certificate, you can use other PDF software, such as Acrobat, to decrypt
|
||||
the PDF.
|
||||
@@ -1,166 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
===================
|
||||
PDF security issues
|
||||
===================
|
||||
|
||||
OCRmyPDF should only be used on PDFs you trust. It is not designed to
|
||||
protect you against malware.
|
||||
|
||||
Recognizing that many users have an interest in handling PDFs and
|
||||
applying OCR to PDFs they did not generate themselves, this article
|
||||
discusses the security implications of PDFs and how users can protect
|
||||
themselves.
|
||||
|
||||
The disclaimer applies: this software has no warranties of any kind.
|
||||
|
||||
PDFs may contain malware
|
||||
========================
|
||||
|
||||
PDF is a rich, complex file format. The official PDF 1.7 specification,
|
||||
ISO 32000:2008, is hundreds of pages long and references several annexes
|
||||
each of which are similar in length. PDFs can contain video, audio, XML,
|
||||
JavaScript and other programming, and forms. In some cases, they can
|
||||
open internet connections to pre-selected URLs. All of these are possible
|
||||
attack vectors.
|
||||
|
||||
In short, PDFs `may contain
|
||||
viruses <https://security.stackexchange.com/questions/64052/can-a-pdf-file-contain-a-virus>`__.
|
||||
|
||||
This
|
||||
`article <https://theinvisiblethings.blogspot.ca/2013/02/converting-untrusted-pdfs-into-trusted.html>`__
|
||||
describes a high-paranoia method which allows potentially hostile PDFs
|
||||
to be viewed and rasterized safely in a disposable virtual machine. A
|
||||
trusted PDF created in this manner is converted to images and loses all
|
||||
information making it searchable and losing all compression. OCRmyPDF
|
||||
could be used to restore searchability.
|
||||
|
||||
How OCRmyPDF processes PDFs
|
||||
===========================
|
||||
|
||||
OCRmyPDF must open and interpret your PDF in order to insert an OCR
|
||||
layer. First, it runs all PDFs through
|
||||
`pikepdf <https://github.com/pikepdf/pikepdf>`__, a library based on
|
||||
`qpdf <https://github.com/qpdf/qpdf>`__, a program that repairs PDFs
|
||||
with syntax errors. This is done because, in the author's experience, a
|
||||
significant number of PDFs in the wild, especially those created by
|
||||
scanners, are not well-formed files. qpdf makes it more likely that
|
||||
OCRmyPDF will succeed, but offers no security guarantees. qpdf is also
|
||||
used to split the PDF into single page PDFs.
|
||||
|
||||
Finally, OCRmyPDF rasterizes each page of the PDF using
|
||||
`Ghostscript <http://ghostscript.com/>`__ in ``-dSAFER`` mode.
|
||||
|
||||
Depending on the options specified, OCRmyPDF may graft the OCR layer
|
||||
into the existing PDF or it may essentially reconstruct ("re-fry") a
|
||||
visually identical PDF that may be quite different at the binary level.
|
||||
That said, OCRmyPDF is not a tool designed for sanitizing PDFs.
|
||||
|
||||
.. _ocr-service:
|
||||
|
||||
Using OCRmyPDF online or as a service
|
||||
=====================================
|
||||
|
||||
OCRmyPDF is not designed for use as a public web service where a
|
||||
malicious user could upload a chosen PDF. In particular, it is not
|
||||
necessarily secure against PDF malware or PDFs that cause denial of
|
||||
service. OCRmyPDF relies on Ghostscript, and therefore, if deployed
|
||||
online one should be prepared to comply with Ghostscript's Affero GPL
|
||||
license, and any other licenses.
|
||||
|
||||
Setting aside these concerns, a side effect of OCRmyPDF is that it may
|
||||
incidentally sanitize PDFs containing certain types of malware. It
|
||||
repairs the PDF with pikepdf/libqpdf, which could correct malformed PDF
|
||||
structures that are part of an attack. When PDF/A output is selected
|
||||
(the default), the input PDF is partially reconstructed by Ghostscript.
|
||||
When ``--force-ocr`` is used, all pages are rasterized and reconverted
|
||||
to PDF, which could remove malware in embedded images.
|
||||
|
||||
OCRmyPDF should be relatively safe to use in a trusted intranet, with
|
||||
some considerations:
|
||||
|
||||
Limiting CPU usage
|
||||
------------------
|
||||
|
||||
OCRmyPDF will attempt to use all available CPUs and storage, so
|
||||
executing ``nice ocrmypdf`` or limiting the number of jobs with the
|
||||
``-j`` argument may ensure the server remains available. Another option
|
||||
would be to run OCRmyPDF jobs inside a Docker container, a virtual machine,
|
||||
or a cloud instance, which can impose its own limits on CPU usage and be
|
||||
terminated "from orbit" if it fails to complete.
|
||||
|
||||
Temporary storage requirements
|
||||
------------------------------
|
||||
|
||||
OCRmyPDF will use a large amount of temporary storage for its work,
|
||||
proportional to the total number of pixels needed to rasterize the PDF.
|
||||
The raster image of a 8.5×11" color page at 300 DPI takes 25 MB
|
||||
uncompressed; OCRmyPDF saves its intermediates as PNG, but that still
|
||||
means it requires about 9 MB per intermediate based on average
|
||||
compression ratios. Multiple intermediates per page are also required,
|
||||
depending on the command line given. A rule of thumb would be to allow
|
||||
100 MB of temporary storage per page in a file – meaning that a small
|
||||
cloud servers or small VM partitions should be provisioned with plenty
|
||||
of extra space, if say, a 500 page file might be sent.
|
||||
|
||||
To check temporary storage usage on actual files, run
|
||||
``ocrmypdf -k ...`` which will preserve and print the path to temporary
|
||||
storage when the job is done.
|
||||
|
||||
To change where temporary files are stored, change the ``TMPDIR``
|
||||
environment variable for ocrmypdf's environment. (Python's
|
||||
``tempfile.gettempdir()`` returns the root directory in which temporary
|
||||
files will be stored.) For example, one could redirect ``TMPDIR`` to a
|
||||
large RAM disk to avoid wear on HDD/SSD and potentially improve
|
||||
performance. On Amazon Web Services, ``TMPDIR`` can be set to `empheral
|
||||
storage <https://docs.aws.amazon.com/AWSEC2/latest/UserGuide/InstanceStorage.html>`__.
|
||||
|
||||
Timeouts
|
||||
--------
|
||||
|
||||
To prevent excessively long OCR jobs consider setting
|
||||
``--tesseract-timeout`` and/or ``--skip-big`` arguments. ``--skip-big``
|
||||
is particularly helpful if your PDFs include documents such as reports
|
||||
on standard page sizes with large images attached - often large images
|
||||
are not worth OCR'ing anyway.
|
||||
|
||||
Commercial alternatives
|
||||
-----------------------
|
||||
|
||||
The author also provides professional services that include OCR and
|
||||
building databases around PDFs, and is happy to provide consultation.
|
||||
|
||||
Abbyy Cloud OCR is viable commercial alternative with a web services
|
||||
API. Amazon Textract, Google Cloud Vision, and Microsoft Azure
|
||||
Computer Vision provide advanced OCR but have less PDF rendering capability.
|
||||
|
||||
Password protection, digital signatures and certification
|
||||
=========================================================
|
||||
|
||||
Password protected PDFs usually have two passwords, and owner and user
|
||||
password. When the user password is set to empty, PDF readers will open
|
||||
the file automatically and marked it as "(SECURED)". While not as
|
||||
reliable as a digital signature, this indicates that whoever set the
|
||||
password approved of the file at that time. When the user password is
|
||||
set, the document cannot be viewed without the password.
|
||||
|
||||
Either way, OCRmyPDF does not remove passwords from PDFs and exits with
|
||||
an error on encountering them.
|
||||
|
||||
``qpdf`` can remove passwords. If the owner and user password are set, a
|
||||
password is required for ``qpdf``. If only the owner password is set, then the
|
||||
password can be stripped, even if one does not have the owner password.
|
||||
|
||||
After OCR is applied, password protection is not permitted on PDF/A
|
||||
documents but the file can be converted to regular PDF.
|
||||
|
||||
Many programs exist which are capable of inserting an image of someone's
|
||||
signature. On its own, this offers no security guarantees. It is trivial
|
||||
to remove the signature image and apply it to other files. This practice
|
||||
offers no real security.
|
||||
|
||||
Important documents can be digitally signed and certified to attest to
|
||||
their authorship. OCRmyPDF cannot do this. Open source tools such as
|
||||
pdfbox (Java) have this capability as does Adobe Acrobat.
|
||||
@@ -0,0 +1,24 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Performance
|
||||
|
||||
Some users have noticed that current versions of OCRmyPDF do not run as
|
||||
quickly as some older versions (specifically 6.x and older). This is
|
||||
because OCRmyPDF added image optimization as a postprocessing step, and
|
||||
it is enabled by default.
|
||||
|
||||
## Speed
|
||||
|
||||
If running OCRmyPDF quickly is your main goal, you can use settings such
|
||||
as:
|
||||
|
||||
- `--optimize 0` to disable file size optimization
|
||||
- `--output-type pdf` to disable PDF/A generation
|
||||
- `--fast-web-view 999999` to disable fast web view optimization
|
||||
- `--skip-big` to skip large images, if some pages have large images
|
||||
|
||||
You can also avoid:
|
||||
|
||||
- `--force-ocr`
|
||||
- Image preprocessing
|
||||
@@ -1,26 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
===========
|
||||
Performance
|
||||
===========
|
||||
|
||||
Some users have noticed that current versions of OCRmyPDF do not run as quickly
|
||||
as some older versions (specifically 6.x and older). This is because OCRmyPDF
|
||||
added image optimization as a postprocessing step, and it is enabled by default.
|
||||
|
||||
Speed
|
||||
=====
|
||||
|
||||
If running OCRmyPDF quickly is your main goal, you can use settings such as:
|
||||
|
||||
* ``--optimize 0`` to disable file size optimization
|
||||
* ``--output-type pdf`` to disable PDF/A generation
|
||||
* ``--fast-web-view 0`` to disable fast web view optimization
|
||||
* ``--skip-big`` to skip large images, if some pages have large images
|
||||
|
||||
You can also avoid:
|
||||
|
||||
* ``--force-ocr``
|
||||
* Image preprocessing
|
||||
+416
@@ -0,0 +1,416 @@
|
||||
% SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
% SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
# Plugins
|
||||
|
||||
> The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL
|
||||
> NOT", "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and
|
||||
> "OPTIONAL" in this document are to be interpreted as described in
|
||||
> RFC 2119.
|
||||
|
||||
You can use plugins to customize the behavior of OCRmyPDF at certain points of
|
||||
interest.
|
||||
|
||||
Currently, it is possible to:
|
||||
|
||||
- add new command line arguments
|
||||
- override the decision for whether or not to perform OCR on a particular file
|
||||
- modify the image is about to be sent for OCR
|
||||
- modify the page image before it is converted to PDF
|
||||
- replace the Tesseract OCR with another OCR engine that has similar behavior
|
||||
- replace Ghostscript with another PDF to image converter (rasterizer) or
|
||||
PDF/A generator
|
||||
|
||||
OCRmyPDF plugins are based on the Python `pluggy` package and conform to its
|
||||
conventions. Note that: plugins installed with as setuptools entrypoints are
|
||||
not checked currently, because OCRmyPDF assumes you may not want to enable
|
||||
plugins for all files.
|
||||
|
||||
See \[OCRmyPDF-EasyOCR\](<https://github.com/ocrmypdf/OCRmyPDF-EasyOCR>) for an
|
||||
example of a straightforward, fully working plugin.
|
||||
|
||||
## Script plugins
|
||||
|
||||
Script plugins may be called from the command line, by specifying the name of a file.
|
||||
Script plugins may be convenient for informal or "one-off" plugins, when a certain
|
||||
batch of files needs a special processing step for example.
|
||||
|
||||
```bash
|
||||
ocrmypdf --plugin ocrmypdf_example_plugin.py input.pdf output.pdf
|
||||
```
|
||||
|
||||
Multiple plugins may be installed by issuing the `--plugin` argument multiple times.
|
||||
|
||||
## Packaged plugins
|
||||
|
||||
Installed plugins may be installed into the same virtual environment as OCRmyPDF
|
||||
is installed into. They may be invoked using Python standard module naming.
|
||||
If you are intending to distribute a plugin, please package it.
|
||||
|
||||
```bash
|
||||
ocrmypdf --plugin ocrmypdf_fancypants.pockets.contents input.pdf output.pdf
|
||||
```
|
||||
|
||||
OCRmyPDF does not automatically import plugins, because the assumption is that
|
||||
plugins affect different files differently and you may not want them activated
|
||||
all the time. The command line or `ocrmypdf.ocr(plugin='...')` must call
|
||||
for them.
|
||||
|
||||
Third parties that wish to distribute packages for ocrmypdf should package them
|
||||
as packaged plugins, and these modules should begin with the name `ocrmypdf_`
|
||||
similar to `pytest` packages such as `pytest-cov` (the package) and
|
||||
`pytest_cov` (the module).
|
||||
|
||||
:::{note}
|
||||
We recommend plugin authors name their plugins with the prefix
|
||||
`ocrmypdf-` (for the package name on PyPI) and `ocrmypdf_` (for the
|
||||
module), just like pytest plugins. At the same time, please make it clear
|
||||
that your package is not official.
|
||||
:::
|
||||
|
||||
## Plugins
|
||||
|
||||
You can also create a plugin that OCRmyPDF will always automatically load if both are
|
||||
installed in the same virtual environment, using a project entrypoint.
|
||||
OCRmyPDF uses the entrypoint namespace "ocrmypdf".
|
||||
|
||||
For example, `pyproject.toml` would need to contain the following, for a plugin named
|
||||
`ocrmypdf-exampleplugin`:
|
||||
|
||||
```toml
|
||||
[project]
|
||||
name = "ocrmypdf-exampleplugin"
|
||||
|
||||
[project.entry-points."ocrmypdf"]
|
||||
exampleplugin = "exampleplugin.pluginmodule"
|
||||
```
|
||||
|
||||
## Plugin requirements
|
||||
|
||||
OCRmyPDF generally uses multiple worker processes. When a new worker is started,
|
||||
Python will import all plugins again, including all plugins that were imported earlier.
|
||||
This means that the global state of a plugin in one worker will not be shared with
|
||||
other workers. As such, plugin hook implementations should be stateless, relying
|
||||
only on their inputs. Hook implementations may use their input parameters to
|
||||
to obtain a reference to shared state prepared by another hook implementation.
|
||||
Plugins must expect that other instances of the plugin will be running
|
||||
simultaneously.
|
||||
|
||||
The `context` object that is passed to many hooks can be used to share information
|
||||
about a file being worked on. Plugins must write private, plugin-specific data to
|
||||
a subfolder named `{options.work_folder}/ocrmypdf-plugin-name`. Plugins MAY
|
||||
read and write files in `options.work_folder`, but should be aware that their
|
||||
semantics are subject to change.
|
||||
|
||||
OCRmyPDF will delete `options.work_folder` when it has finished OCRing
|
||||
a file, unless invoked with `--keep-temporary-files`.
|
||||
|
||||
The documentation for some plugin hooks contain a detailed description of the
|
||||
execution context in which they will be called.
|
||||
|
||||
Plugins should be prepared to work whether executed in worker threads or worker
|
||||
processes. Generally, OCRmyPDF uses processes, but has a semi-hidden threaded
|
||||
argument that simplifies debugging.
|
||||
|
||||
## Plugin hooks
|
||||
|
||||
A plugin may provide the following hooks. Hooks must be decorated with
|
||||
`ocrmypdf.hookimpl`, for example:
|
||||
|
||||
```python
|
||||
from ocrmypdf import hookimpl
|
||||
|
||||
@hookimpl
|
||||
def add_options(parser):
|
||||
pass
|
||||
```
|
||||
|
||||
The following is a complete list of hooks that are available, and when
|
||||
they are called.
|
||||
|
||||
(firstresult)=
|
||||
|
||||
**Note on firstresult hooks**
|
||||
|
||||
If multiple plugins install implementations for this hook, they will be called in
|
||||
the reverse of the order in which they are installed (i.e., last plugin wins).
|
||||
When each hook implementation is called in order, the first implementation that
|
||||
returns a value other than `None` will "win" and prevent execution of all other
|
||||
hooks. As such, you cannot "chain" a series of plugin filters together in this
|
||||
way. Instead, a single hook implementation should be responsible for any such
|
||||
chaining operations.
|
||||
|
||||
## Examples
|
||||
|
||||
- OCRmyPDF's test suite contains several plugins that are used to simulate certain
|
||||
test conditions.
|
||||
- [ocrmypdf-papermerge](https://github.com/papermerge/OCRmyPDF_papermerge) is
|
||||
a production plugin that integrates OCRmyPDF and the Papermerge document
|
||||
management system.
|
||||
|
||||
### Suppressing or overriding other plugins
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.initialize
|
||||
```
|
||||
|
||||
### Custom command line arguments
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.add_options
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.check_options
|
||||
```
|
||||
|
||||
### Plugin option models
|
||||
|
||||
Plugins can define their own option models using Pydantic. This allows plugins to:
|
||||
|
||||
- Define type-safe option structures with validation
|
||||
- Add CLI arguments that map to their option model fields
|
||||
- Access options via nested namespaces (e.g., `options.tesseract.timeout`)
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.register_options
|
||||
```
|
||||
|
||||
Plugin options can be accessed in two ways:
|
||||
|
||||
1. **Flat access** (backward compatible): `options.tesseract_timeout`
|
||||
2. **Nested access**: `options.tesseract.timeout`
|
||||
|
||||
Both access patterns are equivalent and return the same values.
|
||||
|
||||
:::{note}
|
||||
**Plugin Interface Change**: Starting in OCRmyPDF v17.0.0, plugin hooks receive
|
||||
`OcrOptions` objects instead of `argparse.Namespace` objects. Most plugins will
|
||||
continue working due to duck-typing compatibility, but plugin developers should
|
||||
update their type hints accordingly.
|
||||
:::
|
||||
|
||||
### Migration guide for plugin developers
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
**Update imports:**
|
||||
|
||||
```python
|
||||
from ocrmypdf._options import OcrOptions
|
||||
```
|
||||
|
||||
**Update type hints:**
|
||||
|
||||
```python
|
||||
# Before (v16 and earlier)
|
||||
def check_options(options: argparse.Namespace) -> None:
|
||||
...
|
||||
|
||||
# After (v17+)
|
||||
def check_options(options: OcrOptions) -> None:
|
||||
...
|
||||
```
|
||||
|
||||
**Attribute access unchanged:**
|
||||
|
||||
```python
|
||||
# These work exactly as before
|
||||
options.languages
|
||||
options.output_type
|
||||
options.tesseract_timeout
|
||||
```
|
||||
|
||||
**Remove in-place modifications:**
|
||||
|
||||
```python
|
||||
# Before (v16 pattern - no longer recommended)
|
||||
def check_options(options):
|
||||
options.some_computed_value = compute_value(options)
|
||||
|
||||
# After (v17 pattern - compute at point of use)
|
||||
def some_function(options):
|
||||
computed = compute_value(options)
|
||||
use_computed(computed)
|
||||
```
|
||||
|
||||
### Execution and progress reporting
|
||||
|
||||
```{eval-rst}
|
||||
.. autoclass:: ocrmypdf.pluginspec.ProgressBar
|
||||
:members:
|
||||
:special-members: __init__, __enter__, __exit__
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autoclass:: ocrmypdf.pluginspec.Executor
|
||||
:members:
|
||||
:special-members: __call__
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.get_logging_console
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.get_executor
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.get_progressbar_class
|
||||
```
|
||||
|
||||
### Applying special behavior before processing
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.validate
|
||||
```
|
||||
|
||||
### PDF page to image
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.rasterize_pdf_page
|
||||
```
|
||||
|
||||
### Modifying intermediate images
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.filter_ocr_image
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.filter_page_image
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.filter_pdf_page
|
||||
```
|
||||
|
||||
### OCR engine
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.get_ocr_engine
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autoclass:: ocrmypdf.pluginspec.OcrEngine
|
||||
:members:
|
||||
|
||||
.. automethod:: __str__
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autoclass:: ocrmypdf.pluginspec.OrientationConfidence
|
||||
```
|
||||
|
||||
### PDF/A production
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.generate_pdfa
|
||||
```
|
||||
|
||||
### PDF optimization
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.optimize_pdf
|
||||
```
|
||||
|
||||
```{eval-rst}
|
||||
.. autofunction:: ocrmypdf.pluginspec.is_optimization_enabled
|
||||
```
|
||||
|
||||
### Working with OcrElement trees
|
||||
|
||||
:::{versionadded} 17.0.0
|
||||
:::
|
||||
|
||||
OCRmyPDF v17 introduces the `OcrElement` dataclass for representing OCR
|
||||
output in an engine-agnostic format. This enables plugins to work with
|
||||
OCR results without parsing hOCR XML.
|
||||
|
||||
**Key classes:**
|
||||
|
||||
```python
|
||||
from ocrmypdf import OcrElement, OcrClass, BoundingBox
|
||||
|
||||
# OcrElement - represents any OCR structural unit
|
||||
page = OcrElement(
|
||||
ocr_class=OcrClass.PAGE,
|
||||
bbox=BoundingBox(0, 0, 612, 792),
|
||||
children=[...]
|
||||
)
|
||||
|
||||
# BoundingBox - axis-aligned bounding box (left, top, right, bottom)
|
||||
bbox = BoundingBox(left=100, top=50, right=300, bottom=80)
|
||||
|
||||
# OcrClass - constants for element types
|
||||
OcrClass.PAGE # "ocr_page"
|
||||
OcrClass.LINE # "ocr_line"
|
||||
OcrClass.WORD # "ocrx_word"
|
||||
OcrClass.PARAGRAPH # "ocr_par"
|
||||
```
|
||||
|
||||
**Navigating the tree:**
|
||||
|
||||
```python
|
||||
# Get all words in a page
|
||||
words = page.words # Returns list[OcrElement]
|
||||
|
||||
# Get all lines
|
||||
lines = page.lines
|
||||
|
||||
# Get combined text
|
||||
text = page.get_text_recursive()
|
||||
|
||||
# Iterate by class
|
||||
for para in page.paragraphs:
|
||||
print(para.get_text_recursive())
|
||||
```
|
||||
|
||||
**OCR engine plugins:**
|
||||
|
||||
Plugins implementing custom OCR engines can now output `OcrElement` trees
|
||||
directly via the `generate_ocr()` method, bypassing hOCR entirely:
|
||||
|
||||
```python
|
||||
from pathlib import Path
|
||||
from ocrmypdf.pluginspec import OcrEngine
|
||||
from ocrmypdf import OcrElement, OcrClass, BoundingBox
|
||||
|
||||
class MyOcrEngine(OcrEngine):
|
||||
def generate_ocr(
|
||||
self,
|
||||
input_file: Path,
|
||||
options,
|
||||
context,
|
||||
) -> OcrElement:
|
||||
# Perform OCR and return OcrElement tree directly
|
||||
# No need to generate hOCR XML
|
||||
return OcrElement(
|
||||
ocr_class=OcrClass.PAGE,
|
||||
bbox=BoundingBox(0, 0, width, height),
|
||||
dpi=300,
|
||||
children=[
|
||||
OcrElement(
|
||||
ocr_class=OcrClass.LINE,
|
||||
bbox=BoundingBox(100, 50, 500, 80),
|
||||
children=[
|
||||
OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
bbox=BoundingBox(100, 50, 200, 80),
|
||||
text="Hello",
|
||||
),
|
||||
# ... more words
|
||||
]
|
||||
),
|
||||
# ... more lines
|
||||
]
|
||||
)
|
||||
|
||||
def supports_generate_ocr(self) -> bool:
|
||||
return True # Indicate this engine uses generate_ocr()
|
||||
```
|
||||
|
||||
This approach is simpler than generating hOCR and allows modern OCR
|
||||
engines to integrate more naturally with OCRmyPDF.
|
||||
@@ -1,230 +0,0 @@
|
||||
.. SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
..
|
||||
.. SPDX-License-Identifier: CC-BY-SA-4.0
|
||||
|
||||
=======
|
||||
Plugins
|
||||
=======
|
||||
|
||||
The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL
|
||||
NOT", "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and
|
||||
"OPTIONAL" in this document are to be interpreted as described in
|
||||
RFC 2119.
|
||||
|
||||
You can use plugins to customize the behavior of OCRmyPDF at certain points of
|
||||
interest.
|
||||
|
||||
Currently, it is possible to:
|
||||
|
||||
- add new command line arguments
|
||||
- override the decision for whether or not to perform OCR on a particular file
|
||||
- modify the image is about to be sent for OCR
|
||||
- modify the page image before it is converted to PDF
|
||||
- replace the Tesseract OCR with another OCR engine that has similar behavior
|
||||
- replace Ghostscript with another PDF to image converter (rasterizer) or
|
||||
PDF/A generator
|
||||
|
||||
OCRmyPDF plugins are based on the Python ``pluggy`` package and conform to its
|
||||
conventions. Note that: plugins installed with as setuptools entrypoints are
|
||||
not checked currently, because OCRmyPDF assumes you may not want to enable
|
||||
plugins for all files.
|
||||
|
||||
Script plugins
|
||||
==============
|
||||
|
||||
Script plugins may be called from the command line, by specifying the name of a file.
|
||||
Script plugins may be convenient for informal or "one-off" plugins, when a certain
|
||||
batch of files needs a special processing step for example.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --plugin ocrmypdf_example_plugin.py input.pdf output.pdf
|
||||
|
||||
Multiple plugins may be installed by issuing the ``--plugin`` argument multiple times.
|
||||
|
||||
Packaged plugins
|
||||
================
|
||||
|
||||
Installed plugins may be installed into the same virtual environment as OCRmyPDF
|
||||
is installed into. They may be invoked using Python standard module naming.
|
||||
If you are intending to distribute a plugin, please package it.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --plugin ocrmypdf_fancypants.pockets.contents input.pdf output.pdf
|
||||
|
||||
OCRmyPDF does not automatically import plugins, because the assumption is that
|
||||
plugins affect different files differently and you may not want them activated
|
||||
all the time. The command line or ``ocrmypdf.ocr(plugin='...')`` must call
|
||||
for them.
|
||||
|
||||
Third parties that wish to distribute packages for ocrmypdf should package them
|
||||
as packaged plugins, and these modules should begin with the name ``ocrmypdf_``
|
||||
similar to ``pytest`` packages such as ``pytest-cov`` (the package) and
|
||||
``pytest_cov`` (the module).
|
||||
|
||||
.. note::
|
||||
|
||||
We recommend plugin authors name their plugins with the prefix
|
||||
``ocrmypdf-`` (for the package name on PyPI) and ``ocrmypdf_`` (for the
|
||||
module), just like pytest plugins. At the same time, please make it clear
|
||||
that your package is not official.
|
||||
|
||||
Setuptools plugins
|
||||
==================
|
||||
|
||||
You can also create a plugin that OCRmyPDF will always automatically load if both are
|
||||
installed in the same virtual environment, using a setuptools entrypoint.
|
||||
|
||||
Your package's ``pyproject.toml`` would need to contain the following, for a plugin
|
||||
named ``ocrmypdf-exampleplugin``:
|
||||
|
||||
.. code-block:: toml
|
||||
|
||||
[project]
|
||||
name = "ocrmypdf-exampleplugin"
|
||||
|
||||
[project.entry-points."ocrmypdf"]
|
||||
exampleplugin = "exampleplugin.pluginmodule"
|
||||
|
||||
.. code-block:: ini
|
||||
|
||||
# equivalent setup.cfg
|
||||
[options.entry_points]
|
||||
ocrmypdf =
|
||||
exampleplugin = exampleplugin.pluginmodule
|
||||
|
||||
Plugin requirements
|
||||
===================
|
||||
|
||||
OCRmyPDF generally uses multiple worker processes. When a new worker is started,
|
||||
Python will import all plugins again, including all plugins that were imported earlier.
|
||||
This means that the global state of a plugin in one worker will not be shared with
|
||||
other workers. As such, plugin hook implementations should be stateless, relying
|
||||
only on their inputs. Hook implementations may use their input parameters to
|
||||
to obtain a reference to shared state prepared by another hook implementation.
|
||||
Plugins must expect that other instances of the plugin will be running
|
||||
simultaneously.
|
||||
|
||||
The ``context`` object that is passed to many hooks can be used to share information
|
||||
about a file being worked on. Plugins must write private, plugin-specific data to
|
||||
a subfolder named ``{options.work_folder}/ocrmypdf-plugin-name``. Plugins MAY
|
||||
read and write files in ``options.work_folder``, but should be aware that their
|
||||
semantics are subject to change.
|
||||
|
||||
OCRmyPDF will delete ``options.work_folder`` when it has finished OCRing
|
||||
a file, unless invoked with ``--keep-temporary-files``.
|
||||
|
||||
The documentation for some plugin hooks contain a detailed description of the
|
||||
execution context in which they will be called.
|
||||
|
||||
Plugins should be prepared to work whether executed in worker threads or worker
|
||||
processes. Generally, OCRmyPDF uses processes, but has a semi-hidden threaded
|
||||
argument that simplifies debugging.
|
||||
|
||||
|
||||
Plugin hooks
|
||||
============
|
||||
|
||||
A plugin may provide the following hooks. Hooks must be decorated with
|
||||
``ocrmypdf.hookimpl``, for example:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from ocrmpydf import hookimpl
|
||||
|
||||
@hookimpl
|
||||
def add_options(parser):
|
||||
pass
|
||||
|
||||
The following is a complete list of hooks that are available, and when
|
||||
they are called.
|
||||
|
||||
.. _firstresult:
|
||||
|
||||
**Note on firstresult hooks**
|
||||
|
||||
If multiple plugins install implementations for this hook, they will be called in
|
||||
the reverse of the order in which they are installed (i.e., last plugin wins).
|
||||
When each hook implementation is called in order, the first implementation that
|
||||
returns a value other than ``None`` will "win" and prevent execution of all other
|
||||
hooks. As such, you cannot "chain" a series of plugin filters together in this
|
||||
way. Instead, a single hook implementation should be responsible for any such
|
||||
chaining operations.
|
||||
|
||||
Examples
|
||||
========
|
||||
|
||||
* OCRmyPDF's test suite contains several plugins that are used to simulate certain
|
||||
test conditions.
|
||||
* `ocrmypdf-papermerge <https://github.com/papermerge/OCRmyPDF_papermerge>`_ is
|
||||
a production plugin that integrates OCRmyPDF and the Papermerge document
|
||||
management system.
|
||||
|
||||
|
||||
Suppressing or overriding other plugins
|
||||
---------------------------------------
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.initialize
|
||||
|
||||
Custom command line arguments
|
||||
-----------------------------
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.add_options
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.check_options
|
||||
|
||||
Execution and progress reporting
|
||||
--------------------------------
|
||||
|
||||
.. autoclass:: ocrmypdf.pluginspec.Executor
|
||||
:members:
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.get_logging_console
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.get_executor
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.get_progressbar_class
|
||||
|
||||
Applying special behavior before processing
|
||||
-------------------------------------------
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.validate
|
||||
|
||||
PDF page to image
|
||||
-----------------
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.rasterize_pdf_page
|
||||
|
||||
Modifying intermediate images
|
||||
-----------------------------
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.filter_ocr_image
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.filter_page_image
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.filter_pdf_page
|
||||
|
||||
OCR engine
|
||||
----------
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.get_ocr_engine
|
||||
|
||||
.. autoclass:: ocrmypdf.pluginspec.OcrEngine
|
||||
:members:
|
||||
|
||||
.. automethod:: __str__
|
||||
|
||||
.. autoclass:: ocrmypdf.pluginspec.OrientationConfidence
|
||||
|
||||
PDF/A production
|
||||
----------------
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.generate_pdfa
|
||||
|
||||
PDF optimization
|
||||
----------------
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.optimize_pdf
|
||||
|
||||
.. autofunction:: ocrmypdf.pluginspec.is_optimization_enabled
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,240 @@
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
"""This is a simple web service/HTTP wrapper for OCRmyPDF.
|
||||
|
||||
This may be more convenient than the command line tool for some Docker users.
|
||||
Note that OCRmyPDF uses Ghostscript, which is licensed under AGPLv3+. While
|
||||
OCRmyPDF is under GPLv3, this file is distributed under the Affero GPLv3+ license,
|
||||
to emphasize that SaaS deployments should make sure they comply with
|
||||
Ghostscript's license as well as OCRmyPDF's.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from functools import partial
|
||||
from operator import getitem
|
||||
from pathlib import Path
|
||||
from tempfile import NamedTemporaryFile
|
||||
|
||||
import pikepdf
|
||||
import streamlit as st
|
||||
|
||||
from ocrmypdf._defaults import DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
|
||||
|
||||
def get_host_url_with_port(port: int) -> str:
|
||||
"""Get the host URL for the web service. Hacky."""
|
||||
host_url = st.context.headers["host"]
|
||||
try:
|
||||
host, _streamlit_port = host_url.split(":", maxsplit=1)
|
||||
except ValueError:
|
||||
host = host_url
|
||||
return f"//{host}:{port}" # Use the same protocol
|
||||
|
||||
|
||||
st.title("OCRmyPDF Web Service")
|
||||
|
||||
uploaded = st.file_uploader("Upload input PDF or image", type=["pdf"], key="file")
|
||||
|
||||
mode = st.selectbox("Mode", options=["normal", "skip-text", "force-ocr", "redo-ocr"])
|
||||
|
||||
pages = st.text_input(
|
||||
"Pages", value="", help="Comma-separated list of pages to process"
|
||||
)
|
||||
|
||||
with st.expander("Input options"):
|
||||
invalidate_digital_signatures = st.checkbox(
|
||||
"Invalidate digital signatures", value=False
|
||||
)
|
||||
language = st.selectbox("Language", options=["eng", "deu", "fra", "spa"])
|
||||
|
||||
image_dpi = st.slider(
|
||||
"Image DPI", value=300, key="image_dpi", min_value=1, max_value=5000, step=50
|
||||
)
|
||||
with st.expander("Preprocessing"):
|
||||
skip_big = st.checkbox("Skip OCR on big pages", value=False, key="skip_big")
|
||||
oversample = st.slider("Oversample", min_value=0, max_value=5000, value=0, step=50)
|
||||
rotate_pages = st.checkbox("Rotate pages", value=False, key="rotate")
|
||||
deskew = st.checkbox("Deskew pages", value=False, key="deskew")
|
||||
clean = st.checkbox("Clean pages before OCR", value=False, key="clean")
|
||||
clean_final = st.checkbox("Clean final", value=False, key="clean_final")
|
||||
remove_vectors = st.checkbox("Remove vectors", value=False, key="remove_vectors")
|
||||
|
||||
|
||||
with st.expander("Output options"):
|
||||
output_type = st.selectbox(
|
||||
"Output type", options=["pdfa", "pdf", "pdfa-1", "pdfa-2", "pdfa-3", "none"]
|
||||
)
|
||||
|
||||
pdf_renderer = st.selectbox(
|
||||
"PDF renderer", options=["auto", "hocr", "hocrdebug", "sandwich"]
|
||||
)
|
||||
|
||||
optimize = st.selectbox("Optimize", options=["0", "1", "2", "3"])
|
||||
|
||||
st.selectbox("PDF/A compression", options=["auto", "jpeg", "lossless"])
|
||||
|
||||
with st.expander("Metadata"):
|
||||
title = author = keywords = subject = None
|
||||
if uploaded:
|
||||
with pikepdf.open(uploaded) as pdf, pdf.open_metadata() as meta:
|
||||
st.code(str(meta), language="xml")
|
||||
title = st.text_input("Title", value=meta.get('dc:title', ''))
|
||||
author = st.text_input("Author", value=meta.get('dc:creator', ''))
|
||||
keywords = st.text_input("Keywords", value=meta.get('dc:subject', ''))
|
||||
subject = st.text_input("Subject", value=meta.get('dc:description', ''))
|
||||
|
||||
|
||||
with st.expander("Optimization after OCR"):
|
||||
jpeg_quality = st.slider(
|
||||
"JPEG quality", min_value=0, max_value=100, value=75, key="jpeg_quality"
|
||||
)
|
||||
png_quality = st.slider(
|
||||
"PNG quality", min_value=0, max_value=100, value=75, key="png_quality"
|
||||
)
|
||||
jbig2_threshold = st.number_input(
|
||||
"JBIG2 threshold", value=0.85, key="jbig2_threshold"
|
||||
)
|
||||
|
||||
with st.expander("Advanced options"):
|
||||
jobs = st.slider(
|
||||
"Threads",
|
||||
min_value=1,
|
||||
max_value=os.cpu_count(),
|
||||
value=os.cpu_count(),
|
||||
key="threads",
|
||||
)
|
||||
max_image_mpixels = st.number_input(
|
||||
"Max image size",
|
||||
value=250.0,
|
||||
min_value=0.0,
|
||||
help="Maximum image size in megapixels",
|
||||
)
|
||||
rotate_pages_threshold = st.number_input(
|
||||
"Rotate pages threshold",
|
||||
value=DEFAULT_ROTATE_PAGES_THRESHOLD,
|
||||
min_value=0.0,
|
||||
max_value=1000.0,
|
||||
help="Threshold for automatic page rotation",
|
||||
)
|
||||
fast_web_view = st.number_input(
|
||||
"Fast web view",
|
||||
value=1.0,
|
||||
min_value=0.0,
|
||||
help="Linearize files above this size in MB",
|
||||
)
|
||||
continue_on_soft_render_error = st.checkbox(
|
||||
"Continue on soft render error", value=True
|
||||
)
|
||||
verbose_labels = ["quiet", "default", "debug", "debug_all"]
|
||||
verbose = st.selectbox(
|
||||
"Verbosity level",
|
||||
options=[-1, 0, 1, 2],
|
||||
index=1,
|
||||
format_func=partial(getitem, verbose_labels),
|
||||
)
|
||||
|
||||
if uploaded:
|
||||
args = []
|
||||
if mode and mode != 'normal':
|
||||
args.append(f"--{mode}")
|
||||
if language:
|
||||
args.append(f"--language={language}")
|
||||
if not uploaded.name.lower().endswith(".pdf") and image_dpi:
|
||||
args.append(f"--image-dpi={image_dpi}")
|
||||
if skip_big:
|
||||
args.append("--skip-big")
|
||||
if oversample:
|
||||
args.append(f"--oversample={oversample}")
|
||||
if rotate_pages:
|
||||
args.append("--rotate-pages")
|
||||
if deskew:
|
||||
args.append("--deskew")
|
||||
if clean:
|
||||
args.append("--clean")
|
||||
if clean_final:
|
||||
args.append("--clean-final")
|
||||
if remove_vectors:
|
||||
args.append("--remove-vectors")
|
||||
if output_type:
|
||||
args.append(f"--output-type={output_type}")
|
||||
if pdf_renderer:
|
||||
args.append(f"--pdf-renderer={pdf_renderer}")
|
||||
if optimize:
|
||||
args.append(f"--optimize={optimize}")
|
||||
if title:
|
||||
args.append(f"--title={title}")
|
||||
if author:
|
||||
args.append(f"--author={author}")
|
||||
if keywords:
|
||||
args.append(f"--keywords={keywords}")
|
||||
if subject:
|
||||
args.append(f"--subject={subject}")
|
||||
if pages:
|
||||
args.append(f"--pages={pages}")
|
||||
if max_image_mpixels:
|
||||
args.append(f"--max-image-mpixels={max_image_mpixels}")
|
||||
if rotate_pages_threshold:
|
||||
args.append(f"--rotate-pages-threshold={rotate_pages_threshold}")
|
||||
if fast_web_view:
|
||||
args.append(f"--fast-web-view={fast_web_view}")
|
||||
if continue_on_soft_render_error:
|
||||
args.append("--continue-on-soft-render-error")
|
||||
if verbose:
|
||||
args.append(f"--verbose={verbose}")
|
||||
if optimize > '0' and jpeg_quality:
|
||||
args.append(f"--jpeg-quality={jpeg_quality}")
|
||||
if optimize > '0' and png_quality:
|
||||
args.append(f"--png-quality={png_quality}")
|
||||
if jbig2_threshold:
|
||||
args.append(f"--jbig2-threshold={jbig2_threshold}")
|
||||
if jobs:
|
||||
args.append(f"--jobs={jobs}")
|
||||
with NamedTemporaryFile(delete=True, suffix=f"_{uploaded.name}") as input_file:
|
||||
input_file.write(uploaded.getvalue())
|
||||
input_file.flush()
|
||||
input_file.seek(0)
|
||||
args.append(str(input_file.name))
|
||||
with NamedTemporaryFile(delete=True, suffix=".pdf") as output_file:
|
||||
args.append(str(output_file.name))
|
||||
|
||||
st.session_state['running'] = (
|
||||
'run_button' in st.session_state and st.session_state.run_button
|
||||
)
|
||||
if st.button(
|
||||
"Run OCRmyPDF",
|
||||
disabled=st.session_state.get("running", False),
|
||||
key='run_button',
|
||||
):
|
||||
st.session_state['running'] = True
|
||||
args = [sys.executable, '-u', '-m', "ocrmypdf"] + args
|
||||
|
||||
proc = subprocess.Popen(
|
||||
args, stdout=subprocess.PIPE, stderr=subprocess.PIPE
|
||||
)
|
||||
with st.container(border=True):
|
||||
while proc.poll() is None:
|
||||
line = proc.stderr.readline()
|
||||
if line:
|
||||
st.html("<code>" + line.decode().strip() + "</code>")
|
||||
|
||||
if proc.returncode != 0:
|
||||
st.error(f"ocrmypdf failed with exit code {proc.returncode}")
|
||||
st.session_state['running'] = False
|
||||
st.stop()
|
||||
|
||||
if Path(output_file.name).stat().st_size == 0:
|
||||
st.error("No output PDF file was generated")
|
||||
st.stop()
|
||||
|
||||
st.download_button(
|
||||
label="Download output PDF",
|
||||
data=output_file.read(),
|
||||
file_name=uploaded.name,
|
||||
mime="application/pdf",
|
||||
)
|
||||
st.session_state['running'] = False
|
||||
+57
-15
@@ -1,12 +1,24 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-FileCopyrightText: 2016 findingorder <https://github.com/findingorder>
|
||||
# SPDX-FileCopyrightText: 2024 nilsro <https://github.com/nilsro>
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Example of using ocrmypdf as a library in a script.
|
||||
|
||||
This script will recursively search a directory for PDF files and run OCR on
|
||||
them. It will log the results. It runs OCR on every file, even if it already
|
||||
has text. OCRmyPDF will detect files that already have text.
|
||||
|
||||
You should edit this script to meet your needs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
# This script must be edited to meet your needs.
|
||||
import filecmp
|
||||
import logging
|
||||
import os
|
||||
import posixpath
|
||||
import shutil
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
@@ -15,32 +27,62 @@ import ocrmypdf
|
||||
# pylint: disable=logging-format-interpolation
|
||||
# pylint: disable=logging-not-lazy
|
||||
|
||||
script_dir = Path(__file__).parent
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
start_dir = Path(sys.argv[1])
|
||||
else:
|
||||
start_dir = Path('.')
|
||||
def filecompare(a, b):
|
||||
try:
|
||||
return filecmp.cmp(a, b, shallow=True)
|
||||
except FileNotFoundError:
|
||||
return False
|
||||
|
||||
|
||||
script_dir = Path(__file__).parent
|
||||
# set archive_dir to a path for backup original documents. Leave empty if not required.
|
||||
archive_dir = "/pdfbak"
|
||||
|
||||
start_dir = Path(sys.argv[1]) if len(sys.argv) > 1 else Path(".")
|
||||
|
||||
if len(sys.argv) > 2:
|
||||
log_file = Path(sys.argv[2])
|
||||
else:
|
||||
log_file = script_dir.with_name('ocr-tree.log')
|
||||
log_file = script_dir.with_name("ocr-tree.log")
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s %(message)s',
|
||||
format="%(asctime)s %(message)s",
|
||||
filename=log_file,
|
||||
filemode='a',
|
||||
filemode="a",
|
||||
)
|
||||
|
||||
logging.info(f"Start directory {start_dir}")
|
||||
|
||||
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
||||
|
||||
for filename in start_dir.glob("**/*.py"):
|
||||
for filename in start_dir.glob("**/*.pdf"):
|
||||
logging.info(f"Processing {filename}")
|
||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||
if result == ocrmypdf.ExitCode.already_done_ocr:
|
||||
logging.error("Skipped document because it already contained text")
|
||||
elif result == ocrmypdf.ExitCode.ok:
|
||||
if ocrmypdf.pdfa.file_claims_pdfa(filename)["pass"]:
|
||||
logging.info("Skipped document because it already contained text")
|
||||
else:
|
||||
archive_filename = archive_dir + str(filename)
|
||||
if len(archive_dir) > 0 and not filecompare(filename, archive_filename):
|
||||
logging.info(f"Archiving document to {archive_filename}")
|
||||
try:
|
||||
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||
except OSError:
|
||||
os.makedirs(posixpath.dirname(archive_filename))
|
||||
shutil.copy2(filename, posixpath.dirname(archive_filename))
|
||||
try:
|
||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||
logging.info(result)
|
||||
except ocrmypdf.exceptions.EncryptedPdfError:
|
||||
logging.info("Skipped document because it is encrypted")
|
||||
except ocrmypdf.exceptions.PriorOcrFoundError:
|
||||
logging.info("Skipped document because it already contained text")
|
||||
except ocrmypdf.exceptions.DigitalSignatureError:
|
||||
logging.info("Skipped document because it has a digital signature")
|
||||
except ocrmypdf.exceptions.TaggedPDFError:
|
||||
logging.info(
|
||||
"Skipped document because it does not need ocr as it is tagged"
|
||||
)
|
||||
except Exception:
|
||||
logging.error("Unhandled error occured")
|
||||
logging.info("OCR complete")
|
||||
logging.info(result)
|
||||
|
||||
@@ -0,0 +1,43 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Helper script for bisecting PDFs to find a page with an issue."""
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
|
||||
import pikepdf
|
||||
|
||||
if len(sys.argv) != 2:
|
||||
print(f"Usage: {sys.argv[0]} <input.pdf>")
|
||||
sys.exit(1)
|
||||
|
||||
with pikepdf.open(sys.argv[1]) as pdf:
|
||||
num_pages = len(pdf.pages)
|
||||
low = 0
|
||||
high = num_pages - 1
|
||||
while low <= high:
|
||||
mid = (low + high) // 2
|
||||
with pikepdf.new() as new_pdf:
|
||||
new_pdf.pages.extend(pdf.pages[low : mid + 1])
|
||||
new_pdf.save(f"bisect-issue-{low + 1}-{mid + 1}.pdf")
|
||||
print(f"Is bisect-issue-{low + 1}-{mid + 1}.pdf good or bad?", end=" ")
|
||||
while True:
|
||||
response = input().lower()
|
||||
if response == "good":
|
||||
low = mid + 1
|
||||
break
|
||||
elif response == "bad":
|
||||
high = mid - 1
|
||||
break
|
||||
else:
|
||||
print("Please respond with 'good' or 'bad'.")
|
||||
print(f"The issue is on page {low + 1} of the original PDF.")
|
||||
with pikepdf.new() as new_pdf:
|
||||
new_pdf.pages.extend(pdf.pages[low])
|
||||
new_pdf.save(f"bisect-issue-bad-{low + 1}.pdf")
|
||||
with pikepdf.new() as new_pdf:
|
||||
new_pdf.pages.extend(pdf.pages[:low])
|
||||
new_pdf.pages.extend(pdf.pages[low + 1 :])
|
||||
new_pdf.save(f"bisect-issue-good-{low + 1}.pdf")
|
||||
@@ -6,52 +6,56 @@ set -o errexit
|
||||
|
||||
__ocrmypdf_arguments()
|
||||
{
|
||||
local arguments="--help (show help message)
|
||||
--language (language(s) of the file to be OCRed)
|
||||
--image-dpi (assume this DPI if input image DPI is unknown)
|
||||
--output-type (select PDF output options)
|
||||
--sidecar (write OCR to text file)
|
||||
--version (print program version and exit)
|
||||
--jobs (how many worker processes to use)
|
||||
--quiet (suppress INFO messages)
|
||||
--verbose (set verbosity level)
|
||||
--title (set metadata)
|
||||
--author (set metadata)
|
||||
--subject (set metadata)
|
||||
--keywords (set metadata)
|
||||
--rotate-pages (rotate pages to correct orientation)
|
||||
--remove-background (attempt to remove background from pages)
|
||||
--deskew (fix small horizontal alignment skew)
|
||||
--clean (clean document images before OCR)
|
||||
--clean-final (clean document images and keep result)
|
||||
--unpaper-args (a quoted string of arguments to pass to unpaper)
|
||||
--oversample (oversample images to this DPI)
|
||||
--remove-vectors (don\'t send vector objects to OCR)
|
||||
--threshold (threshold images before OCR)
|
||||
--force-ocr (OCR documents that already have printable text)
|
||||
--skip-text (skip OCR on any pages that already contain text)
|
||||
--redo-ocr (redo OCR on any pages that seem to have OCR already)
|
||||
--skip-big (skip OCR on pages larger than this many MPixels)
|
||||
--optimize (select optimization level)
|
||||
--jpeg-quality (JPEG quality [0..100])
|
||||
--png-quality (PNG quality [0..100])
|
||||
--jbig2-lossy (enable lossy JBIG2 (see docs))
|
||||
--pages (apply OCR to only the specified pages)
|
||||
--max-image-mpixels (image decompression bomb threshold)
|
||||
--pdf-renderer (select PDF renderer options)
|
||||
--rotate-pages-threshold (page rotation confidence)
|
||||
--pdfa-image-compression (set PDF/A image compression options)
|
||||
--fast-web-view (if file size if above this amount in MB linearize PDF)
|
||||
--plugin (name of plugin to import)
|
||||
--keep-temporary-files (keep temporary files (debug)
|
||||
--tesseract-config (set custom tesseract config file)
|
||||
--tesseract-pagesegmode (set tesseract --psm)
|
||||
--tesseract-oem (set tesseract --oem)
|
||||
--tesseract-thresholding (set tesseract image thresholding)
|
||||
--tesseract-timeout (maximum number of seconds to wait for OCR)
|
||||
--user-words (specify location of user words file)
|
||||
--user-patterns (specify location of user patterns file)
|
||||
--no-progress-bar (disable the progress bar)
|
||||
local arguments="\
|
||||
--help (show help message)
|
||||
--language (language(s) of the file to be OCRed)
|
||||
--image-dpi (assume this DPI if input image DPI is unknown)
|
||||
--output-type (select PDF output options)
|
||||
--sidecar (write OCR to text file)
|
||||
--version (print program version and exit)
|
||||
--jobs (how many worker processes to use)
|
||||
--quiet (suppress INFO messages)
|
||||
--verbose (set verbosity level)
|
||||
--title (set metadata)
|
||||
--author (set metadata)
|
||||
--subject (set metadata)
|
||||
--keywords (set metadata)
|
||||
--rotate-pages (rotate pages to correct orientation)
|
||||
--remove-background (attempt to remove background from pages)
|
||||
--deskew (fix small horizontal alignment skew)
|
||||
--clean (clean document images before OCR)
|
||||
--clean-final (clean document images and keep result)
|
||||
--unpaper-args (a quoted string of arguments to pass to unpaper)
|
||||
--oversample (oversample images to this DPI)
|
||||
--remove-vectors (don\'t send vector objects to OCR)
|
||||
--threshold (threshold images before OCR)
|
||||
--force-ocr (OCR documents that already have printable text)
|
||||
--skip-text (skip OCR on any pages that already contain text)
|
||||
--redo-ocr (redo OCR on any pages that seem to have OCR already)
|
||||
--invalidate-digital-signatures (remove digital signatures from PDF)
|
||||
--skip-big (skip OCR on pages larger than this many MPixels)
|
||||
--optimize (select optimization level)
|
||||
--jpeg-quality (JPEG quality [0..100])
|
||||
--png-quality (PNG quality [0..100])
|
||||
--jbig2-lossy (enable lossy JBIG2 (see docs))
|
||||
--jbig2-threshold (set JBIG2 threshold (see docs))
|
||||
--pages (apply OCR to only the specified pages)
|
||||
--max-image-mpixels (image decompression bomb threshold)
|
||||
--pdf-renderer (select PDF renderer options)
|
||||
--rotate-pages-threshold (page rotation confidence)
|
||||
--pdfa-image-compression (set PDF/A image compression options)
|
||||
--fast-web-view (if file size if above this amount in MB linearize PDF)
|
||||
--plugin (name of plugin to import)
|
||||
--keep-temporary-files (keep temporary files (debug)
|
||||
--tesseract-config (set custom tesseract config file)
|
||||
--tesseract-pagesegmode (set tesseract --psm)
|
||||
--tesseract-oem (set tesseract --oem)
|
||||
--tesseract-thresholding (set tesseract image thresholding)
|
||||
--tesseract-timeout (maximum number of seconds to wait for OCR)
|
||||
--user-words (specify location of user words file)
|
||||
--user-patterns (specify location of user patterns file)
|
||||
--no-progress-bar (disable the progress bar)
|
||||
--color-conversion-strategy (select color conversion strategy)
|
||||
"
|
||||
|
||||
COMPREPLY=( $( compgen -W "$arguments" -- "$cur") )
|
||||
@@ -191,6 +195,20 @@ sauvola (use Sauvola thresholding)"
|
||||
fi
|
||||
}
|
||||
|
||||
__ocrmypdf_color-conversion-strategy()
|
||||
{
|
||||
local choices="LeaveColorUnchanged (default)
|
||||
CMYK (convert to CMYK)
|
||||
Gray (convert to grayscale)
|
||||
RGB (convert to RGB)
|
||||
UseDeviceIndependentColor (convert with device independent color)"
|
||||
|
||||
COMPREPLY=( $( compgen -W "$choices" -- "$cur") )
|
||||
# Remove description if only one completion exists
|
||||
if [[ ${#COMPREPLY[*]} -eq 1 ]]; then
|
||||
COMPREPLY=( ${COMPREPLY[0]%% *} )
|
||||
fi
|
||||
}
|
||||
|
||||
__ocrmypdf_check_previous()
|
||||
{
|
||||
@@ -250,6 +268,10 @@ __ocrmypdf_check_previous()
|
||||
_filedir
|
||||
return 0
|
||||
;;
|
||||
--color-conversion-strategy)
|
||||
__ocrmypdf_color-conversion-strategy
|
||||
return 0
|
||||
;;
|
||||
esac
|
||||
|
||||
return 1
|
||||
|
||||
@@ -14,8 +14,9 @@ complete -c ocrmypdf -s i -l clean-final -d "clean document images and keep resu
|
||||
complete -c ocrmypdf -l remove-vectors -d "don't send vector objects to OCR"
|
||||
|
||||
complete -c ocrmypdf -s f -l force-ocr -d "OCR documents that already have printable text"
|
||||
complete -c ocrmypdf -s s -l skip-ocr -d "skip OCR on pages that text, otherwise try OCR"
|
||||
complete -c ocrmypdf -s s -l skip-text -d "skip OCR on any pages that already contain text"
|
||||
complete -c ocrmypdf -l redo-ocr -d "redo OCR on any pages that seem to have OCR already"
|
||||
complete -c ocrmypdf -l invalidate-digital-signatures -d "invalidate digital signatures and allow OCR to proceed"
|
||||
|
||||
complete -c ocrmypdf -s k -l keep-temporary-files -d "keep temporary files (debug)"
|
||||
|
||||
@@ -83,6 +84,7 @@ complete -c ocrmypdf -x -l skip-big -d "skip OCR on pages larger than this many
|
||||
complete -c ocrmypdf -x -l jpeg-quality -d "JPEG quality [0..100]"
|
||||
complete -c ocrmypdf -x -l png-quality -d "PNG quality [0..100]"
|
||||
complete -c ocrmypdf -x -l jbig2-lossy -d "enable lossy JBIG2 (see docs)"
|
||||
complete -c ocrmypdf -x -l jbig2-threshold -d "JBIG2 compression threshold (see docs)"
|
||||
complete -c ocrmypdf -x -l max-image-mpixels -d "image decompression bomb threshold"
|
||||
complete -c ocrmypdf -x -l pages -d "apply OCR to only the specified pages"
|
||||
complete -c ocrmypdf -x -l tesseract-config -d "set custom tesseract config file"
|
||||
@@ -128,4 +130,27 @@ complete -c ocrmypdf -r -l user-words -d "specify location of user words file"
|
||||
complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file"
|
||||
complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF"
|
||||
|
||||
complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf; __fish_complete_suffix .PDF; __fish_complete_suffix .jpg; __fish_complete_suffix .png)"
|
||||
function __fish_ocrmypdf_color_conversion_strategy
|
||||
echo -e "LeaveColorUnchanged\t"(_ "do not convert color spaces (default)")
|
||||
echo -e "CMYK\t"(_ "convert all color spaces to CMYK")
|
||||
echo -e "Gray\t"(_ "convert all color spaces to grayscale")
|
||||
echo -e "RGB\t"(_ "convert all color spaces to RGB")
|
||||
echo -e "UseDeviceIndependentColor\t"(_ "convert all color spaces to ICC-based color spaces")
|
||||
end
|
||||
|
||||
complete -c ocrmypdf -x -l color-conversion-strategy -a '(__fish_ocrmypdf_color_conversion_strategy)' -d "set color conversion strategy"
|
||||
|
||||
function __fish_ocrmypdf_input_file_given
|
||||
set -l tokens (commandline -opc)
|
||||
for token in $tokens
|
||||
if string match -q -r '^-' -- $token
|
||||
continue
|
||||
end
|
||||
if test -f "$token"
|
||||
return 0
|
||||
end
|
||||
end
|
||||
return 1
|
||||
end
|
||||
|
||||
complete -c ocrmypdf -x -n 'not __fish_ocrmypdf_input_file_given' -a "(__fish_complete_suffix .pdf)" -d "input file"
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R Barlow: https://github.com/jbarlow83
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""
|
||||
An example of an OCRmyPDF plugin.
|
||||
"""An example of an OCRmyPDF plugin.
|
||||
|
||||
This plugin adds two new command line arguments
|
||||
--grayscale-ocr: converts the image to grayscale before performing OCR on it
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<component type="console-application">
|
||||
<id>io.ocrmypdf.ocrmypdf</id>
|
||||
|
||||
<name>OCRmyPDF</name>
|
||||
<summary>Adds an OCR text layer to scanned PDF files, allowing them to be searched</summary>
|
||||
|
||||
<developer id="io.ocrmypdf">
|
||||
<name>OCRmyPDF Developers</name>
|
||||
</developer>
|
||||
|
||||
<url type="homepage">https://github.com/ocrmypdf/ocrmypdf</url>
|
||||
<url type="bugtracker">https://github.com/ocrmypdf/OCRmyPDF/issues</url>
|
||||
|
||||
<content_rating type="oars-1.1" />
|
||||
|
||||
<metadata_license>CC0-1.0</metadata_license>
|
||||
<project_license>MPL-2.0</project_license>
|
||||
|
||||
<description>
|
||||
<ul>
|
||||
<li>Generates a searchable PDF/A file from a regular PDF</li>
|
||||
<li>Places OCR text accurately below the image to ease copy / paste</li>
|
||||
<li>Keeps the exact resolution of the original embedded images</li>
|
||||
<li>When possible, inserts OCR information as a lossless operation without disrupting any other content</li>
|
||||
<li>Optimizes PDF images, often producing files smaller than the input file If requested, deskews and/or cleans the image before performing OCR</li>
|
||||
<li>Validates input and output files</li>
|
||||
<li>Distributes work across all available CPU cores</li>
|
||||
<li>Uses Tesseract OCR engine to recognize more than 100 languages</li>
|
||||
<li>Keeps your private data private</li>
|
||||
<li>Scales properly to handle files with thousands of pages</li>
|
||||
<li>Battle-tested on millions of PDFs</li>
|
||||
</ul>
|
||||
</description>
|
||||
|
||||
<provides>
|
||||
<binary>ocrmypdf</binary>
|
||||
</provides>
|
||||
|
||||
<icon type="stock">io.ocrmypdf.ocrmypdf</icon>
|
||||
|
||||
<screenshots>
|
||||
<screenshot type="default">
|
||||
<image>https://raw.githubusercontent.com/ocrmypdf/OCRmyPDF/f7ad5f16bd0340b0b1803dada0c02f9f40542bd8/misc/flatpak/sample_screenshot.png</image>
|
||||
<caption>Sample usage of OCRmyPDF</caption>
|
||||
</screenshot>
|
||||
</screenshots>
|
||||
|
||||
<categories>
|
||||
<category>Office</category>
|
||||
<category>Utility</category>
|
||||
</categories>
|
||||
|
||||
<keywords>
|
||||
<keyword>ocr</keyword>
|
||||
<keyword>pdf</keyword>
|
||||
<keyword>tool</keyword>
|
||||
</keywords>
|
||||
|
||||
<releases>
|
||||
<release version="16.8.0" date="2025-01-05"/>
|
||||
</releases>
|
||||
</component>
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 166 KiB |
@@ -0,0 +1,128 @@
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Run OCRmyPDF on the same PDF with different options."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shlex
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from subprocess import check_output, run
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
import pikepdf
|
||||
import pymupdf
|
||||
import streamlit as st
|
||||
from lxml import etree
|
||||
from streamlit_pdf_viewer import pdf_viewer
|
||||
|
||||
|
||||
def do_column(label, suffix, d):
|
||||
cli = st.text_area(
|
||||
f"Command line arguments for {label}",
|
||||
key=f"args{suffix}",
|
||||
value="ocrmypdf {in_} {out}",
|
||||
)
|
||||
env_text = st.text_area(f"Environment variables for {label}", key=f"env{suffix}")
|
||||
env = os.environ.copy()
|
||||
for line in env_text.splitlines():
|
||||
if line:
|
||||
try:
|
||||
k, v = line.split("=", 1)
|
||||
except ValueError:
|
||||
st.error(f"Invalid environment variable: {line}")
|
||||
break
|
||||
env[k] = v
|
||||
args = shlex.split(
|
||||
cli.format(
|
||||
in_=os.path.join(d, "input.pdf"),
|
||||
out=os.path.join(d, f"output{suffix}.pdf"),
|
||||
)
|
||||
)
|
||||
with st.expander("Environment variables", expanded=bool(env_text.strip())):
|
||||
st.code('\n'.join(f"{k}={v}" for k, v in env.items()))
|
||||
st.code(shlex.join(args))
|
||||
return env, args
|
||||
|
||||
|
||||
def main():
|
||||
st.set_page_config(layout="wide")
|
||||
|
||||
st.title("OCRmyPDF Compare")
|
||||
st.write("Run OCRmyPDF on the same PDF with different options.")
|
||||
st.warning("This is a testing tool and is not intended for production use.")
|
||||
|
||||
uploaded_pdf = st.file_uploader("Upload a PDF", type=["pdf"])
|
||||
if uploaded_pdf is None:
|
||||
return
|
||||
|
||||
pdf_bytes = uploaded_pdf.read()
|
||||
|
||||
with pikepdf.open(BytesIO(pdf_bytes)) as p, TemporaryDirectory() as d:
|
||||
with st.expander("PDF Metadata"):
|
||||
with p.open_metadata() as meta:
|
||||
xml_txt = str(meta)
|
||||
parser = etree.XMLParser(remove_blank_text=True)
|
||||
tree = etree.fromstring(xml_txt, parser=parser)
|
||||
st.code(
|
||||
etree.tostring(tree, pretty_print=True).decode("utf-8"),
|
||||
language="xml",
|
||||
)
|
||||
st.write(p.docinfo)
|
||||
st.write("Number of pages:", len(p.pages))
|
||||
|
||||
col1, col2 = st.columns(2)
|
||||
with col1:
|
||||
env1, args1 = do_column("A", "1", d)
|
||||
with col2:
|
||||
env2, args2 = do_column("B", "2", d)
|
||||
|
||||
if not st.button("Execute and Compare"):
|
||||
return
|
||||
with st.spinner("Executing..."):
|
||||
Path(d, "input.pdf").write_bytes(pdf_bytes)
|
||||
run(args1, env=env1)
|
||||
run(args2, env=env2)
|
||||
|
||||
col1, col2 = st.columns(2)
|
||||
with col1:
|
||||
st.text(
|
||||
"Ghostscript version A: "
|
||||
+ check_output(
|
||||
["gs", "--version"],
|
||||
env=env1,
|
||||
text=True,
|
||||
)
|
||||
)
|
||||
with col2:
|
||||
st.text(
|
||||
"Ghostscript version B: "
|
||||
+ check_output(
|
||||
["gs", "--version"],
|
||||
env=env2,
|
||||
text=True,
|
||||
)
|
||||
)
|
||||
|
||||
doc1 = pymupdf.open(os.path.join(d, "output1.pdf"))
|
||||
doc2 = pymupdf.open(os.path.join(d, "output2.pdf"))
|
||||
for i, page1_2 in enumerate(zip(doc1, doc2, strict=False)):
|
||||
st.write(f"Page {i+1}")
|
||||
page1, page2 = page1_2
|
||||
col1, col2 = st.columns(2)
|
||||
with col1, st.container(border=True):
|
||||
st.write(page1.get_text())
|
||||
with col2, st.container(border=True):
|
||||
st.write(page2.get_text())
|
||||
|
||||
col1, col2 = st.columns(2)
|
||||
with col1, st.expander("PDF Viewer"):
|
||||
pdf_viewer(Path(d, "output1.pdf"))
|
||||
with col2, st.expander("PDF Viewer"):
|
||||
pdf_viewer(Path(d, "output2.pdf"))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,83 @@
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Compare two PDFs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
import pikepdf
|
||||
import pymupdf
|
||||
import streamlit as st
|
||||
from lxml import etree
|
||||
from streamlit_pdf_viewer import pdf_viewer
|
||||
|
||||
|
||||
def do_metadata(pdf):
|
||||
with pikepdf.open(pdf) as pdf:
|
||||
with pdf.open_metadata() as meta:
|
||||
xml_txt = str(meta)
|
||||
parser = etree.XMLParser(remove_blank_text=True)
|
||||
tree = etree.fromstring(xml_txt, parser=parser)
|
||||
st.code(
|
||||
etree.tostring(tree, pretty_print=True).decode("utf-8"),
|
||||
language="xml",
|
||||
)
|
||||
st.write(pdf.docinfo)
|
||||
st.write("Number of pages:", len(pdf.pages))
|
||||
|
||||
|
||||
def main():
|
||||
st.set_page_config(layout="wide")
|
||||
|
||||
st.title("PDF Compare")
|
||||
st.write("Compare two PDFs.")
|
||||
|
||||
col1, col2 = st.columns(2)
|
||||
with col1:
|
||||
uploaded_pdf1 = st.file_uploader("Upload a PDF", type=["pdf"], key='pdf1')
|
||||
with col2:
|
||||
uploaded_pdf2 = st.file_uploader("Upload a PDF", type=["pdf"], key='pdf2')
|
||||
if uploaded_pdf1 is None or uploaded_pdf2 is None:
|
||||
return
|
||||
|
||||
pdf_bytes1 = uploaded_pdf1.getvalue()
|
||||
pdf_bytes2 = uploaded_pdf2.getvalue()
|
||||
|
||||
with st.expander("PDF Metadata"):
|
||||
col1, col2 = st.columns(2)
|
||||
with col1:
|
||||
do_metadata(BytesIO(pdf_bytes1))
|
||||
with col2:
|
||||
do_metadata(BytesIO(pdf_bytes2))
|
||||
|
||||
with TemporaryDirectory() as d:
|
||||
Path(d, "1.pdf").write_bytes(pdf_bytes1)
|
||||
Path(d, "2.pdf").write_bytes(pdf_bytes2)
|
||||
|
||||
with st.expander("Text"):
|
||||
doc1 = pymupdf.open(os.path.join(d, "1.pdf"))
|
||||
doc2 = pymupdf.open(os.path.join(d, "2.pdf"))
|
||||
for i, page1_2 in enumerate(zip(doc1, doc2, strict=False)):
|
||||
st.write(f"Page {i+1}")
|
||||
page1, page2 = page1_2
|
||||
col1, col2 = st.columns(2)
|
||||
with col1, st.container(border=True):
|
||||
st.write(page1.get_text())
|
||||
with col2, st.container(border=True):
|
||||
st.write(page2.get_text())
|
||||
|
||||
with st.expander("PDF Viewer"):
|
||||
col1, col2 = st.columns(2)
|
||||
with col1:
|
||||
pdf_viewer(Path(d, "1.pdf"), key='pdf_viewer1', render_text=True)
|
||||
with col2:
|
||||
pdf_viewer(Path(d, "2.pdf"), key='pdf_viewer2', render_text=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,57 @@
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Compare text in PDFs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from subprocess import run
|
||||
from tempfile import NamedTemporaryFile
|
||||
from typing import Annotated
|
||||
|
||||
import cyclopts
|
||||
|
||||
app = cyclopts.App()
|
||||
|
||||
|
||||
@app.default
|
||||
def main(
|
||||
pdf1: Annotated[Path, cyclopts.Parameter()],
|
||||
pdf2: Annotated[Path, cyclopts.Parameter()],
|
||||
*,
|
||||
engine: Annotated[str, cyclopts.Parameter()] = 'pdftotext',
|
||||
):
|
||||
"""Compare text in PDFs."""
|
||||
with open(pdf1, 'rb') as f1, open(pdf2, 'rb') as f2:
|
||||
text1 = run(
|
||||
['pdftotext', '-layout', '-', '-'],
|
||||
stdin=f1,
|
||||
capture_output=True,
|
||||
check=True,
|
||||
)
|
||||
text2 = run(
|
||||
['pdftotext', '-layout', '-', '-'],
|
||||
stdin=f2,
|
||||
capture_output=True,
|
||||
check=True,
|
||||
)
|
||||
|
||||
with NamedTemporaryFile() as t1, NamedTemporaryFile() as t2:
|
||||
t1.write(text1.stdout)
|
||||
t1.flush()
|
||||
t2.write(text2.stdout)
|
||||
t2.flush()
|
||||
diff = run(
|
||||
['diff', '--color=always', '--side-by-side', t1.name, t2.name],
|
||||
capture_output=True,
|
||||
)
|
||||
run(['less', '-R'], input=diff.stdout, check=True)
|
||||
if text1.stdout.strip() != text2.stdout.strip():
|
||||
return 1
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
app()
|
||||
@@ -0,0 +1,31 @@
|
||||
<!-- SPDX-License-Identifier: CC-BY-SA-4.0 -->
|
||||
|
||||
To regenerate
|
||||
=============
|
||||
|
||||
Using asciinema and svg-term (`npm install -g svg-term-cli`).
|
||||
|
||||
Create `~/.config/asciinema/config` to disable prompt.
|
||||
|
||||
```
|
||||
[record]
|
||||
|
||||
command = fish --init-command 'alias fish_prompt="echo \>\ "'
|
||||
```
|
||||
|
||||
Run asciinema
|
||||
|
||||
```
|
||||
asciinema rec new_input.cast
|
||||
```
|
||||
|
||||
Re-record faster version with fewer pauses
|
||||
|
||||
```
|
||||
asciinema rec demo.cast -c "asciinema play new_input.cast --speed 2 --idle-time-limit 0.5"
|
||||
```
|
||||
|
||||
Convert to SVG
|
||||
```
|
||||
svg-term --in=misc/screencast/demo.cast --out=misc/screencast/demo.svg --window
|
||||
```
|
||||
@@ -0,0 +1,65 @@
|
||||
{"version": 2, "width": 131, "height": 24, "timestamp": 1687247006, "env": {"SHELL": "/usr/bin/fish", "TERM": "xterm-256color"}}
|
||||
[0.103649, "o", "\u001b[?2004h\u001b]7; \u0007"]
|
||||
[0.104223, "o", "\u001b]0;fish \u0007\u001b[30m\u001b(B\u001b[m\r> \u001b[K\r\u001b[C\u001b[C"]
|
||||
[0.604542, "o", "o\r\u001b[3C\b\u001b[38;2;255;0;0mo\r\u001b[3C\u001b[30m\u001b(B\u001b[m\u001b[38;2;85;85;85mcrmypdf multipage.pdf multipage_with_ocr.pdf\r\u001b[3C\u001b[30m\u001b(B\u001b[m"]
|
||||
[0.679571, "o", "\u001b[38;2;255;0;0mc\u001b[38;2;85;85;85mrmypdf multipage.pdf multipage_with_ocr.pdf\r\u001b[4C\u001b[30m\u001b(B\u001b[m"]
|
||||
[0.767271, "o", "\u001b[38;2;255;0;0mr\u001b[38;2;85;85;85mmypdf multipage.pdf multipage_with_ocr.pdf\r\u001b[5C\u001b[30m\u001b(B\u001b[m"]
|
||||
[0.814505, "o", "\u001b[38;2;255;0;0mm\u001b[38;2;85;85;85mypdf multipage.pdf multipage_with_ocr.pdf\r\u001b[6C\u001b[30m\u001b(B\u001b[m"]
|
||||
[0.938919, "o", "\u001b[38;2;255;0;0my\u001b[38;2;85;85;85mpdf multipage.pdf multipage_with_ocr.pdf\r\u001b[7C\u001b[30m\u001b(B\u001b[m"]
|
||||
[0.967347, "o", "\u001b[38;2;255;0;0mp\u001b[38;2;85;85;85mdf multipage.pdf multipage_with_ocr.pdf\r\u001b[8C\u001b[30m\u001b(B\u001b[m"]
|
||||
[1.009954, "o", "\u001b[38;2;255;0;0md\u001b[38;2;85;85;85mf multipage.pdf multipage_with_ocr.pdf\r\u001b[9C\u001b[30m\u001b(B\u001b[m"]
|
||||
[1.034488, "o", "\u001b[38;2;255;0;0mf\u001b[38;2;85;85;85m multipage.pdf multipage_with_ocr.pdf\r\u001b[10C\u001b[30m\u001b(B\u001b[m\b\b\b\b\b\b\b\b\u001b[38;2;0;95;215mocrmypdf\u001b[38;2;85;85;85m multipage.pdf multipage_with_ocr.pdf\r\u001b[10C\u001b[30m\u001b(B\u001b[m"]
|
||||
[1.069226, "o", "\u001b[38;2;0;95;215m \u001b[38;2;85;85;85mmultipage.pdf multipage_with_ocr.pdf\r\u001b[11C\u001b[30m\u001b(B\u001b[m\b \u001b[38;2;85;85;85mmultipage.pdf multipage_with_ocr.pdf\r\u001b[11C\u001b[30m\u001b(B\u001b[m"]
|
||||
[1.569682, "o", "-\u001b[K\r\u001b[12C\u001b[38;2;85;85;85m-version\r\u001b[12C\u001b[30m\u001b(B\u001b[m\b\u001b[38;2;0;175;255m-\u001b[38;2;85;85;85m-version\r\u001b[12C\u001b[30m\u001b(B\u001b[m"]
|
||||
[1.642096, "o", "\u001b[38;2;0;175;255m-\u001b[38;2;85;85;85mversion\r\u001b[13C\u001b[30m\u001b(B\u001b[m"]
|
||||
[1.71793, "o", "\u001b[38;2;0;175;255ms\u001b[30m\u001b(B\u001b[m\u001b[K\r\u001b[14C"]
|
||||
[1.771483, "o", "\u001b[38;2;0;175;255mk\r\u001b[15C\u001b[30m\u001b(B\u001b[m"]
|
||||
[1.864664, "o", "\u001b[38;2;0;175;255mi\r\u001b[16C\u001b[30m\u001b(B\u001b[m"]
|
||||
[1.876085, "o", "\u001b[38;2;0;175;255mp\r\u001b[17C\u001b[30m\u001b(B\u001b[m"]
|
||||
[2.092979, "o", "\u001b[38;2;0;175;255m-\r\u001b[18C\u001b[30m\u001b(B\u001b[m"]
|
||||
[2.138821, "o", "\u001b[38;2;0;175;255mt\r\u001b[19C\u001b[30m\u001b(B\u001b[m"]
|
||||
[2.18017, "o", "\u001b[38;2;0;175;255me\r\u001b[20C\u001b[30m\u001b(B\u001b[m"]
|
||||
[2.268222, "o", "\u001b[38;2;0;175;255mx\r\u001b[21C\u001b[30m\u001b(B\u001b[m"]
|
||||
[2.277031, "o", "\u001b[38;2;0;175;255mt\r\u001b[22C\u001b[30m\u001b(B\u001b[m"]
|
||||
[2.322469, "o", "\u001b[38;2;0;175;255m \r\u001b[23C\u001b[30m\u001b(B\u001b[m\b \r\u001b[23C"]
|
||||
[2.824696, "o", "m\r\u001b[24C\b\u001b[38;2;0;175;255m\u001b[4mm\r\u001b[24C\u001b[30m\u001b(B\u001b[m\u001b[38;2;85;85;85masks.pdf \r\u001b[24C\u001b[30m\u001b(B\u001b[m"]
|
||||
[2.923234, "o", "\u001b[38;2;0;175;255m\u001b[4mu\u001b[30m\u001b(B\u001b[m\u001b[K\r\u001b[25C\u001b[38;2;85;85;85mltipage.pdf \r\u001b[25C\u001b[30m\u001b(B\u001b[m"]
|
||||
[2.960685, "o", "\u001b[38;2;0;175;255m\u001b[4ml\u001b[38;2;85;85;85m\u001b[24mtipage.pdf \r\u001b[26C\u001b[30m\u001b(B\u001b[m"]
|
||||
[3.03365, "o", "\u001b[38;2;0;175;255m\u001b[4mt\u001b[38;2;85;85;85m\u001b[24mipage.pdf \r\u001b[27C\u001b[30m\u001b(B\u001b[m"]
|
||||
[3.479338, "o", "\u001b[38;2;0;175;255m\u001b[4mipage.pdf \r\u001b[37C\u001b[30m\u001b(B\u001b[m\b \r\u001b[37C"]
|
||||
[3.754818, "o", "m\r\u001b[38C\b\u001b[38;2;0;175;255m\u001b[4mm\r\u001b[38C\u001b[30m\u001b(B\u001b[m\u001b[38;2;85;85;85masks.pdf \r\u001b[38C\u001b[30m\u001b(B\u001b[m"]
|
||||
[3.873318, "o", "\u001b[38;2;0;175;255m\u001b[4mu\u001b[30m\u001b(B\u001b[m\u001b[K\r\u001b[39C\u001b[38;2;85;85;85mltipage.pdf \r\u001b[39C\u001b[30m\u001b(B\u001b[m"]
|
||||
[3.926829, "o", "\u001b[38;2;0;175;255m\u001b[4ml\u001b[38;2;85;85;85m\u001b[24mtipage.pdf \r\u001b[40C\u001b[30m\u001b(B\u001b[m"]
|
||||
[4.272251, "o", "\u001b[38;2;0;175;255m\u001b[4mtipage.pdf \r\u001b[51C\u001b[30m\u001b(B\u001b[m\b \r\u001b[51C"]
|
||||
[4.343464, "o", "\r\u001b[50C"]
|
||||
[4.416286, "o", "\r\u001b[49C"]
|
||||
[4.490574, "o", "\r\u001b[48C"]
|
||||
[4.564115, "o", "\r\u001b[47C"]
|
||||
[4.630398, "o", "\r\u001b[46C"]
|
||||
[4.76825, "o", "\u001b[38;2;0;175;255m\u001b[4m_.pd\u001b[30m\u001b(B\u001b[mf \r\u001b[47C\u001b[10D\u001b[38;2;0;175;255mmultipage_.pdf\u001b[30m\u001b(B\u001b[m \r\u001b[47C"]
|
||||
[5.012506, "o", "\u001b[38;2;0;175;255mo.pd\u001b[30m\u001b(B\u001b[mf \r\u001b[48C\u001b[3C\u001b[38;2;0;175;255mf\u001b[30m\u001b(B\u001b[m \r\u001b[48C"]
|
||||
[5.053615, "o", "\u001b[38;2;0;175;255mc.pd\u001b[30m\u001b(B\u001b[mf \r\u001b[49C\u001b[3C\u001b[38;2;0;175;255mf\u001b[30m\u001b(B\u001b[m \r\u001b[49C"]
|
||||
[5.103957, "o", "\u001b[38;2;0;175;255mr.pd\u001b[30m\u001b(B\u001b[mf \r\u001b[50C\u001b[3C\u001b[38;2;0;175;255mf\u001b[30m\u001b(B\u001b[m \r\u001b[50C"]
|
||||
[5.226183, "o", "\r\u001b[55C"]
|
||||
[5.728321, "o", "\r\n\u001b[30m\u001b(B\u001b[m\u001b[?2004l\u001b]0;ocrmypdf --skip-text multipage.pdf multipage_ocr.pdf /home/jb/src/ocrmypdf/tests/resources\u0007\u001b[30m\u001b(B\u001b[m\r"]
|
||||
[5.801032, "o", "\rScanning contents: 0%| | 0/6 [00:00<?, ?page/s]"]
|
||||
[5.802664, "o", "\rScanning contents: 100%|█████████████████████████████████████████████████████████████████████████| 6/6 [00:00<00:00, 1270.68page/s]\r\n"]
|
||||
[5.802747, "o", "Start processing 6 pages concurrently\r\n"]
|
||||
[5.803488, "o", "\rOCR: 0%| | 0.0/6.0 [00:00<?, ?page/s]"]
|
||||
[5.804896, "o", "\r \r 4 skipping all processing on this page\r\n\rOCR: 0%| | 0.0/6.0 [00:00<?, ?page/s]"]
|
||||
[5.896969, "o", "\rOCR: 25%|█████████████████████▎ | 1.5/6.0 [00:00<00:00, 8.12page/s]"]
|
||||
[6.170021, "o", "\rOCR: 42%|███████████████████████████████████▍ | 2.5/6.0 [00:00<00:01, 3.05page/s]"]
|
||||
[6.292338, "o", "\rOCR: 58%|█████████████████████████████████████████████████▌ | 3.5/6.0 [00:00<00:00, 3.39page/s]"]
|
||||
[6.586017, "o", "\rOCR: 75%|███████████████████████████████████████████████████████████████▊ | 4.5/6.0 [00:01<00:00, 2.49page/s]"]
|
||||
[7.087058, "o", "\rOCR: 92%|█████████████████████████████████████████████████████████████████████████████▉ | 5.5/6.0 [00:06<00:00, 1.98s/page]\rOCR: 100%|█████████████████████████████████████████████████████████████████████████████████████| 6.0/6.0 [00:06<00:00, 1.09s/page]\r\nPostprocessing...\r\n"]
|
||||
[7.104927, "o", "\rPDF/A conversion: 0%| | 0/6 [00:00<?, ?page/s]"]
|
||||
[7.607392, "o", "\rPDF/A conversion: 50%|██████████████████████████████████████ | 3/6 [00:01<00:01, 1.61page/s]"]
|
||||
[7.653781, "o", "\rPDF/A conversion: 83%|███████████████████████████████████████████████████████████████▎ | 5/6 [00:01<00:00, 2.90page/s]"]
|
||||
[7.774532, "o", "\rPDF/A conversion: 100%|████████████████████████████████████████████████████████████████████████████| 6/6 [00:02<00:00, 2.71page/s]\r\n"]
|
||||
[7.778252, "o", "\u001b[33mSome input metadata could not be copied because it is not permitted in PDF/A. You may wish to examine the output PDF's XMP metadata.\u001b[0m\r\n"]
|
||||
[8.280789, "o", "\rRecompressing JPEGs: 0image [00:00, ?image/s]\rRecompressing JPEGs: 0image [00:00, ?image/s]\r\n\rDeflating JPEGs: 0%| | 0/4 [00:00<?, ?image/s]\rDeflating JPEGs: 100%|███████████████████████████████████████████████████████████████████████████| 4/4 [00:00<00:00, 238.28image/s]\r\n"]
|
||||
[8.28149, "o", "\rJBIG2: 0item [00:00, ?item/s]\rJBIG2: 0item [00:00, ?item/s]\r\n"]
|
||||
[8.289998, "o", "Image optimization ratio: 1.01 savings: 1.3%\r\nTotal file size ratio: 1.02 savings: 1.6%\r\n"]
|
||||
[8.291209, "o", "Output file is a PDF/A-2b (as expected)\r\n"]
|
||||
[8.361316, "o", "\u001b[2m⏎\u001b(B\u001b[m \r⏎ \r\u001b[K\u001b[?2004h\u001b]0;fish /home/jb/src/ocrmypdf/tests/resources\u0007\u001b[30m\u001b(B\u001b[m> \u001b[K\r\u001b[C\u001b[C"]
|
||||
[8.862206, "o", "\r\n\u001b[30m\u001b(B\u001b[m\u001b[30m\u001b(B\u001b[m\u001b[?2004l"]
|
||||
File diff suppressed because one or more lines are too long
|
After Width: | Height: | Size: 29 KiB |
+5
-4
@@ -2,7 +2,7 @@
|
||||
# SPDX-FileCopyrightText: 2017 Enantiomerie
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Example OCRmyPDF for Synology NAS"""
|
||||
"""Example OCRmyPDF for Synology NAS."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -53,9 +53,10 @@ for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||
]
|
||||
logging.info(cmd)
|
||||
full_path_ocr = os.path.join(dir_name, filename_ocr)
|
||||
with open(filename, 'rb') as input_file, open(
|
||||
full_path_ocr, 'wb'
|
||||
) as output_file:
|
||||
with (
|
||||
open(filename, 'rb') as input_file,
|
||||
open(full_path_ocr, 'wb') as output_file,
|
||||
):
|
||||
proc = subprocess.run(
|
||||
cmd,
|
||||
stdin=input_file,
|
||||
|
||||
+230
-81
@@ -3,106 +3,130 @@
|
||||
# SPDX-FileCopyrightText: 2020 James R Barlow <https://github.com/jbarlow83>
|
||||
# SPDX-License-Identifier: MIT
|
||||
|
||||
"""Watch a directory for new PDFs and OCR them."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime as dt
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import Annotated, Any
|
||||
|
||||
import cyclopts
|
||||
import pikepdf
|
||||
from dotenv import load_dotenv
|
||||
from watchdog.events import PatternMatchingEventHandler
|
||||
from watchdog.observers import Observer
|
||||
from watchdog.observers.polling import PollingObserver
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
load_dotenv()
|
||||
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
|
||||
|
||||
def getenv_bool(name: str, default: str = 'False'):
|
||||
return os.getenv(name, default).lower() in ('true', 'yes', 'y', '1')
|
||||
|
||||
|
||||
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
|
||||
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
|
||||
ARCHIVE_DIRECTORY = os.getenv('OCR_ARCHIVE_DIRECTORY', '/processed')
|
||||
OUTPUT_DIRECTORY_YEAR_MONTH = getenv_bool('OCR_OUTPUT_DIRECTORY_YEAR_MONTH')
|
||||
ON_SUCCESS_DELETE = getenv_bool('OCR_ON_SUCCESS_DELETE')
|
||||
ON_SUCCESS_ARCHIVE = getenv_bool('OCR_ON_SUCCESS_ARCHIVE')
|
||||
DESKEW = getenv_bool('OCR_DESKEW')
|
||||
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
||||
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
||||
USE_POLLING = getenv_bool('OCR_USE_POLLING')
|
||||
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
|
||||
PATTERNS = ['*.pdf', '*.PDF']
|
||||
app = cyclopts.App(name="ocrmypdf-watcher")
|
||||
|
||||
log = logging.getLogger('ocrmypdf-watcher')
|
||||
|
||||
|
||||
def get_output_dir(root, basename):
|
||||
if OUTPUT_DIRECTORY_YEAR_MONTH:
|
||||
today = datetime.today()
|
||||
output_directory_year_month = (
|
||||
Path(root) / str(today.year) / f'{today.month:02d}'
|
||||
)
|
||||
class LoggingLevelEnum(str, Enum):
|
||||
"""Enum for logging levels."""
|
||||
|
||||
DEBUG = "DEBUG"
|
||||
INFO = "INFO"
|
||||
WARNING = "WARNING"
|
||||
ERROR = "ERROR"
|
||||
CRITICAL = "CRITICAL"
|
||||
|
||||
|
||||
def get_output_path(root: Path, basename: str, output_dir_year_month: bool) -> Path:
|
||||
assert '/' not in basename, "basename must not contain '/'"
|
||||
if output_dir_year_month:
|
||||
today = dt.datetime.today()
|
||||
output_directory_year_month = root / str(today.year) / f'{today.month:02d}'
|
||||
if not output_directory_year_month.exists():
|
||||
output_directory_year_month.mkdir(parents=True, exist_ok=True)
|
||||
output_path = Path(output_directory_year_month) / basename
|
||||
output_path = Path(output_directory_year_month) / Path(basename).with_suffix(
|
||||
'.pdf'
|
||||
)
|
||||
else:
|
||||
output_path = Path(OUTPUT_DIRECTORY) / basename
|
||||
output_path = root / Path(basename).with_suffix('.pdf')
|
||||
return output_path
|
||||
|
||||
|
||||
def wait_for_file_ready(file_path):
|
||||
def wait_for_file_ready(
|
||||
file_path: Path, poll_new_file_seconds: int, retries_loading_file: int
|
||||
):
|
||||
# This loop waits to make sure that the file is completely loaded on
|
||||
# disk before attempting to read. Docker sometimes will publish the
|
||||
# watchdog event before the file is actually fully on disk, causing
|
||||
# pikepdf to fail.
|
||||
|
||||
retries = 5
|
||||
while retries:
|
||||
tries = retries_loading_file + 1
|
||||
while tries:
|
||||
try:
|
||||
pdf = pikepdf.open(file_path)
|
||||
except (FileNotFoundError, pikepdf.PdfError) as e:
|
||||
with pikepdf.Pdf.open(file_path) as pdf:
|
||||
log.debug(f"{file_path} ready with {pdf.pages} pages")
|
||||
return True
|
||||
except (FileNotFoundError, OSError) as e:
|
||||
log.info(f"File {file_path} is not ready yet")
|
||||
log.debug("Exception was", exc_info=e)
|
||||
time.sleep(POLL_NEW_FILE_SECONDS)
|
||||
retries -= 1
|
||||
else:
|
||||
pdf.close()
|
||||
return True
|
||||
time.sleep(poll_new_file_seconds)
|
||||
tries -= 1
|
||||
except pikepdf.PdfError as e:
|
||||
log.info(f"File {file_path} is not full written yet")
|
||||
log.debug("Exception was", exc_info=e)
|
||||
time.sleep(poll_new_file_seconds)
|
||||
tries -= 1
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def execute_ocrmypdf(file_path):
|
||||
file_path = Path(file_path)
|
||||
output_path = get_output_dir(OUTPUT_DIRECTORY, file_path.name)
|
||||
def execute_ocrmypdf(
|
||||
*,
|
||||
file_path: Path,
|
||||
archive_dir: Path,
|
||||
output_dir: Path,
|
||||
ocrmypdf_kwargs: dict[str, Any],
|
||||
on_success_delete: bool,
|
||||
on_success_archive: bool,
|
||||
poll_new_file_seconds: int,
|
||||
retries_loading_file: int,
|
||||
output_dir_year_month: bool,
|
||||
):
|
||||
output_path = get_output_path(output_dir, file_path.name, output_dir_year_month)
|
||||
|
||||
log.info("-" * 20)
|
||||
log.info(f'New file: {file_path}. Waiting until fully loaded...')
|
||||
if not wait_for_file_ready(file_path):
|
||||
log.info(f'New file: {file_path}. Waiting until fully written...')
|
||||
if not wait_for_file_ready(file_path, poll_new_file_seconds, retries_loading_file):
|
||||
log.info(f"Gave up waiting for {file_path} to become ready")
|
||||
return
|
||||
log.info(f'Attempting to OCRmyPDF to: {output_path}')
|
||||
|
||||
log.debug(
|
||||
f'OCRmyPDF input_file={file_path} output_file={output_path} '
|
||||
f'kwargs: {ocrmypdf_kwargs}'
|
||||
)
|
||||
exit_code = ocrmypdf.ocr(
|
||||
input_file=file_path,
|
||||
output_file=output_path,
|
||||
deskew=DESKEW,
|
||||
**OCR_JSON_SETTINGS,
|
||||
ocrmypdf.OcrOptions(
|
||||
input_file=file_path,
|
||||
output_file=output_path,
|
||||
**ocrmypdf_kwargs,
|
||||
)
|
||||
)
|
||||
if exit_code == 0:
|
||||
if ON_SUCCESS_DELETE:
|
||||
if on_success_delete:
|
||||
log.info(f'OCR is done. Deleting: {file_path}')
|
||||
file_path.unlink()
|
||||
elif ON_SUCCESS_ARCHIVE:
|
||||
log.info(f'OCR is done. Archiving {file_path.name} to {ARCHIVE_DIRECTORY}')
|
||||
shutil.move(file_path, f'{ARCHIVE_DIRECTORY}/{file_path.name}')
|
||||
elif on_success_archive:
|
||||
log.info(f'OCR is done. Archiving {file_path.name} to {archive_dir}')
|
||||
shutil.move(file_path, f'{archive_dir}/{file_path.name}')
|
||||
else:
|
||||
log.info('OCR is done')
|
||||
else:
|
||||
@@ -110,60 +134,185 @@ def execute_ocrmypdf(file_path):
|
||||
|
||||
|
||||
class HandleObserverEvent(PatternMatchingEventHandler):
|
||||
def __init__( # noqa: D107
|
||||
self,
|
||||
patterns=None,
|
||||
ignore_patterns=None,
|
||||
ignore_directories=False,
|
||||
case_sensitive=False,
|
||||
settings=None,
|
||||
):
|
||||
super().__init__(
|
||||
patterns=patterns,
|
||||
ignore_patterns=ignore_patterns,
|
||||
ignore_directories=ignore_directories,
|
||||
case_sensitive=case_sensitive,
|
||||
)
|
||||
self._settings = settings if settings else {}
|
||||
|
||||
def on_any_event(self, event):
|
||||
if event.event_type in ['created']:
|
||||
execute_ocrmypdf(event.src_path)
|
||||
execute_ocrmypdf(file_path=Path(event.src_path), **self._settings)
|
||||
|
||||
|
||||
def main():
|
||||
@app.default
|
||||
def main(
|
||||
input_dir: Annotated[
|
||||
Path,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_INPUT_DIRECTORY',
|
||||
),
|
||||
] = Path('/input'),
|
||||
output_dir: Annotated[
|
||||
Path,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_OUTPUT_DIRECTORY',
|
||||
),
|
||||
] = Path('/output'),
|
||||
archive_dir: Annotated[
|
||||
Path,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_ARCHIVE_DIRECTORY',
|
||||
),
|
||||
] = Path('/processed'),
|
||||
*,
|
||||
output_dir_year_month: Annotated[
|
||||
bool,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_OUTPUT_DIRECTORY_YEAR_MONTH',
|
||||
help='Create a subdirectory in the output directory for each year/month',
|
||||
),
|
||||
] = False,
|
||||
on_success_delete: Annotated[
|
||||
bool,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_ON_SUCCESS_DELETE',
|
||||
help='Delete the input file after successful OCR',
|
||||
),
|
||||
] = False,
|
||||
on_success_archive: Annotated[
|
||||
bool,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_ON_SUCCESS_ARCHIVE',
|
||||
help='Archive the input file after successful OCR',
|
||||
),
|
||||
] = False,
|
||||
deskew: Annotated[
|
||||
bool,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_DESKEW',
|
||||
help='Deskew the input file before OCR',
|
||||
),
|
||||
] = False,
|
||||
ocr_json_settings: Annotated[
|
||||
str | None,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_JSON_SETTINGS',
|
||||
help='JSON settings to pass to OCRmyPDF (JSON string or file path)',
|
||||
),
|
||||
] = None,
|
||||
poll_new_file_seconds: Annotated[
|
||||
int,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_POLL_NEW_FILE_SECONDS',
|
||||
help='Seconds to wait before polling a new file',
|
||||
),
|
||||
] = 1,
|
||||
use_polling: Annotated[
|
||||
bool,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_USE_POLLING',
|
||||
help='Use polling instead of filesystem events',
|
||||
),
|
||||
] = False,
|
||||
retries_loading_file: Annotated[
|
||||
int,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_RETRIES_LOADING_FILE',
|
||||
help='Number of times to retry loading a file before giving up',
|
||||
),
|
||||
] = 5,
|
||||
loglevel: Annotated[
|
||||
LoggingLevelEnum,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_LOGLEVEL',
|
||||
help='Logging level',
|
||||
),
|
||||
] = LoggingLevelEnum.INFO,
|
||||
patterns: Annotated[
|
||||
str,
|
||||
cyclopts.Parameter(
|
||||
env_var='OCR_PATTERNS',
|
||||
help='File patterns to watch',
|
||||
),
|
||||
] = '*.pdf,*.PDF',
|
||||
):
|
||||
ocrmypdf.configure_logging(
|
||||
verbosity=(
|
||||
ocrmypdf.Verbosity.default
|
||||
if LOGLEVEL != 'DEBUG'
|
||||
if loglevel != LoggingLevelEnum.DEBUG
|
||||
else ocrmypdf.Verbosity.debug
|
||||
),
|
||||
manage_root_logger=True,
|
||||
)
|
||||
log.setLevel(LOGLEVEL)
|
||||
log.setLevel(loglevel.value)
|
||||
log.info(
|
||||
f"Starting OCRmyPDF watcher with config:\n"
|
||||
f"Input Directory: {INPUT_DIRECTORY}\n"
|
||||
f"Output Directory: {OUTPUT_DIRECTORY}\n"
|
||||
f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}\n"
|
||||
f"Archive Directory: {ARCHIVE_DIRECTORY}"
|
||||
f"Input Directory: {input_dir}\n"
|
||||
f"Output Directory: {output_dir}\n"
|
||||
f"Output Directory Year & Month: {output_dir_year_month}\n"
|
||||
f"Archive Directory: {archive_dir}"
|
||||
)
|
||||
log.debug(
|
||||
f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n"
|
||||
f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n"
|
||||
f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n"
|
||||
f"ARCHIVE_DIRECTORY: {ARCHIVE_DIRECTORY}\n"
|
||||
f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n"
|
||||
f"ON_SUCCESS_ARCHIVE: {ON_SUCCESS_ARCHIVE}\n"
|
||||
f"DESKEW: {DESKEW}\n"
|
||||
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
||||
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
||||
f"USE_POLLING: {USE_POLLING}\n"
|
||||
f"LOGLEVEL: {LOGLEVEL}"
|
||||
log.info(
|
||||
f"INPUT_DIRECTORY: {input_dir}\n"
|
||||
f"OUTPUT_DIRECTORY: {output_dir}\n"
|
||||
f"ARCHIVE_DIRECTORY: {archive_dir}\n"
|
||||
f"OUTPUT_DIRECTORY_YEAR_MONTH: {output_dir_year_month}\n"
|
||||
f"ON_SUCCESS_DELETE: {on_success_delete}\n"
|
||||
f"ON_SUCCESS_ARCHIVE: {on_success_archive}\n"
|
||||
f"DESKEW: {deskew}\n"
|
||||
f"ARGS: {ocr_json_settings}\n"
|
||||
f"POLL_NEW_FILE_SECONDS: {poll_new_file_seconds}\n"
|
||||
f"RETRIES_LOADING_FILE: {retries_loading_file}\n"
|
||||
f"USE_POLLING: {use_polling}\n"
|
||||
f"LOGLEVEL: {loglevel.value}"
|
||||
)
|
||||
|
||||
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS:
|
||||
log.error('OCR_JSON_SETTINGS should not specify input file or output file')
|
||||
if ocr_json_settings and Path(ocr_json_settings).exists():
|
||||
json_settings = json.loads(Path(ocr_json_settings).read_text())
|
||||
else:
|
||||
json_settings = json.loads(ocr_json_settings or '{}')
|
||||
|
||||
if 'input_file' in json_settings or 'output_file' in json_settings:
|
||||
log.error(
|
||||
'OCR_JSON_SETTINGS (--ocr-json-settings) may not specify input/output file'
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
handler = HandleObserverEvent(patterns=PATTERNS)
|
||||
if USE_POLLING:
|
||||
observer = PollingObserver()
|
||||
else:
|
||||
observer = Observer()
|
||||
observer.schedule(handler, INPUT_DIRECTORY, recursive=True)
|
||||
handler = HandleObserverEvent(
|
||||
patterns=patterns.split(','),
|
||||
settings={
|
||||
'archive_dir': archive_dir,
|
||||
'output_dir': output_dir,
|
||||
'ocrmypdf_kwargs': json_settings | {'deskew': deskew},
|
||||
'on_success_delete': on_success_delete,
|
||||
'on_success_archive': on_success_archive,
|
||||
'poll_new_file_seconds': poll_new_file_seconds,
|
||||
'retries_loading_file': retries_loading_file,
|
||||
'output_dir_year_month': output_dir_year_month,
|
||||
},
|
||||
)
|
||||
observer = PollingObserver() if use_polling else Observer()
|
||||
observer.schedule(handler, input_dir, recursive=True)
|
||||
observer.start()
|
||||
print(f"Watching {input_dir} for new PDFs. Press Ctrl+C to exit.")
|
||||
try:
|
||||
while True:
|
||||
time.sleep(1)
|
||||
time.sleep(30)
|
||||
except KeyboardInterrupt:
|
||||
observer.stop()
|
||||
observer.join()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
app()
|
||||
|
||||
Regular → Executable
+23
-99
@@ -1,107 +1,31 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-FileCopyrightText: 2019 James R. Barlow
|
||||
#!/usr/bin/env python
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: AGPL-3.0-or-later
|
||||
|
||||
"""This is a simple web service/HTTP wrapper for OCRmyPDF
|
||||
|
||||
This may be more convenient than the command line tool for some Docker users.
|
||||
Note that OCRmyPDF uses Ghostscript, which is licensed under AGPLv3+. While
|
||||
OCRmyPDF is under GPLv3, this file is distributed under the Affero GPLv3+ license,
|
||||
to emphasize that SaaS deployments should make sure they comply with
|
||||
Ghostscript's license as well as OCRmyPDF's.
|
||||
"""
|
||||
"""Run the OCRmyPDF web service."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shlex
|
||||
from subprocess import PIPE, run
|
||||
from tempfile import TemporaryDirectory
|
||||
import sys
|
||||
|
||||
from flask import Flask, Response, request, send_from_directory
|
||||
from werkzeug.utils import secure_filename
|
||||
try:
|
||||
import streamlit # noqa: F401
|
||||
except ImportError:
|
||||
raise ImportError(
|
||||
'You need to install streamlit in the Python environment '
|
||||
'to run the web service.\n'
|
||||
) from None
|
||||
|
||||
app = Flask(__name__)
|
||||
app.secret_key = "secret"
|
||||
app.config['MAX_CONTENT_LENGTH'] = 50_000_000
|
||||
app.config.from_envvar("OCRMYPDF_WEBSERVICE_SETTINGS", silent=True)
|
||||
|
||||
ALLOWED_EXTENSIONS = {"pdf"}
|
||||
|
||||
|
||||
def allowed_file(filename):
|
||||
return "." in filename and filename.rsplit(".", 1)[1].lower() in ALLOWED_EXTENSIONS
|
||||
|
||||
|
||||
def do_ocrmypdf(file):
|
||||
uploaddir = TemporaryDirectory(prefix="ocrmypdf-upload")
|
||||
downloaddir = TemporaryDirectory(prefix="ocrmypdf-download")
|
||||
|
||||
filename = secure_filename(file.filename)
|
||||
up_file = os.path.join(uploaddir.name, filename)
|
||||
file.save(up_file)
|
||||
|
||||
down_file = os.path.join(downloaddir.name, filename)
|
||||
|
||||
cmd_args = [arg for arg in shlex.split(request.form["params"])]
|
||||
if "--sidecar" in cmd_args:
|
||||
return Response("--sidecar not supported", 501, mimetype='text/plain')
|
||||
|
||||
ocrmypdf_args = ["ocrmypdf", *cmd_args, up_file, down_file]
|
||||
proc = run(ocrmypdf_args, capture_output=True, encoding="utf-8")
|
||||
if proc.returncode != 0:
|
||||
stderr = proc.stderr
|
||||
return Response(stderr, 400, mimetype='text/plain')
|
||||
|
||||
return send_from_directory(downloaddir.name, filename)
|
||||
|
||||
|
||||
@app.route("/", methods=["GET", "POST"])
|
||||
def upload_file():
|
||||
if request.method == "POST":
|
||||
if "file" not in request.files:
|
||||
return Response("No file in POST", 400, mimetype='text/plain')
|
||||
file = request.files["file"]
|
||||
if file.filename == "":
|
||||
return Response("Empty filename", 400, mimetype='text/plain')
|
||||
if not allowed_file(file.filename):
|
||||
return Response("Invalid filename", 400, mimetype='text/plain')
|
||||
if file and allowed_file(file.filename):
|
||||
return do_ocrmypdf(file)
|
||||
return Response("Some other problem", 400, mimetype='text/plain')
|
||||
|
||||
return """
|
||||
<!doctype html>
|
||||
<title>OCRmyPDF webservice</title>
|
||||
<h1>Upload a PDF (debug UI)</h1>
|
||||
<form method=post enctype=multipart/form-data>
|
||||
<label for="args">Command line parameters</label>
|
||||
<input type=textbox name=params>
|
||||
<label for="file">File to upload</label>
|
||||
<input type=file name=file>
|
||||
<input type=submit value=Upload>
|
||||
</form>
|
||||
<h4>Notice</h2>
|
||||
<div style="font-size: 70%; max-width: 34em;">
|
||||
<p>This is a webservice wrapper for OCRmyPDF.</p>
|
||||
<p>Copyright 2019 James R. Barlow</p>
|
||||
<p>This program is free software: you can redistribute it and/or modify
|
||||
it under the terms of the GNU Affero General Public License as published by
|
||||
the Free Software Foundation, either version 3 of the License, or
|
||||
(at your option) any later version.
|
||||
</p>
|
||||
<p>This program is distributed in the hope that it will be useful,
|
||||
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
GNU General Public License for more details.
|
||||
</p>
|
||||
<p>
|
||||
You should have received a copy of the GNU Affero General Public License
|
||||
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
</p>
|
||||
</div>
|
||||
"""
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app.run(host='0.0.0.0', port=5000)
|
||||
if __name__ == '__main__':
|
||||
os.execvp(
|
||||
sys.executable,
|
||||
[
|
||||
sys.executable,
|
||||
'-m',
|
||||
'streamlit',
|
||||
'run',
|
||||
'misc/_webservice.py',
|
||||
*sys.argv[1:],
|
||||
],
|
||||
)
|
||||
|
||||
+108
-109
@@ -1,126 +1,78 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
[build-system]
|
||||
requires = [
|
||||
"setuptools >= 61",
|
||||
"setuptools_scm[toml] >= 7.0.5",
|
||||
"wheel"
|
||||
]
|
||||
build-backend = "setuptools.build_meta"
|
||||
requires = ["hatchling", "hatch-vcs"]
|
||||
build-backend = "hatchling.build"
|
||||
|
||||
[project]
|
||||
name = "ocrmypdf"
|
||||
dynamic = ["version"]
|
||||
description = "OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched"
|
||||
readme = "README.md"
|
||||
license = {text = "MPL-2.0"}
|
||||
requires-python = ">=3.8"
|
||||
license = "MPL-2.0"
|
||||
requires-python = ">=3.11"
|
||||
dependencies = [
|
||||
"Pillow>=8.2.0",
|
||||
"coloredlogs>=14.0",
|
||||
"deprecation>=2.1.0",
|
||||
"img2pdf>=0.3.0", # pure Python
|
||||
"fpdf2>=2.8.0",
|
||||
"img2pdf>=0.5",
|
||||
"packaging>=20",
|
||||
"pdfminer.six>=20201018",
|
||||
"pikepdf>=5.0.1",
|
||||
"pluggy>=0.13.0",
|
||||
"reportlab>=3.5.66",
|
||||
"tqdm>=4",
|
||||
"importlib-resources>=5;python_version<'3.9'", # until Python 3.9
|
||||
"typing-extensions>=4;python_version<'3.10'",
|
||||
"pdfminer.six>=20220319",
|
||||
"pi-heif", # Heif image format - maintainers: if this is removed, it will NOT break
|
||||
"pikepdf>=10",
|
||||
"Pillow>=10.0.1",
|
||||
"pluggy>=1",
|
||||
"pydantic>=2.12.5",
|
||||
"pypdfium2>=5.0.0",
|
||||
"rich>=13",
|
||||
"uharfbuzz>=0.53.2",
|
||||
]
|
||||
authors = [{name = "James R. Barlow", email="james@purplerock.ca"}]
|
||||
authors = [{ name = "James R. Barlow", email = "james@purplerock.ca" }]
|
||||
classifiers = [
|
||||
"Development Status :: 5 - Production/Stable",
|
||||
"Environment :: Console",
|
||||
"Intended Audience :: End Users/Desktop",
|
||||
"Intended Audience :: Science/Research",
|
||||
"Intended Audience :: System Administrators",
|
||||
"License :: OSI Approved :: Mozilla Public License 2.0 (MPL 2.0)",
|
||||
"Operating System :: MacOS :: MacOS X",
|
||||
"Operating System :: Microsoft :: Windows :: Windows 10",
|
||||
"Operating System :: MacOS",
|
||||
"Operating System :: Microsoft :: Windows",
|
||||
"Operating System :: POSIX",
|
||||
"Operating System :: POSIX :: BSD",
|
||||
"Operating System :: POSIX :: Linux",
|
||||
"Programming Language :: Python :: 3",
|
||||
"Programming Language :: Python :: 3 :: Only",
|
||||
"Programming Language :: Python :: 3.8",
|
||||
"Programming Language :: Python :: 3.9",
|
||||
"Programming Language :: Python :: 3.10",
|
||||
"Topic :: Scientific/Engineering :: Image Recognition",
|
||||
"Topic :: Text Processing :: Indexing",
|
||||
"Topic :: Text Processing :: Linguistic",
|
||||
]
|
||||
keywords = [
|
||||
"PDF",
|
||||
"OCR",
|
||||
"optical character recognition",
|
||||
"PDF/A",
|
||||
"scanning",
|
||||
]
|
||||
keywords = ["PDF", "OCR", "optical character recognition", "PDF/A", "scanning"]
|
||||
|
||||
[project.urls]
|
||||
Documentation = "https://ocrmypdf.readthedocs.io/"
|
||||
Source = "https://github.com/ocrmypdf/OCRmyPDF"
|
||||
Tracker = "https://github.com/ocrmypdf/OCRmyPDF/issues"
|
||||
Changelog = "https://github.com/ocrmypdf/OCRmyPDF/docs/release_notes.md"
|
||||
|
||||
[project.optional-dependencies]
|
||||
docs = ["sphinx", "sphinx-issues", "sphinx-rtd-theme"]
|
||||
extended_test = ["PyMuPDF==1.19.1"]
|
||||
test = [
|
||||
"coverage[toml]>=5",
|
||||
"pytest>=6.0.0",
|
||||
"pytest-cov>=2.11.1",
|
||||
"pytest-xdist>=2.2.0",
|
||||
"python-xmp-toolkit==2.0.1", # also requires apt-get install libexempi3
|
||||
"types-Pillow",
|
||||
"types-humanfriendly",
|
||||
]
|
||||
watcher = ["watchdog>=1.0.2"]
|
||||
webservice = ["Flask>=1"]
|
||||
# User-installable features - use `uv sync --extra <name>` or `pip install ocrmypdf[name]`
|
||||
watcher = ["watchdog>=1.0.2", "cyclopts>=3", "python-dotenv"]
|
||||
webservice = ["streamlit>=1.41.0"]
|
||||
|
||||
[project.scripts]
|
||||
ocrmypdf = "ocrmypdf.__main__:run"
|
||||
|
||||
[tool.setuptools.package-data]
|
||||
ocrmypdf = ["data/sRGB.icc", "py.typed"]
|
||||
[tool.hatch.version]
|
||||
source = "vcs"
|
||||
|
||||
[tool.setuptools.packages.find]
|
||||
where = ["src"]
|
||||
namespaces = false
|
||||
|
||||
[tool.setuptools_scm]
|
||||
[tool.hatch.build.hooks.vcs]
|
||||
version-file = "src/ocrmypdf/_version.py"
|
||||
|
||||
[tool.distutils.bdist_wheel]
|
||||
python-tag = "py38"
|
||||
|
||||
[tool.black]
|
||||
line-length = 88
|
||||
target-version = ["py38", "py39", "py310", "py311"]
|
||||
skip-string-normalization = true
|
||||
include = '\.pyi?$'
|
||||
exclude = '''
|
||||
/(
|
||||
\.eggs
|
||||
| \.git
|
||||
| \.hg
|
||||
| \.mypy_cache
|
||||
| \.tox
|
||||
| \.venv
|
||||
| _build
|
||||
| buck-out
|
||||
| build
|
||||
| dist
|
||||
| docs
|
||||
| misc
|
||||
| \.egg-info
|
||||
)/
|
||||
'''
|
||||
python-tag = "py311"
|
||||
|
||||
[tool.coverage.run]
|
||||
branch = true
|
||||
parallel = true
|
||||
concurrency = ["multiprocessing"]
|
||||
concurrency = ["multiprocessing", "thread"]
|
||||
sigterm = true
|
||||
|
||||
[tool.coverage.paths]
|
||||
source = ["src/ocrmypdf"]
|
||||
@@ -137,54 +89,101 @@ exclude_lines = [
|
||||
"if 0:",
|
||||
"if False:",
|
||||
"if __name__ == .__main__.:",
|
||||
"if TYPE_CHECKING:"
|
||||
]
|
||||
|
||||
[tool.isort]
|
||||
profile = "black"
|
||||
known_first_party = "ocrmypdf"
|
||||
known_third_party = [
|
||||
"PIL",
|
||||
"flask",
|
||||
"img2pdf",
|
||||
"ocrmypdf",
|
||||
"pdfminer",
|
||||
"pikepdf",
|
||||
"pkg_resources",
|
||||
"pluggy",
|
||||
"pytest",
|
||||
"reportlab",
|
||||
"setuptools",
|
||||
"sphinx_rtd_theme",
|
||||
"tqdm",
|
||||
"watchdog",
|
||||
"werkzeug"
|
||||
"if TYPE_CHECKING:",
|
||||
]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
minversion = "6.0"
|
||||
norecursedirs = ["lib", ".pc", ".git", "venv", "output", "cache", "resources"]
|
||||
testpaths = ["tests"]
|
||||
addopts = "-n auto"
|
||||
markers = ["slow"]
|
||||
filterwarnings = ["ignore:.*XMLParser.*:DeprecationWarning"]
|
||||
filterwarnings = [
|
||||
"ignore:.*XMLParser.*:DeprecationWarning",
|
||||
"ignore:.*ast.NameConstant.*:DeprecationWarning:reportlab",
|
||||
"ignore:.*distutils.*:DeprecationWarning:libxmp",
|
||||
]
|
||||
|
||||
[tool.mypy]
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = [
|
||||
'pluggy',
|
||||
'tqdm',
|
||||
'coloredlogs',
|
||||
'img2pdf',
|
||||
'pdfminer.*',
|
||||
'reportlab.*',
|
||||
'fitz',
|
||||
'libxmp.utils'
|
||||
'libxmp.utils',
|
||||
]
|
||||
ignore_missing_imports = true
|
||||
|
||||
[tool.pylint.basic]
|
||||
good-names = ["i", "j", "k", "ex", "Run", "_", "e", "p", "im", "w", "h", "m", "x", "y", "a", "b", "fp", "n", "f", "s", "v", "q", "dx", "dy"]
|
||||
logging-format-style = "old"
|
||||
disable = ["raw-checker-failed", "bad-inline-option", "locally-disabled", "file-ignored", "suppressed-message", "useless-suppression", "deprecated-pragma", "use-symbolic-message-instead", "logging-fstring-interpolation", "missing-function-docstring", "too-few-public-methods"]
|
||||
[tool.ruff]
|
||||
target-version = "py311"
|
||||
exclude = ["src/ocrmypdf/_version.py"] # Autogenerated
|
||||
|
||||
[tool.ruff.lint]
|
||||
"select" = [
|
||||
"D", # pydocstyle
|
||||
"E", # pycodestyle
|
||||
"W", # pycodestyle
|
||||
"F", # pyflakes
|
||||
"I", # isort
|
||||
"UP", # pyupgrade
|
||||
"SIM", # simplify
|
||||
"B", # flake8-bugbear
|
||||
"ICN", # flake8-import-conventions
|
||||
]
|
||||
ignore = [
|
||||
"B028", # warning with no explicit stacklevel
|
||||
# rule is key in dict instead of key in dict.keys(); but pikepdf semantics differ
|
||||
"SIM118",
|
||||
]
|
||||
|
||||
[tool.ruff.lint.isort]
|
||||
known-first-party = ["ocrmypdf"]
|
||||
required-imports = ["from __future__ import annotations"]
|
||||
|
||||
[tool.ruff.lint.flake8-import-conventions]
|
||||
# Prohibit explicit imports from the 'datetime' module
|
||||
banned-from = ["datetime"]
|
||||
# Optionally, suggest an alias for 'import datetime' (e.g., as dt)
|
||||
extend-aliases = { "datetime" = "dt" }
|
||||
|
||||
[tool.ruff.lint.pydocstyle]
|
||||
convention = "google"
|
||||
|
||||
[tool.ruff.lint.per-file-ignores]
|
||||
"docs/conf.py" = ["D100", "D101", "D105"]
|
||||
"tests/*.py" = ["D100", "D101", "D102", "D103", "D105", "E501"]
|
||||
"misc/*.py" = ["D103", "D101", "D102"]
|
||||
"src/ocrmypdf/builtin_plugins/*.py" = ["D103", "D102", "D105"]
|
||||
|
||||
[tool.ruff.format]
|
||||
quote-style = "preserve"
|
||||
|
||||
[dependency-groups]
|
||||
# Developer-only tools - use `uv sync --group <name>`
|
||||
dev = ["mypy>=1.13.0", "ipykernel>=6.29.5", "reportlab>=4.4.4"]
|
||||
test = [
|
||||
# Core testing framework
|
||||
"coverage[toml]>=6.2",
|
||||
"hypothesis>=6.36.0",
|
||||
"pytest>=6.2.5",
|
||||
"pytest-cov>=3.0.0",
|
||||
"pytest-xdist>=2.5.0",
|
||||
# Test dependencies
|
||||
"python-xmp-toolkit==2.0.1", # also requires apt-get install libexempi3
|
||||
"reportlab>=3.6.8",
|
||||
# Type stubs for testing
|
||||
"types-Pillow",
|
||||
"types-humanfriendly",
|
||||
# Extended test capabilities (merged from extended_test)
|
||||
"pymupdf>=1.24.14",
|
||||
]
|
||||
docs = [
|
||||
"myst-parser>=4.0.1",
|
||||
"sphinx",
|
||||
"sphinx-issues",
|
||||
"sphinx-rtd-theme",
|
||||
"sphinxcontrib-mermaid",
|
||||
]
|
||||
streamlit-dev = ["streamlit>=1.40.2", "streamlit-pdf-viewer>=0.0.19"]
|
||||
|
||||
@@ -0,0 +1,231 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Generate the Occulta glyphless font for OCRmyPDF.
|
||||
|
||||
Occulta (Latin for "hidden") is a glyphless font designed for invisible text layers
|
||||
in searchable PDFs. It has proper Unicode cmap coverage using format 13 (many-to-one)
|
||||
for efficient mapping of all BMP codepoints to a small set of width-specific glyphs.
|
||||
|
||||
Features:
|
||||
- Full BMP coverage (U+0000 to U+FFFF)
|
||||
- Width-aware glyphs for proper text selection:
|
||||
- Zero-width for combining marks and invisible characters
|
||||
- Regular width (500 units) for Latin, Greek, Cyrillic, Arabic, Hebrew, etc.
|
||||
- Double width (1000 units) for CJK and fullwidth characters
|
||||
- Uses cmap format 13 (many-to-one) for ~12KB size vs ~780KB with format 12
|
||||
- Compatible with fpdf2 and other modern PDF libraries
|
||||
|
||||
Usage:
|
||||
python scripts/generate_glyphless_font.py
|
||||
|
||||
Output:
|
||||
src/ocrmypdf/data/Occulta.ttf
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
|
||||
from fontTools.fontBuilder import FontBuilder
|
||||
from fontTools.ttLib import TTFont
|
||||
from fontTools.ttLib.tables._c_m_a_p import CmapSubtable
|
||||
from fontTools.ttLib.tables._g_l_y_f import Glyph
|
||||
|
||||
# Output path relative to this script
|
||||
OUTPUT_PATH = Path(__file__).parent.parent / "src" / "ocrmypdf" / "data" / "Occulta.ttf"
|
||||
|
||||
# Font metrics (units per em = 1000)
|
||||
UNITS_PER_EM = 1000
|
||||
ASCENT = 800
|
||||
DESCENT = -200
|
||||
|
||||
# Glyph definitions: (name, advance_width, left_side_bearing)
|
||||
GLYPHS = [
|
||||
(".notdef", 500, 0), # Required, used for unmapped characters
|
||||
("space", 500, 0), # U+0020 SPACE
|
||||
("nbspace", 500, 0), # U+00A0 NO-BREAK SPACE
|
||||
("blank0", 0, 0), # Zero-width (combining marks, ZWNJ, ZWJ, BOM)
|
||||
("blank1", 500, 0), # Regular width (most scripts)
|
||||
("blank2", 1000, 0), # Double width (CJK, fullwidth)
|
||||
]
|
||||
|
||||
# Explicit zero-width character codepoints
|
||||
ZERO_WIDTH_CHARS = frozenset(
|
||||
[
|
||||
0x200B, # ZERO WIDTH SPACE
|
||||
0x200C, # ZERO WIDTH NON-JOINER
|
||||
0x200D, # ZERO WIDTH JOINER
|
||||
0xFEFF, # ZERO WIDTH NO-BREAK SPACE (BOM)
|
||||
0x200E, # LEFT-TO-RIGHT MARK
|
||||
0x200F, # RIGHT-TO-LEFT MARK
|
||||
0x202A, # LEFT-TO-RIGHT EMBEDDING
|
||||
0x202B, # RIGHT-TO-LEFT EMBEDDING
|
||||
0x202C, # POP DIRECTIONAL FORMATTING
|
||||
0x202D, # LEFT-TO-RIGHT OVERRIDE
|
||||
0x202E, # RIGHT-TO-LEFT OVERRIDE
|
||||
0x2060, # WORD JOINER
|
||||
0x2061, # FUNCTION APPLICATION
|
||||
0x2062, # INVISIBLE TIMES
|
||||
0x2063, # INVISIBLE SEPARATOR
|
||||
0x2064, # INVISIBLE PLUS
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def classify_codepoint(codepoint: int) -> str:
|
||||
"""Classify a Unicode codepoint into one of our glyph categories.
|
||||
|
||||
Args:
|
||||
codepoint: Unicode codepoint (0x0000 to 0xFFFF)
|
||||
|
||||
Returns:
|
||||
Glyph name to map this codepoint to
|
||||
"""
|
||||
# Special cases first
|
||||
if codepoint == 0x0020:
|
||||
return "space"
|
||||
if codepoint == 0x00A0:
|
||||
return "nbspace"
|
||||
if codepoint in ZERO_WIDTH_CHARS:
|
||||
return "blank0"
|
||||
|
||||
# Use Unicode properties for the rest
|
||||
char = chr(codepoint)
|
||||
try:
|
||||
category = unicodedata.category(char)
|
||||
east_asian_width = unicodedata.east_asian_width(char)
|
||||
|
||||
# Combining marks are zero-width
|
||||
if category.startswith("M"):
|
||||
return "blank0"
|
||||
|
||||
# Wide and Fullwidth characters are double-width
|
||||
if east_asian_width in ("W", "F"):
|
||||
return "blank2"
|
||||
|
||||
# Everything else is regular width
|
||||
return "blank1"
|
||||
|
||||
except (ValueError, TypeError):
|
||||
# Fallback for any edge cases
|
||||
return "blank1"
|
||||
|
||||
|
||||
def build_cmap() -> dict[int, str]:
|
||||
"""Build the Unicode to glyph name mapping for the entire BMP.
|
||||
|
||||
Returns:
|
||||
Dictionary mapping codepoints to glyph names
|
||||
"""
|
||||
return {cp: classify_codepoint(cp) for cp in range(0x10000)}
|
||||
|
||||
|
||||
def create_font() -> TTFont:
|
||||
"""Create the Occulta glyphless font.
|
||||
|
||||
Returns:
|
||||
TTFont object ready to be saved
|
||||
"""
|
||||
glyph_names = [g[0] for g in GLYPHS]
|
||||
|
||||
# Start building the font
|
||||
fb = FontBuilder(UNITS_PER_EM, isTTF=True)
|
||||
fb.setupGlyphOrder(glyph_names)
|
||||
|
||||
# Create empty (invisible) glyphs
|
||||
glyphs = {}
|
||||
for name, _, _ in GLYPHS:
|
||||
glyph = Glyph()
|
||||
glyph.numberOfContours = 0
|
||||
glyphs[name] = glyph
|
||||
fb.setupGlyf(glyphs)
|
||||
|
||||
# Set up horizontal metrics
|
||||
metrics = {name: (width, lsb) for name, width, lsb in GLYPHS}
|
||||
fb.setupHorizontalMetrics(metrics)
|
||||
|
||||
# Minimal cmap to satisfy FontBuilder (we'll replace it later)
|
||||
fb.setupCharacterMap({0x0020: "space", 0x00A0: "nbspace"})
|
||||
|
||||
# Set up other required tables
|
||||
fb.setupHorizontalHeader(ascent=ASCENT, descent=DESCENT)
|
||||
fb.setupOS2(
|
||||
sTypoAscender=ASCENT,
|
||||
sTypoDescender=DESCENT,
|
||||
sTypoLineGap=0,
|
||||
usWinAscent=UNITS_PER_EM,
|
||||
usWinDescent=abs(DESCENT),
|
||||
sxHeight=500,
|
||||
sCapHeight=700,
|
||||
)
|
||||
import time
|
||||
|
||||
# Use current time for font timestamps
|
||||
now = int(time.time())
|
||||
fb.setupHead(unitsPerEm=UNITS_PER_EM, created=now, modified=now)
|
||||
fb.setupPost()
|
||||
fb.setupNameTable(
|
||||
{
|
||||
"familyName": "Occulta",
|
||||
"styleName": "Regular",
|
||||
"uniqueFontIdentifier": "OCRmyPDF;Occulta-Regular;2026",
|
||||
"fullName": "Occulta Regular",
|
||||
"version": "Version 2.0",
|
||||
"psName": "Occulta-Regular",
|
||||
}
|
||||
)
|
||||
|
||||
# Build the font
|
||||
font = fb.font
|
||||
|
||||
# Now replace the cmap with format 13 for efficient many-to-one mapping
|
||||
char_to_glyph = build_cmap()
|
||||
|
||||
cmap13 = CmapSubtable.newSubtable(13)
|
||||
cmap13.platformID = 3 # Windows
|
||||
cmap13.platEncID = 10 # Unicode full repertoire
|
||||
cmap13.language = 0
|
||||
cmap13.cmap = char_to_glyph
|
||||
|
||||
font["cmap"].tables = [cmap13]
|
||||
|
||||
return font
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Generate the Occulta font and save it."""
|
||||
print("Generating Occulta glyphless font...")
|
||||
|
||||
font = create_font()
|
||||
|
||||
# Create output directory if needed
|
||||
OUTPUT_PATH.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Save the font
|
||||
font.save(str(OUTPUT_PATH))
|
||||
font.close()
|
||||
|
||||
# Report statistics
|
||||
size = OUTPUT_PATH.stat().st_size
|
||||
print(f"Saved to: {OUTPUT_PATH}")
|
||||
print(f"Size: {size:,} bytes")
|
||||
|
||||
# Verify cmap
|
||||
font = TTFont(str(OUTPUT_PATH))
|
||||
for table in font["cmap"].tables:
|
||||
print(
|
||||
f"cmap: Platform {table.platformID}, "
|
||||
f"Encoding {table.platEncID}, "
|
||||
f"Format {table.format}, "
|
||||
f"{len(table.cmap)} mappings"
|
||||
)
|
||||
font.close()
|
||||
|
||||
print("Done!")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+16
-10
@@ -1,24 +1,26 @@
|
||||
# SPDX-FileCopyrightText: 2022 Alexander Langanke
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-FileCopyrightText: 2023 林博仁(Buo-ren, Lin) <Buo.Ren.Lin@gmail.com>
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
name: ocrmypdf
|
||||
title: OCRmyPDF
|
||||
base: core20
|
||||
base: core24
|
||||
version: git
|
||||
summary: OCRmyPDF adds optical character recognition (OCR) to PDFs
|
||||
summary: OCRmyPDF adds a searchable text layer to scanned PDF files
|
||||
description: OCRmyPDF packaged for snap
|
||||
grade: stable
|
||||
confinement: strict
|
||||
icon: docs/images/logo-square-256.svg
|
||||
license: MPL-2.0
|
||||
|
||||
architectures: [amd64]
|
||||
platforms:
|
||||
amd64:
|
||||
|
||||
environment:
|
||||
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/4.00/tessdata
|
||||
GS_LIB: $SNAP/usr/share/ghostscript/9.50/Resource/Init
|
||||
GS_FONTPATH: $SNAP/usr/share/ghostscript/9.50/Resource/Font
|
||||
TESSDATA_PREFIX: $SNAP/usr/share/tesseract-ocr/5/tessdata
|
||||
GS_LIB: $SNAP/usr/share/ghostscript/10.02.1/Resource/Init
|
||||
GS_FONTPATH: $SNAP/usr/share/ghostscript/10.02.1/Resource/Font
|
||||
LD_LIBRARY_PATH: $SNAP/usr/lib/x86_64-linux-gnu
|
||||
|
||||
apps:
|
||||
@@ -48,13 +50,16 @@ parts:
|
||||
jbig2enc:
|
||||
plugin: autotools
|
||||
source: https://github.com/agl/jbig2enc.git
|
||||
source-tag: '0.29'
|
||||
source-tag: "0.29"
|
||||
build-packages:
|
||||
- libleptonica-dev
|
||||
|
||||
ocrmypdf:
|
||||
plugin: python
|
||||
source: https://github.com/ocrmypdf/OCRmyPDF.git
|
||||
source: .
|
||||
|
||||
build-packages:
|
||||
- python3-pip
|
||||
|
||||
stage-packages:
|
||||
- ghostscript
|
||||
@@ -77,7 +82,8 @@ parts:
|
||||
- setuptools
|
||||
- tqdm
|
||||
- pipe
|
||||
- wheel
|
||||
|
||||
override-build: |
|
||||
snapcraftctl build
|
||||
ln -sf ../usr/lib/libsnapcraft-preload.so $SNAPCRAFT_PART_INSTALL/lib/libsnapcraft-preload.so
|
||||
craftctl default
|
||||
ln -sf ../usr/lib/libsnapcraft-preload.so $CRAFT_PART_INSTALL/lib/libsnapcraft-preload.so
|
||||
|
||||
@@ -9,9 +9,18 @@ from pluggy import HookimplMarker as _HookimplMarker
|
||||
|
||||
from ocrmypdf import helpers, hocrtransform, pdfa, pdfinfo
|
||||
from ocrmypdf._concurrent import Executor
|
||||
from ocrmypdf._defaults import PROGRAM_NAME
|
||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||
from ocrmypdf._version import PROGRAM_NAME, __version__
|
||||
from ocrmypdf.api import Verbosity, configure_logging, ocr
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf._pipelines._common import (
|
||||
configure_debug_logging,
|
||||
)
|
||||
from ocrmypdf._version import __version__
|
||||
from ocrmypdf.api import (
|
||||
Verbosity,
|
||||
configure_logging,
|
||||
ocr,
|
||||
)
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
DpiError,
|
||||
@@ -26,6 +35,50 @@ from ocrmypdf.exceptions import (
|
||||
TesseractConfigError,
|
||||
UnsupportedImageFormatError,
|
||||
)
|
||||
from ocrmypdf.models.ocr_element import (
|
||||
Baseline,
|
||||
BoundingBox,
|
||||
FontInfo,
|
||||
OcrClass,
|
||||
OcrElement,
|
||||
)
|
||||
from ocrmypdf.pluginspec import OcrEngine, OrientationConfidence
|
||||
|
||||
hookimpl = _HookimplMarker('ocrmypdf')
|
||||
|
||||
__all__ = [
|
||||
'__version__',
|
||||
'BadArgsError',
|
||||
'Baseline',
|
||||
'BoundingBox',
|
||||
'configure_debug_logging',
|
||||
'configure_logging',
|
||||
'DpiError',
|
||||
'EncryptedPdfError',
|
||||
'Executor',
|
||||
'ExitCode',
|
||||
'ExitCodeException',
|
||||
'FontInfo',
|
||||
'helpers',
|
||||
'hocrtransform',
|
||||
'hookimpl',
|
||||
'InputFileError',
|
||||
'MissingDependencyError',
|
||||
'ocr',
|
||||
'OcrClass',
|
||||
'OcrElement',
|
||||
'OcrEngine',
|
||||
'OcrOptions',
|
||||
'OrientationConfidence',
|
||||
'OutputFileAccessError',
|
||||
'PageContext',
|
||||
'pdfa',
|
||||
'PdfContext',
|
||||
'pdfinfo',
|
||||
'PriorOcrFoundError',
|
||||
'PROGRAM_NAME',
|
||||
'SubprocessOutputError',
|
||||
'TesseractConfigError',
|
||||
'UnsupportedImageFormatError',
|
||||
'Verbosity',
|
||||
]
|
||||
|
||||
@@ -7,17 +7,17 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import multiprocessing
|
||||
import os
|
||||
import signal
|
||||
import sys
|
||||
from contextlib import suppress
|
||||
from multiprocessing import set_start_method
|
||||
|
||||
from ocrmypdf import __version__
|
||||
from ocrmypdf._plugin_manager import get_parser_options_plugins
|
||||
from ocrmypdf._sync import run_pipeline
|
||||
from ocrmypdf._pipelines.ocr import run_pipeline_cli
|
||||
from ocrmypdf._validation import check_options
|
||||
from ocrmypdf.api import Verbosity, configure_logging
|
||||
from ocrmypdf.cli import get_options_and_plugins
|
||||
from ocrmypdf.exceptions import (
|
||||
BadArgsError,
|
||||
ExitCode,
|
||||
@@ -29,11 +29,17 @@ log = logging.getLogger('ocrmypdf')
|
||||
|
||||
|
||||
def sigbus(*args):
|
||||
"""Handle SIGBUS signals.
|
||||
|
||||
pikepdf, depending on configuration, may use mmap so SIGBUS is a
|
||||
possibility.
|
||||
"""
|
||||
raise InputFileError("Lost access to the input file")
|
||||
|
||||
|
||||
def run(args=None):
|
||||
_parser, options, plugin_manager = get_parser_options_plugins(args=args)
|
||||
"""Run the ocrmypdf command line interface."""
|
||||
options, plugin_manager = get_options_and_plugins(args=args)
|
||||
|
||||
with suppress(AttributeError, PermissionError):
|
||||
os.nice(5)
|
||||
@@ -66,9 +72,13 @@ def run(args=None):
|
||||
with suppress(AttributeError, OSError):
|
||||
signal.signal(signal.SIGBUS, sigbus)
|
||||
|
||||
result = run_pipeline(options=options, plugin_manager=plugin_manager)
|
||||
result = run_pipeline_cli(options=options, plugin_manager=plugin_manager)
|
||||
return result
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
multiprocessing.freeze_support()
|
||||
if sys.platform not in ('win32', 'darwin'):
|
||||
with suppress(RuntimeError):
|
||||
multiprocessing.set_start_method('forkserver')
|
||||
sys.exit(run())
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""OCRmyPDF PDF annotation cleanup."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
|
||||
from pikepdf import Dictionary, Name, NameTree, Pdf
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def remove_broken_goto_annotations(pdf: Pdf) -> bool:
|
||||
"""Remove broken goto annotations from a PDF.
|
||||
|
||||
If a PDF contains a GoTo Action that points to a named destination that does not
|
||||
exist, Ghostscript PDF/A conversion will fail. In any event, a named destination
|
||||
that is not defined is not useful.
|
||||
|
||||
Args:
|
||||
pdf: Opened PDF file.
|
||||
|
||||
Returns:
|
||||
bool: True if the file was modified, False if not.
|
||||
"""
|
||||
modified = False
|
||||
|
||||
# Check if there are any named destinations
|
||||
if Name.Names not in pdf.Root:
|
||||
return modified
|
||||
if Name.Dests not in pdf.Root[Name.Names]:
|
||||
return modified
|
||||
|
||||
dests = pdf.Root[Name.Names][Name.Dests]
|
||||
if not isinstance(dests, Dictionary):
|
||||
return modified
|
||||
nametree = NameTree(dests)
|
||||
|
||||
# Create a set of all named destinations
|
||||
names = set(k for k in nametree.keys())
|
||||
|
||||
for n, page in enumerate(pdf.pages):
|
||||
if Name.Annots not in page:
|
||||
continue
|
||||
for annot in page[Name.Annots]:
|
||||
if not isinstance(annot, Dictionary):
|
||||
continue
|
||||
if Name.A not in annot or Name.D not in annot[Name.A]:
|
||||
continue
|
||||
# We found an annotation that points to a named destination
|
||||
named_destination = str(annot[Name.A][Name.D])
|
||||
if named_destination not in names:
|
||||
# If there is no corresponding named destination, remove the
|
||||
# annotation. Having no destination set is still valid and just
|
||||
# makes the link non-functional.
|
||||
log.warning(
|
||||
f"Disabling a hyperlink annotation on page {n + 1} to a "
|
||||
"non-existent named destination "
|
||||
f"{named_destination}."
|
||||
)
|
||||
del annot[Name.A][Name.D]
|
||||
modified = True
|
||||
|
||||
return modified
|
||||
+22
-31
@@ -7,27 +7,20 @@ from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from abc import ABC, abstractmethod
|
||||
from typing import Callable, Iterable
|
||||
from collections.abc import Callable, Iterable
|
||||
from typing import Any, TypeVar
|
||||
|
||||
from ocrmypdf._progressbar import NullProgressBar, ProgressBar
|
||||
|
||||
T = TypeVar('T')
|
||||
|
||||
|
||||
def _task_noop(*_args, **_kwargs):
|
||||
def _task_noop(*_args, **_kwargs) -> None:
|
||||
return
|
||||
|
||||
|
||||
class NullProgressBar:
|
||||
"""Progress bar API that takes no actions."""
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
pass
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return False
|
||||
|
||||
def update(self, _arg=None):
|
||||
return
|
||||
def _task_finished_noop(_result: Any, pbar: ProgressBar):
|
||||
pbar.update()
|
||||
|
||||
|
||||
class Executor(ABC):
|
||||
@@ -45,14 +38,13 @@ class Executor(ABC):
|
||||
*,
|
||||
use_threads: bool,
|
||||
max_workers: int,
|
||||
tqdm_kwargs: dict,
|
||||
progress_kwargs: dict,
|
||||
worker_initializer: Callable | None = None,
|
||||
task: Callable | None = None,
|
||||
task: Callable[..., T] | None = None,
|
||||
task_arguments: Iterable | None = None,
|
||||
task_finished: Callable | None = None,
|
||||
task_finished: Callable[[T, ProgressBar], None] | None = None,
|
||||
) -> None:
|
||||
"""
|
||||
Set up parallel execution and progress reporting.
|
||||
"""Set up parallel execution and progress reporting.
|
||||
|
||||
Args:
|
||||
use_threads: If ``False``, the workload is the sort that will benefit from
|
||||
@@ -60,7 +52,7 @@ class Executor(ABC):
|
||||
heavily, and parallelizing it with threads is not expected to be
|
||||
performant).
|
||||
max_workers: The maximum number of workers that should be run.
|
||||
tdqm_kwargs: Arguments to set up the progress bar.
|
||||
progress_kwargs: Arguments to set up the progress bar.
|
||||
worker_initializer: Called when a worker is initialized, in the worker's
|
||||
execution context. If the child workers are processes, it must be
|
||||
possible to marshall/pickle the worker initializer.
|
||||
@@ -73,13 +65,12 @@ class Executor(ABC):
|
||||
task. This runs in the parent's context, but the parameters must be
|
||||
marshallable to the worker.
|
||||
"""
|
||||
|
||||
if not task_arguments:
|
||||
return # Nothing to do!
|
||||
if not worker_initializer:
|
||||
worker_initializer = _task_noop
|
||||
if not task_finished:
|
||||
task_finished = _task_noop
|
||||
task_finished = _task_finished_noop
|
||||
if not task:
|
||||
task = _task_noop
|
||||
|
||||
@@ -87,7 +78,7 @@ class Executor(ABC):
|
||||
self._execute(
|
||||
use_threads=use_threads,
|
||||
max_workers=max_workers,
|
||||
tqdm_kwargs=tqdm_kwargs,
|
||||
progress_kwargs=progress_kwargs,
|
||||
worker_initializer=worker_initializer,
|
||||
task=task,
|
||||
task_arguments=task_arguments,
|
||||
@@ -100,7 +91,7 @@ class Executor(ABC):
|
||||
*,
|
||||
use_threads: bool,
|
||||
max_workers: int,
|
||||
tqdm_kwargs: dict,
|
||||
progress_kwargs: dict,
|
||||
worker_initializer: Callable,
|
||||
task: Callable,
|
||||
task_arguments: Iterable,
|
||||
@@ -110,8 +101,8 @@ class Executor(ABC):
|
||||
|
||||
|
||||
def setup_executor(plugin_manager) -> Executor:
|
||||
pbar_class = plugin_manager.hook.get_progressbar_class()
|
||||
return plugin_manager.hook.get_executor(progressbar_class=pbar_class)
|
||||
pbar_class = plugin_manager.get_progressbar_class()
|
||||
return plugin_manager.get_executor(progressbar_class=pbar_class)
|
||||
|
||||
|
||||
class SerialExecutor(Executor):
|
||||
@@ -126,13 +117,13 @@ class SerialExecutor(Executor):
|
||||
*,
|
||||
use_threads: bool,
|
||||
max_workers: int,
|
||||
tqdm_kwargs: dict,
|
||||
progress_kwargs: dict,
|
||||
worker_initializer: Callable,
|
||||
task: Callable,
|
||||
task_arguments: Iterable,
|
||||
task_finished: Callable,
|
||||
): # pylint: disable=unused-argument
|
||||
with self.pbar_class(**tqdm_kwargs) as pbar:
|
||||
with self.pbar_class(**progress_kwargs) as pbar:
|
||||
for args in task_arguments:
|
||||
result = task(args)
|
||||
result = task(*args)
|
||||
task_finished(result, pbar)
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
# Enforce English hegemony
|
||||
from __future__ import annotations
|
||||
|
||||
DEFAULT_LANGUAGE = 'eng'
|
||||
|
||||
# Default rotation threshold
|
||||
DEFAULT_ROTATE_PAGES_THRESHOLD = 14.0
|
||||
|
||||
PROGRAM_NAME = 'OCRmyPDF'
|
||||
@@ -1,6 +1,6 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Manage third party executables"""
|
||||
"""Manage third party executables."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -1,15 +1,14 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Interface to Ghostscript executable"""
|
||||
"""Interface to Ghostscript executable."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from io import BytesIO
|
||||
from collections import deque
|
||||
from os import fspath
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, CalledProcessError
|
||||
@@ -17,35 +16,64 @@ from subprocess import PIPE, CalledProcessError
|
||||
from packaging.version import Version
|
||||
from PIL import Image, UnidentifiedImageError
|
||||
|
||||
from ocrmypdf.exceptions import SubprocessOutputError
|
||||
from ocrmypdf.exceptions import (
|
||||
ColorConversionNeededError,
|
||||
InputFileError,
|
||||
SubprocessOutputError,
|
||||
)
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.pluginspec import GhostscriptRasterDevice
|
||||
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
||||
|
||||
# Remove this workaround when we require Pillow >= 10
|
||||
try:
|
||||
Transpose = Image.Transpose # type: ignore
|
||||
except AttributeError:
|
||||
# Pillow 9 shim
|
||||
Transpose = Image # type: ignore
|
||||
COLOR_CONVERSION_STRATEGIES = frozenset(
|
||||
[
|
||||
'CMYK',
|
||||
'Gray',
|
||||
'LeaveColorUnchanged',
|
||||
'RGB',
|
||||
'UseDeviceIndependentColor',
|
||||
]
|
||||
)
|
||||
# Ghostscript executable - gswin32c is not supported
|
||||
GS = 'gswin64c' if os.name == 'nt' else 'gs'
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Most reliable what to get the bitness of Python interpreter, according to Python docs
|
||||
_IS_64BIT = sys.maxsize > 2**32
|
||||
|
||||
_GSWIN = None
|
||||
if os.name == 'nt':
|
||||
if _IS_64BIT:
|
||||
_GSWIN = 'gswin64c'
|
||||
else:
|
||||
_GSWIN = 'gswin32c'
|
||||
class DuplicateFilter(logging.Filter):
|
||||
"""Filter out duplicate log messages.
|
||||
|
||||
GS = _GSWIN if _GSWIN else 'gs'
|
||||
del _GSWIN
|
||||
A context window of default 5 messages is used to determine if a message is a
|
||||
duplicate. This is because some Ghostscript messages are word wrapped.
|
||||
"""
|
||||
|
||||
def __init__(self, logger: logging.Logger, context_window=5):
|
||||
self.window: deque[str] = deque([], maxlen=context_window)
|
||||
self.logger = logger
|
||||
self.levelno = logging.DEBUG
|
||||
self.count = 0
|
||||
|
||||
def filter(self, record):
|
||||
if record.msg in self.window:
|
||||
self.count += 1
|
||||
self.levelno = record.levelno
|
||||
return False
|
||||
else:
|
||||
if self.count >= 1:
|
||||
rep_msg = f"(suppressed {self.count} repeated lines)"
|
||||
self.count = 0 # Avoid infinite recursion
|
||||
self.logger.log(self.levelno, rep_msg)
|
||||
self.window.clear()
|
||||
self.window.append(record.msg)
|
||||
return True
|
||||
|
||||
|
||||
def version():
|
||||
return get_version(GS)
|
||||
log.addFilter(DuplicateFilter(log))
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
return Version(get_version(GS))
|
||||
|
||||
|
||||
def _gs_error_reported(stream) -> bool:
|
||||
@@ -53,18 +81,48 @@ def _gs_error_reported(stream) -> bool:
|
||||
return bool(match)
|
||||
|
||||
|
||||
def _gs_devicen_reported(stream) -> bool:
|
||||
"""Did Ghostscript warn about a DeviceN with inappropriate alternate?
|
||||
|
||||
If so, we need the user to select a color conversion, or the resulting PDF will
|
||||
not present correctly in some PDF viewers.
|
||||
"""
|
||||
match = re.search(
|
||||
r'DeviceN.*inappropriate alternate',
|
||||
stream,
|
||||
flags=re.IGNORECASE | re.MULTILINE,
|
||||
)
|
||||
return bool(match)
|
||||
|
||||
|
||||
def rasterize_pdf(
|
||||
input_file: os.PathLike,
|
||||
output_file: os.PathLike,
|
||||
*,
|
||||
raster_device: str,
|
||||
raster_device: GhostscriptRasterDevice,
|
||||
raster_dpi: Resolution,
|
||||
pageno: int = 1,
|
||||
page_dpi: Resolution | None = None,
|
||||
rotation: int | None = None,
|
||||
filter_vector: bool = False,
|
||||
stop_on_error: bool = False,
|
||||
use_cropbox: bool = False,
|
||||
):
|
||||
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units."""
|
||||
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units.
|
||||
|
||||
Args:
|
||||
input_file: The PDF file to rasterize.
|
||||
output_file: The file to write the rasterized PDF to.
|
||||
raster_device: The Ghostscript raster device to use to rasterize the PDF.
|
||||
raster_dpi: Resolution in dots per inch at which to rasterize page.
|
||||
pageno: Page number to rasterize (beginning at page 1).
|
||||
page_dpi: Resolution, overriding output image DPI.
|
||||
rotation: Cardinal angle, clockwise, to rotate page.
|
||||
filter_vector: If True, remove vector graphics objects.
|
||||
stop_on_error: If True, stop rasterizing on the first error.
|
||||
use_cropbox: If True, rasterize the CropBox instead of MediaBox.
|
||||
Default is False (use MediaBox).
|
||||
"""
|
||||
raster_dpi = raster_dpi.round(6)
|
||||
if not page_dpi:
|
||||
page_dpi = raster_dpi
|
||||
@@ -72,7 +130,6 @@ def rasterize_pdf(
|
||||
args_gs = (
|
||||
[
|
||||
GS,
|
||||
'-dQUIET',
|
||||
'-dSAFER',
|
||||
'-dBATCH',
|
||||
'-dNOPAUSE',
|
||||
@@ -82,10 +139,12 @@ def rasterize_pdf(
|
||||
f'-dLastPage={pageno}',
|
||||
f'-r{raster_dpi.x:f}x{raster_dpi.y:f}',
|
||||
]
|
||||
+ (['-dUseCropBox'] if use_cropbox else [])
|
||||
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
||||
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||
+ [
|
||||
'-o',
|
||||
'-',
|
||||
fspath(output_file),
|
||||
'-sstdout=%stderr', # Literal %s, not string interpolation
|
||||
'-dAutoRotatePages=/None', # Probably has no effect on raster
|
||||
'-f',
|
||||
@@ -97,33 +156,48 @@ def rasterize_pdf(
|
||||
p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True)
|
||||
except CalledProcessError as e:
|
||||
log.error(e.stderr.decode(errors='replace'))
|
||||
raise SubprocessOutputError('Ghostscript rasterizing failed') from e
|
||||
else:
|
||||
stderr = p.stderr.decode(errors='replace')
|
||||
if _gs_error_reported(stderr):
|
||||
log.error(stderr)
|
||||
Path(output_file).unlink(missing_ok=True)
|
||||
raise SubprocessOutputError("Ghostscript rasterizing failed") from e
|
||||
|
||||
stderr = p.stderr.decode(errors='replace')
|
||||
if _gs_error_reported(stderr):
|
||||
log.error(stderr)
|
||||
if stop_on_error and "recoverable image error" in stderr:
|
||||
Path(output_file).unlink(missing_ok=True)
|
||||
raise InputFileError(
|
||||
"Ghostscript rasterizing failed. The input file contains errors that "
|
||||
"cause PDF viewers to interpret it differently and incorrectly. "
|
||||
"Try using --continue-on-soft-render-error and manually inspect the "
|
||||
"input and output files to check for visual differences or errors."
|
||||
)
|
||||
|
||||
try:
|
||||
with Image.open(BytesIO(p.stdout)) as im:
|
||||
with Image.open(output_file) as im:
|
||||
if rotation is not None:
|
||||
log.debug("Rotating output by %i", rotation)
|
||||
# rotation is a clockwise angle and Image.ROTATE_* is
|
||||
# counterclockwise so this cancels out the rotation
|
||||
if rotation == 90:
|
||||
im = im.transpose(Transpose.ROTATE_90)
|
||||
im = im.transpose(Image.Transpose.ROTATE_90)
|
||||
elif rotation == 180:
|
||||
im = im.transpose(Transpose.ROTATE_180)
|
||||
im = im.transpose(Image.Transpose.ROTATE_180)
|
||||
elif rotation == 270:
|
||||
im = im.transpose(Transpose.ROTATE_270)
|
||||
im = im.transpose(Image.Transpose.ROTATE_270)
|
||||
if rotation % 180 == 90:
|
||||
page_dpi = page_dpi.flip_axis()
|
||||
im.save(fspath(output_file), dpi=page_dpi)
|
||||
im.save(output_file, dpi=page_dpi)
|
||||
except UnidentifiedImageError:
|
||||
log.error(
|
||||
f"Ghostscript (using {raster_device} at {raster_dpi} dpi) produced "
|
||||
"an invalid page image file."
|
||||
)
|
||||
raise
|
||||
except OSError as e:
|
||||
log.error(
|
||||
f"Ghostscript (using {raster_device} at {raster_dpi} dpi) produced "
|
||||
"an invalid page image file."
|
||||
)
|
||||
raise UnidentifiedImageError() from e
|
||||
|
||||
|
||||
class GhostscriptFollower:
|
||||
@@ -137,6 +211,17 @@ class GhostscriptFollower:
|
||||
self.progressbar_class = progressbar_class
|
||||
self.progressbar = None
|
||||
|
||||
def __enter__(self):
|
||||
# We can't actually set up the progressbar here, because we don't know
|
||||
# how many pages there are until the first __call__() happens. So we
|
||||
# do it in __call__().
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
if self.progressbar:
|
||||
return self.progressbar.__exit__(exc_type, exc_value, traceback)
|
||||
return False
|
||||
|
||||
def __call__(self, line):
|
||||
if not self.progressbar_class:
|
||||
return
|
||||
@@ -147,7 +232,8 @@ class GhostscriptFollower:
|
||||
self.progressbar = self.progressbar_class(
|
||||
total=self.count, desc="PDF/A conversion", unit='page'
|
||||
)
|
||||
return
|
||||
# Now that we know the count, we can set up the progressbar.
|
||||
self.progressbar.__enter__()
|
||||
else:
|
||||
if self.re_page.match(line.strip()):
|
||||
self.progressbar.update()
|
||||
@@ -158,9 +244,11 @@ def generate_pdfa(
|
||||
output_file: os.PathLike,
|
||||
*,
|
||||
compression: str,
|
||||
color_conversion_strategy: str,
|
||||
pdf_version: str = '1.5',
|
||||
pdfa_part: str = '2',
|
||||
progressbar_class=None,
|
||||
stop_on_error: bool = False,
|
||||
):
|
||||
# Ghostscript's compression is all or nothing. We can either force all images
|
||||
# to JPEG, force all to Flate/PNG, or let it decide how to encode the images.
|
||||
@@ -186,13 +274,16 @@ def generate_pdfa(
|
||||
"-dAutoFilterGrayImages=true",
|
||||
]
|
||||
|
||||
strategy = 'LeaveColorUnchanged'
|
||||
gs_version = Version(version())
|
||||
gs_version = version()
|
||||
if gs_version == Version('9.56.0'):
|
||||
# 9.56.0 breaks our OCR, should be fixed in 9.56.1
|
||||
# https://bugs.ghostscript.com/show_bug.cgi?id=705187
|
||||
compression_args.append('-dNEWPDF=false')
|
||||
|
||||
if os.name == 'nt':
|
||||
# Windows has lots of fatal "permission denied" errors
|
||||
stop_on_error = False
|
||||
|
||||
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
||||
# is set; see:
|
||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699392
|
||||
@@ -202,34 +293,34 @@ def generate_pdfa(
|
||||
"-dBATCH",
|
||||
"-dNOPAUSE",
|
||||
"-dSAFER",
|
||||
"-dCompatibilityLevel=" + str(pdf_version),
|
||||
f"-dCompatibilityLevel={str(pdf_version)}",
|
||||
"-sDEVICE=pdfwrite",
|
||||
"-dAutoRotatePages=/None",
|
||||
"-sColorConversionStrategy=" + strategy,
|
||||
f"-sColorConversionStrategy={color_conversion_strategy}",
|
||||
]
|
||||
+ (['-dPDFSTOPONERROR'] if stop_on_error else [])
|
||||
+ compression_args
|
||||
+ [
|
||||
"-dJPEGQ=95",
|
||||
"-dPDFA=" + pdfa_part,
|
||||
"-dSubsetFonts=false", # Prevents GS from messing up some encodings
|
||||
f"-dPDFA={pdfa_part}",
|
||||
"-dPDFACompatibilityPolicy=1",
|
||||
"-o",
|
||||
"-",
|
||||
fspath(output_file),
|
||||
"-sstdout=%stderr", # Literal %s, not string interpolation
|
||||
]
|
||||
)
|
||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||
|
||||
try:
|
||||
with Path(output_file).open('wb') as output:
|
||||
with GhostscriptFollower(progressbar_class) as pbar:
|
||||
p = run_polling_stderr(
|
||||
args_gs,
|
||||
stdout=output,
|
||||
stderr=PIPE,
|
||||
check=True,
|
||||
text=True,
|
||||
encoding='utf-8',
|
||||
errors='replace',
|
||||
callback=GhostscriptFollower(progressbar_class),
|
||||
callback=pbar,
|
||||
)
|
||||
except CalledProcessError as e:
|
||||
# Ghostscript does not change return code when it fails to create
|
||||
@@ -241,14 +332,11 @@ def generate_pdfa(
|
||||
# If there is an error we log the whole stderr, except for filtering
|
||||
# duplicates.
|
||||
if _gs_error_reported(stderr):
|
||||
last_part = None
|
||||
repcount = 0
|
||||
# Ghostscript outputs the pattern **** Error: .... frequently.
|
||||
# Occasionally the error message is spammed many times. We filter
|
||||
# out duplicates of this message using the filter above. We use
|
||||
# the **** pattern to split the stderr into parts.
|
||||
for part in stderr.split('****'):
|
||||
if part != last_part:
|
||||
if repcount > 1:
|
||||
log.error(f"(previous error message repeated {repcount} times)")
|
||||
repcount = 0
|
||||
log.error(part)
|
||||
else:
|
||||
repcount += 1
|
||||
last_part = part
|
||||
log.error(part)
|
||||
if _gs_devicen_reported(stderr):
|
||||
raise ColorConversionNeededError()
|
||||
|
||||
@@ -1,18 +1,26 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Interface to jbig2 executable"""
|
||||
"""Interface to jbig2 executable."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from subprocess import PIPE
|
||||
from subprocess import PIPE, CalledProcessError
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
|
||||
def version():
|
||||
return get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
def version() -> Version:
|
||||
try:
|
||||
version = get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||
except CalledProcessError as e:
|
||||
# TeX Live for Windows provides an incompatible jbig2.EXE which may
|
||||
# be on the PATH.
|
||||
raise MissingDependencyError('jbig2enc') from e
|
||||
return Version(version)
|
||||
|
||||
|
||||
def available():
|
||||
@@ -23,33 +31,9 @@ def available():
|
||||
return True
|
||||
|
||||
|
||||
def convert_group(*, cwd, infiles, out_prefix):
|
||||
args = [
|
||||
'jbig2',
|
||||
'-b',
|
||||
out_prefix,
|
||||
'-s', # symbol mode (lossy)
|
||||
# '-r', # refinement mode (lossless symbol mode, currently disabled in
|
||||
# jbig2)
|
||||
'-p',
|
||||
]
|
||||
args.extend(infiles)
|
||||
proc = run(args, cwd=cwd, stdout=PIPE, stderr=PIPE)
|
||||
proc.check_returncode()
|
||||
return proc
|
||||
|
||||
|
||||
def convert_group_mp(args):
|
||||
return convert_group(cwd=args[0], infiles=args[1], out_prefix=args[2])
|
||||
|
||||
|
||||
def convert_single(*, cwd, infile, outfile):
|
||||
args = ['jbig2', '-p', infile]
|
||||
def convert_single(cwd, infile, outfile, threshold):
|
||||
args = ['jbig2', '--pdf', '-t', str(threshold), infile]
|
||||
with open(outfile, 'wb') as fstdout:
|
||||
proc = run(args, cwd=cwd, stdout=fstdout, stderr=PIPE)
|
||||
proc.check_returncode()
|
||||
return proc
|
||||
|
||||
|
||||
def convert_single_mp(args):
|
||||
return convert_single(cwd=args[0], infile=args[1], outfile=args[2])
|
||||
|
||||
@@ -1,23 +1,21 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Interface to pngquant executable"""
|
||||
"""Interface to pngquant executable."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from contextlib import contextmanager
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE
|
||||
|
||||
from PIL import Image
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
|
||||
def version():
|
||||
return get_version('pngquant', regex=r'(\d+(\.\d+)*).*')
|
||||
def version() -> Version:
|
||||
return Version(get_version('pngquant', regex=r'(\d+(\.\d+)*).*'))
|
||||
|
||||
|
||||
def available():
|
||||
@@ -28,21 +26,16 @@ def available():
|
||||
return True
|
||||
|
||||
|
||||
@contextmanager
|
||||
def input_as_png(input_file: Path):
|
||||
if not input_file.name.endswith('.png'):
|
||||
with Image.open(input_file) as im:
|
||||
bio = BytesIO()
|
||||
im.save(bio, format='png')
|
||||
bio.seek(0)
|
||||
yield bio
|
||||
else:
|
||||
with open(input_file, 'rb') as f:
|
||||
yield f
|
||||
|
||||
|
||||
def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max: int):
|
||||
with input_as_png(input_file) as input_stream:
|
||||
"""Quantize a PNG image using pngquant.
|
||||
|
||||
Args:
|
||||
input_file: Input PNG image
|
||||
output_file: Output PNG image
|
||||
quality_min: Minimum quality to use
|
||||
quality_max: Maximum quality to use
|
||||
"""
|
||||
with open(input_file, 'rb') as input_stream:
|
||||
args = [
|
||||
'pngquant',
|
||||
'--force',
|
||||
@@ -57,7 +50,3 @@ def quantize(input_file: Path, output_file: Path, quality_min: int, quality_max:
|
||||
if result.returncode == 0:
|
||||
# input_file could be the same as output_file, so we defer the write
|
||||
output_file.write_bytes(result.stdout)
|
||||
|
||||
|
||||
def quantize_mp(args):
|
||||
return quantize(*args)
|
||||
|
||||
+109
-54
@@ -1,19 +1,21 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Interface to Tesseract executable"""
|
||||
"""Interface to Tesseract executable."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from contextlib import suppress
|
||||
from enum import IntEnum
|
||||
from math import pi
|
||||
from os import fspath
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
||||
|
||||
from packaging.version import Version
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf.exceptions import (
|
||||
MissingDependencyError,
|
||||
@@ -26,33 +28,35 @@ from ocrmypdf.subprocess import get_version, run
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
HOCR_TEMPLATE = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||
<head>
|
||||
<title></title>
|
||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 4.1.1' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||
</head>
|
||||
<body>
|
||||
<div class='ocr_page' id='page_1' title='image "_blank.png"; bbox 0 0 {0} {1}; ppageno 0'>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
def _tesseract_env(omp_thread_limit: int | None) -> dict[str, str] | None:
|
||||
"""Create environment dict with OMP_THREAD_LIMIT set for Tesseract subprocesses."""
|
||||
if omp_thread_limit is None:
|
||||
return None
|
||||
env = os.environ.copy()
|
||||
env['OMP_THREAD_LIMIT'] = str(omp_thread_limit)
|
||||
return env
|
||||
|
||||
|
||||
class ThresholdingMethod(IntEnum):
|
||||
"""Tesseract thresholding methods for image binarization."""
|
||||
|
||||
AUTO = 0
|
||||
OTSU = 0 # Alias for AUTO - uses Tesseract's default (legacy Otsu)
|
||||
ADAPTIVE_OTSU = 1
|
||||
SAUVOLA = 2
|
||||
|
||||
|
||||
# Legacy dictionary for backward compatibility
|
||||
TESSERACT_THRESHOLDING_METHODS: dict[str, int] = {
|
||||
'auto': 0,
|
||||
'otsu': 0,
|
||||
'adaptive-otsu': 1,
|
||||
'sauvola': 2,
|
||||
'auto': ThresholdingMethod.AUTO,
|
||||
'otsu': ThresholdingMethod.OTSU,
|
||||
'adaptive-otsu': ThresholdingMethod.ADAPTIVE_OTSU,
|
||||
'sauvola': ThresholdingMethod.SAUVOLA,
|
||||
}
|
||||
|
||||
|
||||
class TesseractLoggerAdapter(logging.LoggerAdapter):
|
||||
"Prepend [tesseract] to messages emitted from tesseract"
|
||||
"""Prepend [tesseract] to messages emitted from tesseract."""
|
||||
|
||||
def process(self, msg, kwargs):
|
||||
kwargs['extra'] = self.extra
|
||||
@@ -104,19 +108,20 @@ TESSERACT_VERSION_PATTERN = r"""
|
||||
|
||||
|
||||
class TesseractVersion(Version):
|
||||
"Modify standard packaging.Version regex to support Tesseract idiosyncracies."
|
||||
"""Modify standard packaging.Version regex to support Tesseract idiosyncrasies."""
|
||||
|
||||
_regex = re.compile(
|
||||
r"^\s*" + TESSERACT_VERSION_PATTERN + r"\s*$", re.VERBOSE | re.IGNORECASE
|
||||
)
|
||||
|
||||
|
||||
def version() -> str:
|
||||
return get_version('tesseract', regex=r'tesseract\s(.+)')
|
||||
def version() -> Version:
|
||||
return TesseractVersion(get_version('tesseract', regex=r'tesseract\s(.+)'))
|
||||
|
||||
|
||||
def has_thresholding() -> bool:
|
||||
"""Does Tesseract have -c thresholding method capability?"""
|
||||
return version() >= '5.0'
|
||||
return version() >= Version('5.0')
|
||||
|
||||
|
||||
def get_languages() -> set[str]:
|
||||
@@ -171,7 +176,10 @@ def _parse_tesseract_output(binary_output: bytes) -> dict[str, str]:
|
||||
|
||||
|
||||
def get_orientation(
|
||||
input_file: Path, engine_mode: int | None, timeout: float
|
||||
input_file: Path,
|
||||
engine_mode: int | None,
|
||||
timeout: float,
|
||||
omp_thread_limit: int | None = None,
|
||||
) -> OrientationConfidence:
|
||||
args_tesseract = tess_base_args(['osd'], engine_mode) + [
|
||||
'--psm',
|
||||
@@ -181,15 +189,24 @@ def get_orientation(
|
||||
]
|
||||
|
||||
try:
|
||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||
p = run(
|
||||
args_tesseract,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
timeout=timeout,
|
||||
check=True,
|
||||
env=_tesseract_env(omp_thread_limit),
|
||||
)
|
||||
except TimeoutExpired:
|
||||
return OrientationConfidence(angle=0, confidence=0.0)
|
||||
except CalledProcessError as e:
|
||||
tesseract_log_output(e.stdout)
|
||||
tesseract_log_output(e.stderr)
|
||||
# Check both stdout (e.output) and stderr for known non-fatal messages
|
||||
all_output = (e.output or b'') + (e.stderr or b'')
|
||||
if (
|
||||
b'Too few characters. Skipping this page' in e.output
|
||||
or b'Image too large' in e.output
|
||||
b'Too few characters. Skipping this page' in all_output
|
||||
or b'Image too large' in all_output
|
||||
):
|
||||
return OrientationConfidence(0, 0)
|
||||
raise SubprocessOutputError() from e
|
||||
@@ -202,8 +219,24 @@ def get_orientation(
|
||||
return orient_conf
|
||||
|
||||
|
||||
def _is_empty_page_error(exc):
|
||||
if b'Empty page!!' in exc.output: # Tesseract 4.x
|
||||
return True
|
||||
|
||||
return exc.returncode == 1 and (
|
||||
# Tesseract 5.0-5.4 or so
|
||||
exc.output == b''
|
||||
# Tesseract 5.5+
|
||||
or exc.output.startswith(b"Error in boxClipToRectangle: box outside rectangle")
|
||||
)
|
||||
|
||||
|
||||
def get_deskew(
|
||||
input_file: Path, languages: list[str], engine_mode: int | None, timeout: float
|
||||
input_file: Path,
|
||||
languages: list[str],
|
||||
engine_mode: int | None,
|
||||
timeout: float,
|
||||
omp_thread_limit: int | None = None,
|
||||
) -> float:
|
||||
"""Gets angle to deskew this page, in degrees."""
|
||||
args_tesseract = tess_base_args(languages, engine_mode) + [
|
||||
@@ -214,28 +247,35 @@ def get_deskew(
|
||||
]
|
||||
|
||||
try:
|
||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||
p = run(
|
||||
args_tesseract,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
timeout=timeout,
|
||||
check=True,
|
||||
env=_tesseract_env(omp_thread_limit),
|
||||
)
|
||||
except TimeoutExpired:
|
||||
return 0.0
|
||||
except CalledProcessError as e:
|
||||
tesseract_log_output(e.stdout)
|
||||
tesseract_log_output(e.stderr)
|
||||
if b'Empty page!!' in e.output or (
|
||||
e.output == b'' and e.returncode == 1
|
||||
): # Not enough info for a skew angle - Tess 4 and 5 return different errors
|
||||
if _is_empty_page_error(e):
|
||||
# Not enough info for a skew angle
|
||||
return 0.0
|
||||
|
||||
raise SubprocessOutputError() from e
|
||||
|
||||
parsed = _parse_tesseract_output(p.stdout)
|
||||
deskew_radians = float(parsed.get('Deskew angle', 0))
|
||||
deskew_degrees = 180 / pi * deskew_radians
|
||||
log.debug(f"Deskew angle: {deskew_degrees:.3f}")
|
||||
return deskew_degrees
|
||||
|
||||
|
||||
def tesseract_log_output(stream: bytes) -> None:
|
||||
tlog = TesseractLoggerAdapter(
|
||||
log, extra=log.extra if hasattr(log, 'extra') else None # type: ignore
|
||||
log,
|
||||
extra=log.extra if hasattr(log, 'extra') else None, # type: ignore
|
||||
)
|
||||
|
||||
if not stream:
|
||||
@@ -247,9 +287,9 @@ def tesseract_log_output(stream: bytes) -> None:
|
||||
|
||||
lines = text.splitlines()
|
||||
for line in lines:
|
||||
if line.startswith("Tesseract Open Source"):
|
||||
continue
|
||||
elif line.startswith("Warning in pixReadMem"):
|
||||
if line.startswith(
|
||||
("Tesseract Open Source", "Warning in pixReadMem")
|
||||
):
|
||||
continue
|
||||
elif 'diacritics' in line:
|
||||
tlog.warning("lots of diacritics - possibly poor OCR")
|
||||
@@ -280,12 +320,11 @@ def page_timedout(timeout: float) -> None:
|
||||
|
||||
|
||||
def _generate_null_hocr(output_hocr: Path, output_text: Path, image: Path) -> None:
|
||||
"""Produce a .hocr file that reports no text detected on a page that is
|
||||
the same size as the input image."""
|
||||
with Image.open(image) as im:
|
||||
w, h = im.size
|
||||
"""Produce an empty .hocr file.
|
||||
|
||||
output_hocr.write_text(HOCR_TEMPLATE.format(w, h), encoding='utf-8')
|
||||
Ensures page is the same size as the input image.
|
||||
"""
|
||||
output_hocr.write_text('', encoding='utf-8')
|
||||
output_text.write_text('[skipped page]', encoding='utf-8')
|
||||
|
||||
|
||||
@@ -299,9 +338,10 @@ def generate_hocr(
|
||||
tessconfig: list[str],
|
||||
timeout: float,
|
||||
pagesegmode: int,
|
||||
thresholding: int,
|
||||
thresholding: ThresholdingMethod,
|
||||
user_words,
|
||||
user_patterns,
|
||||
omp_thread_limit: int | None = None,
|
||||
) -> None:
|
||||
"""Generate a hOCR file, which must be converted to PDF."""
|
||||
prefix = output_hocr.with_suffix('')
|
||||
@@ -311,7 +351,7 @@ def generate_hocr(
|
||||
if pagesegmode is not None:
|
||||
args_tesseract.extend(['--psm', str(pagesegmode)])
|
||||
|
||||
if thresholding != 0 and has_thresholding():
|
||||
if thresholding != ThresholdingMethod.AUTO and has_thresholding():
|
||||
args_tesseract.extend(['-c', f'thresholding_method={thresholding}'])
|
||||
|
||||
if user_words:
|
||||
@@ -325,7 +365,14 @@ def generate_hocr(
|
||||
args_tesseract.extend([fspath(input_file), fspath(prefix), 'hocr', 'txt'])
|
||||
args_tesseract.extend(tessconfig)
|
||||
try:
|
||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||
p = run(
|
||||
args_tesseract,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
timeout=timeout,
|
||||
check=True,
|
||||
env=_tesseract_env(omp_thread_limit),
|
||||
)
|
||||
stdout = p.stdout
|
||||
except TimeoutExpired:
|
||||
# Generate a HOCR file with no recognized text if tesseract times out
|
||||
@@ -344,7 +391,7 @@ def generate_hocr(
|
||||
tesseract_log_output(stdout)
|
||||
# The sidecar text file will get the suffix .txt; rename it to
|
||||
# whatever caller wants it named
|
||||
if prefix.with_suffix('.txt').exists():
|
||||
with suppress(FileNotFoundError):
|
||||
prefix.with_suffix('.txt').replace(output_text)
|
||||
|
||||
|
||||
@@ -365,9 +412,10 @@ def generate_pdf(
|
||||
tessconfig: list[str],
|
||||
timeout: float,
|
||||
pagesegmode: int,
|
||||
thresholding: int,
|
||||
thresholding: ThresholdingMethod,
|
||||
user_words,
|
||||
user_patterns,
|
||||
omp_thread_limit: int | None = None,
|
||||
) -> None:
|
||||
"""Generate a PDF using Tesseract's internal PDF generator.
|
||||
|
||||
@@ -381,7 +429,7 @@ def generate_pdf(
|
||||
|
||||
args_tesseract.extend(['-c', 'textonly_pdf=1'])
|
||||
|
||||
if thresholding != 0 and has_thresholding():
|
||||
if thresholding != ThresholdingMethod.AUTO and has_thresholding():
|
||||
args_tesseract.extend(['-c', f'thresholding_method={thresholding}'])
|
||||
|
||||
if user_words:
|
||||
@@ -398,9 +446,16 @@ def generate_pdf(
|
||||
args_tesseract.extend([fspath(input_file), fspath(prefix), 'pdf', 'txt'])
|
||||
args_tesseract.extend(tessconfig)
|
||||
try:
|
||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||
p = run(
|
||||
args_tesseract,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
timeout=timeout,
|
||||
check=True,
|
||||
env=_tesseract_env(omp_thread_limit),
|
||||
)
|
||||
stdout = p.stdout
|
||||
if prefix.with_suffix('.txt').exists():
|
||||
with suppress(FileNotFoundError):
|
||||
prefix.with_suffix('.txt').replace(output_text)
|
||||
except TimeoutExpired:
|
||||
page_timedout(timeout)
|
||||
|
||||
@@ -1,53 +1,33 @@
|
||||
# SPDX-FileCopyrightText: 2022 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Interface to unpaper executable"""
|
||||
"""Interface to unpaper executable."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import sys
|
||||
from collections.abc import Iterator
|
||||
from contextlib import contextmanager
|
||||
from decimal import Decimal
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT
|
||||
from typing import Iterator, Union
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
from packaging.version import Version
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
||||
from ocrmypdf.exceptions import SubprocessOutputError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
# unpaper documentation:
|
||||
# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md
|
||||
|
||||
|
||||
if sys.version_info >= (3, 10):
|
||||
from tempfile import TemporaryDirectory
|
||||
else:
|
||||
from tempfile import TemporaryDirectory as _TemporaryDirectory
|
||||
|
||||
class TemporaryDirectory(_TemporaryDirectory):
|
||||
"""Shim to consume ignore_cleanup_errors kwarg on Python 3.9 and older.
|
||||
|
||||
The argument is consumed without action. If users are getting errors related
|
||||
to temporary file cleanup, they should upgrade to Python 3.10 which properly
|
||||
cleans up temporary directories on Windows.
|
||||
|
||||
See: https://github.com/python/cpython/pull/24793
|
||||
"""
|
||||
|
||||
def __init__(self, ignore_cleanup_errors=False, **kwargs):
|
||||
super().__init__(**kwargs)
|
||||
|
||||
del _TemporaryDirectory
|
||||
# https://github.com/Flameeyes/unpaper/blob/main/doc/basic-concepts.md
|
||||
|
||||
|
||||
UNPAPER_IMAGE_PIXEL_LIMIT = 256 * 1024 * 1024
|
||||
|
||||
DecFloat = Union[Decimal, float]
|
||||
DecFloat = Decimal | float
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -67,34 +47,8 @@ class UnpaperImageTooLargeError(Exception):
|
||||
super().__init__(self.message)
|
||||
|
||||
|
||||
def version() -> str:
|
||||
return get_version('unpaper')
|
||||
|
||||
|
||||
SUPPORTED_MODES = {'1', 'L', 'RGB'}
|
||||
|
||||
|
||||
def _convert_image(im: Image.Image) -> tuple[Image.Image, bool]:
|
||||
im_modified = False
|
||||
|
||||
if im.mode not in SUPPORTED_MODES:
|
||||
log.info("Converting image to other colorspace")
|
||||
try:
|
||||
if im.mode == 'P' and len(im.getcolors()) == 2:
|
||||
im = im.convert(mode='1')
|
||||
else:
|
||||
im = im.convert(mode='RGB')
|
||||
except OSError as e:
|
||||
raise MissingDependencyError(
|
||||
"Could not convert image with type " + im.mode
|
||||
) from e
|
||||
else:
|
||||
im_modified = True
|
||||
if im.mode not in SUPPORTED_MODES:
|
||||
raise MissingDependencyError(
|
||||
"Failed to convert image to a supported format."
|
||||
) from None
|
||||
return im, im_modified
|
||||
def version() -> Version:
|
||||
return Version(get_version('unpaper', regex=r'(?m).*?(\d+(\.\d+)(\.\d+)?)'))
|
||||
|
||||
|
||||
@contextmanager
|
||||
@@ -102,21 +56,15 @@ def _setup_unpaper_io(input_file: Path) -> Iterator[tuple[Path, Path, Path]]:
|
||||
with Image.open(input_file) as im:
|
||||
if im.width * im.height >= UNPAPER_IMAGE_PIXEL_LIMIT:
|
||||
raise UnpaperImageTooLargeError(w=im.width, h=im.height)
|
||||
im, im_modified = _convert_image(im)
|
||||
|
||||
with TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
||||
tmppath = Path(tmpdir)
|
||||
if im_modified or input_file.suffix != '.png':
|
||||
input_png = tmppath / 'input.png'
|
||||
im.save(input_png, format='PNG')
|
||||
else:
|
||||
# No changes, PNG input, just use the file we already have
|
||||
input_png = input_file
|
||||
|
||||
# unpaper can write .png too, but it seems to write them slowly
|
||||
# adds a few seconds to test suite - so just use pnm
|
||||
output_pnm = tmppath / 'output.pnm'
|
||||
yield input_png, output_pnm, tmppath
|
||||
with TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
|
||||
tmppath = Path(tmpdir)
|
||||
# No changes, PNG input, just use the file we already have
|
||||
input_png = input_file
|
||||
# unpaper can write .png too, but it seems to write them slowly
|
||||
# adds a few seconds to test suite - so just use pnm
|
||||
output_pnm = tmppath / 'output.pnm'
|
||||
yield input_png, output_pnm, tmppath
|
||||
|
||||
|
||||
def run_unpaper(
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Interface to verapdf executable."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE
|
||||
from typing import NamedTuple
|
||||
|
||||
from packaging.version import Version
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class ValidationResult(NamedTuple):
|
||||
"""Result of PDF/A validation."""
|
||||
|
||||
valid: bool
|
||||
failed_rules: int
|
||||
message: str
|
||||
|
||||
|
||||
def version() -> Version:
|
||||
"""Get verapdf version."""
|
||||
return Version(get_version('verapdf', regex=r'veraPDF (\d+(\.\d+)*)'))
|
||||
|
||||
|
||||
def available() -> bool:
|
||||
"""Check if verapdf is available."""
|
||||
try:
|
||||
version()
|
||||
except MissingDependencyError:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def output_type_to_flavour(output_type: str) -> str:
|
||||
"""Map OCRmyPDF output_type to verapdf flavour.
|
||||
|
||||
Args:
|
||||
output_type: One of 'pdfa', 'pdfa-1', 'pdfa-2', 'pdfa-3'
|
||||
|
||||
Returns:
|
||||
verapdf flavour string like '1b', '2b', '3b'
|
||||
"""
|
||||
mapping = {
|
||||
'pdfa': '2b',
|
||||
'pdfa-1': '1b',
|
||||
'pdfa-2': '2b',
|
||||
'pdfa-3': '3b',
|
||||
}
|
||||
return mapping.get(output_type, '2b')
|
||||
|
||||
|
||||
def validate(input_file: Path, flavour: str) -> ValidationResult:
|
||||
"""Validate a PDF against a PDF/A profile.
|
||||
|
||||
Args:
|
||||
input_file: Path to PDF file to validate
|
||||
flavour: verapdf flavour (1a, 1b, 2a, 2b, 2u, 3a, 3b, 3u)
|
||||
|
||||
Returns:
|
||||
ValidationResult with validation status
|
||||
"""
|
||||
args = [
|
||||
'verapdf',
|
||||
'--format',
|
||||
'json',
|
||||
'--flavour',
|
||||
flavour,
|
||||
str(input_file),
|
||||
]
|
||||
|
||||
try:
|
||||
proc = run(args, stdout=PIPE, stderr=PIPE, check=False)
|
||||
except FileNotFoundError as e:
|
||||
raise MissingDependencyError('verapdf') from e
|
||||
|
||||
try:
|
||||
result = json.loads(proc.stdout)
|
||||
jobs = result.get('report', {}).get('jobs', [])
|
||||
if not jobs:
|
||||
return ValidationResult(False, -1, 'No validation jobs in result')
|
||||
validation_results = jobs[0].get('validationResult', [])
|
||||
if not validation_results:
|
||||
return ValidationResult(False, -1, 'No validation result in output')
|
||||
validation_result = validation_results[0]
|
||||
details = validation_result.get('details', {})
|
||||
failed_rules = details.get('failedRules', 0)
|
||||
|
||||
if failed_rules == 0:
|
||||
return ValidationResult(True, 0, 'PDF/A validation passed')
|
||||
else:
|
||||
return ValidationResult(
|
||||
False,
|
||||
failed_rules,
|
||||
f'PDF/A validation failed with {failed_rules} rule violations',
|
||||
)
|
||||
except (json.JSONDecodeError, KeyError, TypeError) as e:
|
||||
log.debug('Failed to parse verapdf output: %s', e)
|
||||
return ValidationResult(False, -1, f'Failed to parse verapdf output: {e}')
|
||||
+488
-191
@@ -7,65 +7,191 @@ from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from contextlib import suppress
|
||||
from dataclasses import dataclass
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from ocrmypdf.hocrtransform import OcrElement
|
||||
|
||||
from pikepdf import (
|
||||
Dictionary,
|
||||
Name,
|
||||
Object,
|
||||
Operator,
|
||||
Page,
|
||||
Pdf,
|
||||
PdfError,
|
||||
PdfMatrix,
|
||||
Stream,
|
||||
parse_content_stream,
|
||||
unparse_content_stream,
|
||||
)
|
||||
|
||||
from ocrmypdf._jobcontext import PdfContext
|
||||
from ocrmypdf._options import ProcessingMode
|
||||
from ocrmypdf._pipeline import VECTOR_PAGE_DPI
|
||||
|
||||
|
||||
class RenderMode(Enum):
|
||||
"""Controls where the OCR text layer is placed relative to page content.
|
||||
|
||||
ON_TOP: Text layer renders above page content (reserved for future use).
|
||||
UNDERNEATH: Text layer renders below page content (current default behavior).
|
||||
"""
|
||||
|
||||
ON_TOP = 0
|
||||
UNDERNEATH = 1
|
||||
|
||||
|
||||
@dataclass
|
||||
class Fpdf2PageInfo:
|
||||
"""Information needed to render and graft an fpdf2 page."""
|
||||
|
||||
pageno: int
|
||||
hocr_path: Path
|
||||
dpi: float
|
||||
autorotate_correction: int
|
||||
emplaced_page: bool
|
||||
|
||||
|
||||
@dataclass
|
||||
class Fpdf2ParsedPage:
|
||||
"""Parsed page data ready for fpdf2 rendering."""
|
||||
|
||||
pageno: int
|
||||
ocr_tree: OcrElement
|
||||
dpi: float
|
||||
autorotate_correction: int
|
||||
emplaced_page: bool
|
||||
|
||||
|
||||
# Alias for backward compatibility with plan documentation
|
||||
Fpdf2DirectPage = Fpdf2ParsedPage
|
||||
|
||||
|
||||
def _compute_text_misalignment(
|
||||
content_rotation: int, autorotate_correction: int, emplaced_page: bool
|
||||
) -> int:
|
||||
"""Compute rotation needed to align text layer with page content.
|
||||
|
||||
Args:
|
||||
content_rotation: Original page /Rotate value (degrees).
|
||||
autorotate_correction: Rotation applied during rasterization (degrees).
|
||||
emplaced_page: Whether the page content was replaced with rasterized image.
|
||||
|
||||
Returns:
|
||||
Rotation in degrees to apply to text layer to align with content.
|
||||
"""
|
||||
if emplaced_page:
|
||||
# New image is upright after autorotation was applied
|
||||
content_rotation = autorotate_correction
|
||||
text_rotation = autorotate_correction
|
||||
return (text_rotation - content_rotation) % 360
|
||||
|
||||
|
||||
def _compute_page_rotation(
|
||||
content_rotation: int, autorotate_correction: int, emplaced_page: bool
|
||||
) -> int:
|
||||
"""Compute final page /Rotate value after grafting.
|
||||
|
||||
Args:
|
||||
content_rotation: Original page /Rotate value (degrees).
|
||||
autorotate_correction: Rotation applied during rasterization (degrees).
|
||||
emplaced_page: Whether the page content was replaced with rasterized image.
|
||||
|
||||
Returns:
|
||||
Final /Rotate value for the page.
|
||||
"""
|
||||
if emplaced_page:
|
||||
content_rotation = autorotate_correction
|
||||
return (content_rotation - autorotate_correction) % 360
|
||||
|
||||
|
||||
def _build_text_layer_ctm(
|
||||
text_width: float,
|
||||
text_height: float,
|
||||
page_width: float,
|
||||
page_height: float,
|
||||
page_origin_x: float,
|
||||
page_origin_y: float,
|
||||
text_rotation: int,
|
||||
):
|
||||
"""Build transformation matrix to align text layer with page content.
|
||||
|
||||
Args:
|
||||
text_width: Width of text layer mediabox.
|
||||
text_height: Height of text layer mediabox.
|
||||
page_width: Width of target page mediabox.
|
||||
page_height: Height of target page mediabox.
|
||||
page_origin_x: X origin of target page mediabox.
|
||||
page_origin_y: Y origin of target page mediabox.
|
||||
text_rotation: Rotation in degrees (clockwise) to apply to text layer.
|
||||
|
||||
Returns:
|
||||
pikepdf.Matrix transformation matrix, or None if no rotation needed.
|
||||
"""
|
||||
if text_rotation == 0:
|
||||
return None
|
||||
|
||||
from pikepdf import Matrix
|
||||
|
||||
wt, ht = text_width, text_height
|
||||
|
||||
# Center text, rotate, scale to fit page, then position at page origin
|
||||
translate = Matrix().translated(-wt / 2, -ht / 2)
|
||||
untranslate = Matrix().translated(page_width / 2, page_height / 2)
|
||||
corner = Matrix().translated(page_origin_x, page_origin_y)
|
||||
|
||||
# Negate rotation because input is clockwise angle
|
||||
rotate = Matrix().rotated(-text_rotation % 360)
|
||||
|
||||
# Swap dimensions if 90 or 270 degree rotation
|
||||
if text_rotation in (90, 270):
|
||||
wt, ht = ht, wt
|
||||
|
||||
# Scale to fit page dimensions
|
||||
scale_x = page_width / wt if wt else 1.0
|
||||
scale_y = page_height / ht if ht else 1.0
|
||||
scale = Matrix().scaled(scale_x, scale_y)
|
||||
|
||||
return translate @ rotate @ scale @ untranslate @ corner
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
MAX_REPLACE_PAGES = 100
|
||||
|
||||
|
||||
def _ensure_dictionary(obj, name):
|
||||
def _ensure_dictionary(obj: Dictionary | Stream, name: Name):
|
||||
if name not in obj:
|
||||
obj[name] = Dictionary({})
|
||||
return obj[name]
|
||||
|
||||
|
||||
def _update_resources(*, obj, font, font_key, procset):
|
||||
"""Update this obj's fonts with a reference to the Glyphless font.
|
||||
|
||||
obj can be a page or Form XObject.
|
||||
"""
|
||||
|
||||
resources = _ensure_dictionary(obj, Name.Resources)
|
||||
fonts = _ensure_dictionary(resources, Name.Font)
|
||||
if font_key is not None and font_key not in fonts:
|
||||
fonts[font_key] = font
|
||||
|
||||
# Reassign /ProcSet to one that just lists everything - ProcSet is
|
||||
# obsolete and doesn't matter but recommended for old viewer support
|
||||
if procset:
|
||||
resources['/ProcSet'] = procset
|
||||
|
||||
|
||||
def strip_invisible_text(pdf, page):
|
||||
def strip_invisible_text(pdf: Pdf, page: Page):
|
||||
stream = []
|
||||
in_text_obj = False
|
||||
render_mode = 0
|
||||
render_mode_stack = []
|
||||
text_objects = []
|
||||
|
||||
for operands, operator in parse_content_stream(page, ''):
|
||||
if operator == Operator('Tr'):
|
||||
render_mode = operands[0]
|
||||
|
||||
if operator == Operator('q'):
|
||||
render_mode_stack.append(render_mode)
|
||||
|
||||
if operator == Operator('Q'):
|
||||
# IndexError is raised if stack is empty; try to carry on
|
||||
with suppress(IndexError):
|
||||
render_mode = render_mode_stack.pop()
|
||||
|
||||
if not in_text_obj:
|
||||
if operator == Operator('BT'):
|
||||
in_text_obj = True
|
||||
render_mode = 0
|
||||
text_objects.append((operands, operator))
|
||||
else:
|
||||
stream.append((operands, operator))
|
||||
else:
|
||||
if operator == Operator('Tr'):
|
||||
render_mode = operands[0]
|
||||
text_objects.append((operands, operator))
|
||||
if operator == Operator('ET'):
|
||||
in_text_obj = False
|
||||
@@ -80,34 +206,50 @@ def strip_invisible_text(pdf, page):
|
||||
class OcrGrafter:
|
||||
"""Manages grafting text-only PDFs onto regular PDFs."""
|
||||
|
||||
def __init__(self, context):
|
||||
def __init__(self, context: PdfContext):
|
||||
self.context = context
|
||||
self.path_base = context.origin
|
||||
|
||||
self.pdf_base = Pdf.open(self.path_base)
|
||||
self.font, self.font_key = None, None
|
||||
|
||||
self.pdfinfo = context.pdfinfo
|
||||
self.output_file = context.get_path('graft_layers.pdf')
|
||||
|
||||
self.procset = self.pdf_base.make_indirect(
|
||||
Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]')
|
||||
)
|
||||
|
||||
self.emplacements = 1
|
||||
self.interim_count = 0
|
||||
self.render_mode = RenderMode.UNDERNEATH
|
||||
|
||||
# Check renderer type
|
||||
pdf_renderer = context.options.pdf_renderer
|
||||
self.use_sandwich_renderer = pdf_renderer == 'sandwich'
|
||||
|
||||
# For fpdf2: accumulate pages before rendering
|
||||
self.fpdf2_hocr_pages: list[Fpdf2PageInfo] = []
|
||||
self.fpdf2_parsed_pages: list[Fpdf2ParsedPage] = []
|
||||
|
||||
def graft_page(
|
||||
self,
|
||||
*,
|
||||
pageno: int,
|
||||
image: Path | None,
|
||||
textpdf: Path | None,
|
||||
ocr_output: Path | None,
|
||||
ocr_tree: OcrElement | None,
|
||||
autorotate_correction: int,
|
||||
):
|
||||
if textpdf and not self.font:
|
||||
self.font, self.font_key = self._find_font(textpdf)
|
||||
"""Graft OCR output onto a page of the base PDF.
|
||||
|
||||
Args:
|
||||
pageno: Zero-based page number.
|
||||
image: Path to the visible page image PDF, or None if not replacing.
|
||||
ocr_output: Path to OCR output file. For fpdf2 renderer this is an
|
||||
hOCR file; for sandwich renderer this is a text-only PDF.
|
||||
ocr_tree: OCR tree for fpdf2 renderer.
|
||||
autorotate_correction: Orientation correction in degrees (0, 90, 180, 270).
|
||||
"""
|
||||
if ocr_output and ocr_tree:
|
||||
raise ValueError(
|
||||
'Cannot specify both ocr_output and ocr_tree for fpdf2 renderer'
|
||||
)
|
||||
# Handle image emplacement first
|
||||
emplaced_page = False
|
||||
content_rotation = self.pdfinfo[pageno].rotation
|
||||
path_image = Path(image).resolve() if image else None
|
||||
@@ -120,190 +262,345 @@ class OcrGrafter:
|
||||
foreign_image_page = pdf_image.pages[0]
|
||||
self.pdf_base.pages.append(foreign_image_page)
|
||||
local_image_page = self.pdf_base.pages[-1]
|
||||
self.pdf_base.pages[pageno].emplace(local_image_page)
|
||||
self.pdf_base.pages[pageno].emplace(
|
||||
local_image_page, retain=(Name.Parent,)
|
||||
)
|
||||
del self.pdf_base.pages[-1]
|
||||
emplaced_page = True
|
||||
|
||||
# Calculate if the text is misaligned compared to the content
|
||||
if emplaced_page:
|
||||
content_rotation = autorotate_correction
|
||||
text_rotation = autorotate_correction
|
||||
text_misaligned = (text_rotation - content_rotation) % 360
|
||||
log.debug(
|
||||
f"Text rotation: (text, autorotate, content) -> text misalignment = "
|
||||
f"({text_rotation}, {autorotate_correction}, {content_rotation}) -> {text_misaligned}"
|
||||
)
|
||||
|
||||
if textpdf and self.font:
|
||||
# Graft the text layer onto this page, whether new or old, possibly
|
||||
# rotating the text layer by the amount is misaligned.
|
||||
strip_old = self.context.options.redo_ocr
|
||||
self._graft_text_layer(
|
||||
page_num=pageno + 1,
|
||||
textpdf=textpdf,
|
||||
font=self.font,
|
||||
font_key=self.font_key,
|
||||
text_rotation=text_misaligned,
|
||||
procset=self.procset,
|
||||
strip_old_text=strip_old,
|
||||
)
|
||||
|
||||
# Correct the overall page rotation if needed, now that the text and content
|
||||
# are aligned
|
||||
page_rotation = (content_rotation - autorotate_correction) % 360
|
||||
self.pdf_base.pages[pageno].Rotate = page_rotation
|
||||
log.debug(
|
||||
f"Page rotation: (content, auto) -> page = "
|
||||
f"({content_rotation}, {autorotate_correction}) -> {page_rotation}"
|
||||
)
|
||||
if self.emplacements % MAX_REPLACE_PAGES == 0:
|
||||
self.save_and_reload()
|
||||
|
||||
def save_and_reload(self):
|
||||
"""Save and reload the Pdf.
|
||||
|
||||
This will keep a lid on our memory usage for very large files. Attach
|
||||
the font to page 1 even if page 1 doesn't use it, so we have a way to get it
|
||||
back.
|
||||
"""
|
||||
|
||||
page0 = self.pdf_base.pages[0]
|
||||
_update_resources(
|
||||
obj=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
||||
)
|
||||
|
||||
# We cannot read and write the same file, that will corrupt it
|
||||
# but we don't to keep more copies than we need to. Delete intermediates.
|
||||
# {interim_count} is the opened file we were updating
|
||||
# {interim_count - 1} can be deleted
|
||||
# {interim_count + 1} is the new file will produce and open
|
||||
old_file = self.output_file.with_suffix(f'.working{self.interim_count - 1}.pdf')
|
||||
if not self.context.options.keep_temporary_files:
|
||||
with suppress(FileNotFoundError):
|
||||
old_file.unlink()
|
||||
|
||||
next_file = self.output_file.with_suffix(
|
||||
f'.working{self.interim_count + 1}.pdf'
|
||||
)
|
||||
self.pdf_base.save(next_file)
|
||||
self.pdf_base.close()
|
||||
|
||||
self.pdf_base = Pdf.open(next_file)
|
||||
self.procset = self.pdf_base.pages[0].Resources.ProcSet
|
||||
self.font, self.font_key = None, None # Ensure we reacquire this information
|
||||
self.interim_count += 1
|
||||
if self.use_sandwich_renderer:
|
||||
# Sandwich renderer: graft pre-rendered PDF immediately
|
||||
if ocr_output:
|
||||
text_misaligned = _compute_text_misalignment(
|
||||
content_rotation, autorotate_correction, emplaced_page
|
||||
)
|
||||
self._graft_sandwich_text_layer(
|
||||
pageno=pageno,
|
||||
textpdf=ocr_output,
|
||||
text_rotation=text_misaligned,
|
||||
)
|
||||
page_rotation = _compute_page_rotation(
|
||||
content_rotation, autorotate_correction, emplaced_page
|
||||
)
|
||||
self.pdf_base.pages[pageno].Rotate = page_rotation
|
||||
else:
|
||||
# fpdf2 renderer: accumulate page info for batch rendering.
|
||||
# The hOCR coordinates are in the corrected (upright) coordinate system.
|
||||
# We store autorotate_correction and emplaced_page to set the final
|
||||
# page /Rotate tag after grafting.
|
||||
if ocr_tree:
|
||||
self.fpdf2_parsed_pages.append(
|
||||
Fpdf2ParsedPage(
|
||||
ocr_tree=ocr_tree,
|
||||
pageno=pageno,
|
||||
autorotate_correction=autorotate_correction,
|
||||
emplaced_page=emplaced_page,
|
||||
dpi=self.pdfinfo[pageno].dpi.to_scalar(),
|
||||
)
|
||||
)
|
||||
if ocr_output:
|
||||
self.fpdf2_hocr_pages.append(
|
||||
Fpdf2PageInfo(
|
||||
hocr_path=ocr_output,
|
||||
pageno=pageno,
|
||||
autorotate_correction=autorotate_correction,
|
||||
emplaced_page=emplaced_page,
|
||||
dpi=self.pdfinfo[pageno].dpi.to_scalar(),
|
||||
)
|
||||
)
|
||||
|
||||
def finalize(self):
|
||||
# Can have hocr OR parsed pages OR neither (no OCR), but not both
|
||||
assert not (
|
||||
self.fpdf2_hocr_pages and self.fpdf2_parsed_pages
|
||||
), "Can't have both hocr and ocrtree pages"
|
||||
|
||||
if self.fpdf2_hocr_pages:
|
||||
# Render all pages with fpdf2, then graft
|
||||
parsed_pages = self._parse_hocr_pages()
|
||||
self.fpdf2_parsed_pages = parsed_pages
|
||||
|
||||
if self.fpdf2_parsed_pages:
|
||||
self._render_and_graft_fpdf2_pages()
|
||||
|
||||
self.pdf_base.save(self.output_file)
|
||||
self.pdf_base.close()
|
||||
return self.output_file
|
||||
|
||||
def _find_font(self, text):
|
||||
"""Copy a font from the filename text into pdf_base"""
|
||||
def _parse_hocr_pages(self):
|
||||
"""Render all pages to multi-page PDF with shared fonts, then graft."""
|
||||
from ocrmypdf.hocrtransform.hocr_parser import HocrParser
|
||||
|
||||
font, font_key = None, None
|
||||
possible_font_names = ('/f-0-0', '/F1')
|
||||
try:
|
||||
with Pdf.open(text) as pdf_text:
|
||||
try:
|
||||
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
||||
except (AttributeError, IndexError, KeyError):
|
||||
return None, None
|
||||
pdf_text_font = None
|
||||
for f in possible_font_names:
|
||||
pdf_text_font = pdf_text_fonts.get(f, None)
|
||||
if pdf_text_font is not None:
|
||||
font_key = f
|
||||
break
|
||||
if pdf_text_font:
|
||||
font = self.pdf_base.copy_foreign(pdf_text_font)
|
||||
return font, font_key
|
||||
except (FileNotFoundError, PdfError):
|
||||
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
||||
return None, None
|
||||
log.info(
|
||||
"Parsing %d pages with HocrParser",
|
||||
len(self.fpdf2_hocr_pages),
|
||||
)
|
||||
|
||||
def _graft_text_layer(
|
||||
# Parse all hOCR files and collect OcrElements
|
||||
pages_data: list[Fpdf2ParsedPage] = []
|
||||
for page_info in self.fpdf2_hocr_pages:
|
||||
if page_info.hocr_path.stat().st_size == 0:
|
||||
continue # Skip empty pages
|
||||
|
||||
# Parse hOCR to OcrElement
|
||||
parser = HocrParser(page_info.hocr_path)
|
||||
ocr_tree = parser.parse()
|
||||
|
||||
# Use DPI from hOCR (scan_res) which reflects actual rasterization DPI.
|
||||
# Fall back to pdfinfo DPI or VECTOR_PAGE_DPI for vector-only pages.
|
||||
effective_dpi = ocr_tree.dpi or page_info.dpi or float(VECTOR_PAGE_DPI)
|
||||
pages_data.append(
|
||||
Fpdf2ParsedPage(
|
||||
pageno=page_info.pageno,
|
||||
ocr_tree=ocr_tree,
|
||||
dpi=effective_dpi,
|
||||
autorotate_correction=page_info.autorotate_correction,
|
||||
emplaced_page=page_info.emplaced_page,
|
||||
)
|
||||
)
|
||||
|
||||
return pages_data
|
||||
|
||||
def _render_and_graft_fpdf2_pages(self):
|
||||
font_dir = Path(__file__).parent / "data"
|
||||
|
||||
# Render all pages to single PDF
|
||||
multi_page_pdf_path = self.context.get_path('fpdf2_multipage.pdf')
|
||||
|
||||
from ocrmypdf.font import MultiFontManager
|
||||
from ocrmypdf.fpdf_renderer import Fpdf2MultiPageRenderer
|
||||
|
||||
multi_font_manager = MultiFontManager(font_dir)
|
||||
# Build renderer input as (pageno, ocr_tree, dpi) tuples
|
||||
renderer_pages_data = [
|
||||
(parsed.pageno, parsed.ocr_tree, parsed.dpi)
|
||||
for parsed in self.fpdf2_parsed_pages
|
||||
]
|
||||
renderer = Fpdf2MultiPageRenderer(
|
||||
pages_data=renderer_pages_data,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=True,
|
||||
)
|
||||
|
||||
renderer.render(multi_page_pdf_path)
|
||||
|
||||
# Now graft each page from the multi-page PDF
|
||||
with Pdf.open(multi_page_pdf_path) as pdf_text:
|
||||
for idx, parsed in enumerate(self.fpdf2_parsed_pages):
|
||||
# Copy page from multi-page PDF
|
||||
text_page = pdf_text.pages[idx]
|
||||
|
||||
content_rotation = self.pdfinfo[parsed.pageno].rotation
|
||||
text_misaligned = _compute_text_misalignment(
|
||||
content_rotation,
|
||||
parsed.autorotate_correction,
|
||||
parsed.emplaced_page,
|
||||
)
|
||||
self._graft_fpdf2_text_layer(parsed.pageno, text_page, text_misaligned)
|
||||
|
||||
page_rotation = _compute_page_rotation(
|
||||
content_rotation,
|
||||
parsed.autorotate_correction,
|
||||
parsed.emplaced_page,
|
||||
)
|
||||
self.pdf_base.pages[parsed.pageno].Rotate = page_rotation
|
||||
|
||||
# Clean up multi-page PDF if not keeping temp files
|
||||
if not self.context.options.keep_temporary_files:
|
||||
with suppress(FileNotFoundError):
|
||||
multi_page_pdf_path.unlink()
|
||||
|
||||
def _graft_fpdf2_text_layer(self, pageno: int, text_page: Page, text_rotation: int):
|
||||
"""Graft a single text page onto the base PDF.
|
||||
|
||||
Similar to existing _graft_text_layer but works with
|
||||
already-rendered pikepdf Page instead of file path.
|
||||
|
||||
Args:
|
||||
pageno: Zero-based page number.
|
||||
text_page: The text-only PDF page to graft.
|
||||
text_rotation: Rotation to apply to align text with content (degrees).
|
||||
"""
|
||||
from pikepdf import Array
|
||||
|
||||
base_page = self.pdf_base.pages[pageno]
|
||||
|
||||
# Extract content stream from text_page
|
||||
text_contents = text_page.Contents.read_bytes()
|
||||
|
||||
# Get the mediabox from the text page
|
||||
mediabox = Array([float(x) for x in text_page.mediabox]) # type: ignore[misc]
|
||||
wt = float(mediabox[2]) - float(mediabox[0])
|
||||
ht = float(mediabox[3]) - float(mediabox[1])
|
||||
|
||||
# Get base page mediabox
|
||||
base_mediabox = base_page.mediabox
|
||||
wp = float(base_mediabox[2]) - float(base_mediabox[0])
|
||||
hp = float(base_mediabox[3]) - float(base_mediabox[1])
|
||||
|
||||
# Create Form XObject from text page content
|
||||
base_resources = _ensure_dictionary(base_page.obj, Name.Resources)
|
||||
base_xobjs = _ensure_dictionary(base_resources, Name.XObject)
|
||||
text_xobj_name = Name.random(prefix="OCR-")
|
||||
xobj = self.pdf_base.make_stream(text_contents)
|
||||
base_xobjs[text_xobj_name] = xobj
|
||||
xobj.Type = Name.XObject
|
||||
xobj.Subtype = Name.Form
|
||||
xobj.FormType = 1
|
||||
xobj.BBox = mediabox
|
||||
|
||||
# Copy resources from text page's Resources to xobj
|
||||
# We need to handle this carefully since text_page is from a foreign PDF
|
||||
if hasattr(text_page, 'Resources') and text_page.Resources:
|
||||
# Create empty Resources dictionary for xobj
|
||||
xobj_resources = _ensure_dictionary(xobj, Name.Resources)
|
||||
|
||||
# Copy fonts if they exist
|
||||
if Name.Font in text_page.Resources:
|
||||
xobj_fonts = _ensure_dictionary(xobj_resources, Name.Font)
|
||||
text_fonts = text_page.Resources[Name.Font]
|
||||
# Copy each font from the foreign PDF
|
||||
for font_name, font_obj in text_fonts.items():
|
||||
xobj_fonts[font_name] = self.pdf_base.copy_foreign(font_obj)
|
||||
|
||||
# Copy ExtGState (graphics state) if it exists - needed for transparency
|
||||
if Name.ExtGState in text_page.Resources:
|
||||
xobj_extstates = _ensure_dictionary(xobj_resources, Name.ExtGState)
|
||||
text_extstates = text_page.Resources[Name.ExtGState]
|
||||
# Copy each graphics state from the foreign PDF
|
||||
for gs_name, gs_obj in text_extstates.items():
|
||||
xobj_extstates[gs_name] = self.pdf_base.copy_foreign(gs_obj)
|
||||
|
||||
# Build transformation matrix for rotation and scaling
|
||||
ctm = _build_text_layer_ctm(
|
||||
wt,
|
||||
ht,
|
||||
wp,
|
||||
hp,
|
||||
float(base_mediabox[0]),
|
||||
float(base_mediabox[1]),
|
||||
text_rotation,
|
||||
)
|
||||
if ctm is not None:
|
||||
pdf_draw_xobj = (
|
||||
(b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'Q\n'
|
||||
)
|
||||
else:
|
||||
pdf_draw_xobj = b'q\n' + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
||||
|
||||
new_text_layer = Stream(self.pdf_base, pdf_draw_xobj)
|
||||
|
||||
# Strip old invisible text if redo mode is enabled
|
||||
if self.context.options.mode == ProcessingMode.redo:
|
||||
strip_invisible_text(self.pdf_base, base_page)
|
||||
|
||||
# Add text layer to base page
|
||||
base_page.contents_coalesce()
|
||||
base_page.contents_add(
|
||||
new_text_layer, prepend=self.render_mode == RenderMode.UNDERNEATH
|
||||
)
|
||||
base_page.contents_coalesce()
|
||||
|
||||
def _graft_sandwich_text_layer(
|
||||
self,
|
||||
*,
|
||||
page_num: int,
|
||||
pageno: int,
|
||||
textpdf: Path,
|
||||
font: Object,
|
||||
font_key: Object,
|
||||
procset: Object,
|
||||
text_rotation: int,
|
||||
strip_old_text: bool,
|
||||
):
|
||||
"""Insert the text layer from text page 0 on to pdf_base at page_num"""
|
||||
"""Graft a pre-rendered text-only PDF onto the base PDF.
|
||||
|
||||
# pylint: disable=invalid-name
|
||||
This is used by the sandwich renderer which generates PDFs directly
|
||||
from Tesseract rather than going through hOCR.
|
||||
"""
|
||||
from pikepdf import PdfError
|
||||
|
||||
log.debug("Grafting")
|
||||
log.debug("Grafting sandwich text layer")
|
||||
if Path(textpdf).stat().st_size == 0:
|
||||
return
|
||||
|
||||
# This is a pointer indicating a specific page in the base file
|
||||
with Pdf.open(textpdf) as pdf_text:
|
||||
pdf_text_contents = pdf_text.pages[0].Contents.read_bytes()
|
||||
try:
|
||||
with Pdf.open(textpdf) as pdf_text:
|
||||
pdf_text_contents = pdf_text.pages[0].Contents.read_bytes()
|
||||
|
||||
base_page = self.pdf_base.pages.p(page_num)
|
||||
base_page = self.pdf_base.pages[pageno]
|
||||
|
||||
# The text page always will be oriented up by this stage but the original
|
||||
# content may have a rotation applied. Wrap the text stream with a rotation
|
||||
# so it will be oriented the same way as the rest of the page content.
|
||||
# (Previous versions OCRmyPDF rotated the content layer to match the text.)
|
||||
mediabox = [float(pdf_text.pages[0].MediaBox[v]) for v in range(4)]
|
||||
wt, ht = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
||||
# Get font from the text PDF
|
||||
pdf_text_fonts = pdf_text.pages[0].Resources.get(
|
||||
Name.Font, Dictionary()
|
||||
)
|
||||
font = None
|
||||
font_key = None
|
||||
for f in ('/f-0-0', '/F1'):
|
||||
pdf_text_font = pdf_text_fonts.get(f, None)
|
||||
if pdf_text_font is not None:
|
||||
font_key = Name(f)
|
||||
font = self.pdf_base.copy_foreign(pdf_text_font)
|
||||
break
|
||||
|
||||
mediabox = [float(base_page.MediaBox[v]) for v in range(4)]
|
||||
wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
||||
# Get mediabox dimensions for rotation calculations
|
||||
mediabox = pdf_text.pages[0].mediabox
|
||||
wt = float(mediabox[2]) - float(mediabox[0])
|
||||
ht = float(mediabox[3]) - float(mediabox[1])
|
||||
|
||||
translate = PdfMatrix().translated(-wt / 2, -ht / 2)
|
||||
untranslate = PdfMatrix().translated(wp / 2, hp / 2)
|
||||
corner = PdfMatrix().translated(mediabox[0], mediabox[1])
|
||||
# -rotation because the input is a clockwise angle and this formula
|
||||
# uses CCW
|
||||
text_rotation = -text_rotation % 360
|
||||
rotate = PdfMatrix().rotated(text_rotation)
|
||||
base_mediabox = base_page.mediabox
|
||||
wp = float(base_mediabox[2]) - float(base_mediabox[0])
|
||||
hp = float(base_mediabox[3]) - float(base_mediabox[1])
|
||||
|
||||
# Because of rounding of DPI, we might get a text layer that is not
|
||||
# identically sized to the target page. Scale to adjust. Normally this
|
||||
# is within 0.998.
|
||||
if text_rotation in (90, 270):
|
||||
wt, ht = ht, wt
|
||||
scale_x = wp / wt
|
||||
scale_y = hp / ht
|
||||
# Build transformation matrix for rotation and scaling
|
||||
ctm = _build_text_layer_ctm(
|
||||
wt,
|
||||
ht,
|
||||
wp,
|
||||
hp,
|
||||
float(base_mediabox[0]),
|
||||
float(base_mediabox[1]),
|
||||
text_rotation,
|
||||
)
|
||||
log.debug("Grafting with ctm %r", ctm)
|
||||
|
||||
# log.debug('%r', scale_x, scale_y)
|
||||
scale = PdfMatrix().scaled(scale_x, scale_y)
|
||||
# Create Form XObject
|
||||
base_resources = _ensure_dictionary(base_page.obj, Name.Resources)
|
||||
base_xobjs = _ensure_dictionary(base_resources, Name.XObject)
|
||||
text_xobj_name = Name.random(prefix="OCR-")
|
||||
xobj = self.pdf_base.make_stream(pdf_text_contents)
|
||||
base_xobjs[text_xobj_name] = xobj
|
||||
xobj.Type = Name.XObject
|
||||
xobj.Subtype = Name.Form
|
||||
xobj.FormType = 1
|
||||
xobj.BBox = base_mediabox
|
||||
|
||||
# Translate the text so it is centered at (0, 0), rotate it there, adjust
|
||||
# for a size different between initial and text PDF, then untranslate, and
|
||||
# finally move the lower left corner to match the mediabox
|
||||
ctm = translate @ rotate @ scale @ untranslate @ corner
|
||||
# Add font to xobj resources
|
||||
if font_key is not None and font is not None:
|
||||
xobj_resources = _ensure_dictionary(xobj, Name.Resources)
|
||||
xobj_fonts = _ensure_dictionary(xobj_resources, Name.Font)
|
||||
if font_key not in xobj_fonts:
|
||||
xobj_fonts[font_key] = font
|
||||
|
||||
base_resources = _ensure_dictionary(base_page, Name.Resources)
|
||||
base_xobjs = _ensure_dictionary(base_resources, Name.XObject)
|
||||
text_xobj_name = Name.random(prefix="OCR-")
|
||||
xobj = self.pdf_base.make_stream(pdf_text_contents)
|
||||
base_xobjs[text_xobj_name] = xobj
|
||||
xobj.Type = Name.XObject
|
||||
xobj.Subtype = Name.Form
|
||||
xobj.FormType = 1
|
||||
xobj.BBox = mediabox
|
||||
_update_resources(
|
||||
obj=xobj, font=font, font_key=font_key, procset=[Name.PDF]
|
||||
)
|
||||
if ctm is not None:
|
||||
pdf_draw_xobj = (
|
||||
(b'q %s cm\n' % ctm.encode())
|
||||
+ (b'%s Do\n' % text_xobj_name)
|
||||
+ b'\nQ\n'
|
||||
)
|
||||
else:
|
||||
pdf_draw_xobj = b'q\n' + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
||||
new_text_layer = Stream(self.pdf_base, pdf_draw_xobj)
|
||||
|
||||
pdf_draw_xobj = (
|
||||
(b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
||||
)
|
||||
new_text_layer = Stream(self.pdf_base, pdf_draw_xobj)
|
||||
if self.context.options.mode == ProcessingMode.redo:
|
||||
strip_invisible_text(self.pdf_base, base_page)
|
||||
base_page.contents_coalesce()
|
||||
base_page.contents_add(
|
||||
new_text_layer, prepend=self.render_mode == RenderMode.UNDERNEATH
|
||||
)
|
||||
base_page.contents_coalesce()
|
||||
|
||||
if strip_old_text:
|
||||
strip_invisible_text(self.pdf_base, base_page)
|
||||
|
||||
base_page.contents_add(new_text_layer, prepend=True)
|
||||
|
||||
_update_resources(
|
||||
obj=base_page, font=font, font_key=font_key, procset=procset
|
||||
)
|
||||
# Add font to page resources
|
||||
if font_key is not None and font is not None:
|
||||
page_resources = _ensure_dictionary(base_page.obj, Name.Resources)
|
||||
page_fonts = _ensure_dictionary(page_resources, Name.Font)
|
||||
if font_key not in page_fonts:
|
||||
page_fonts[font_key] = font
|
||||
except (FileNotFoundError, PdfError):
|
||||
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
||||
pass
|
||||
|
||||
+42
-24
@@ -5,31 +5,31 @@
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
from argparse import Namespace
|
||||
from copy import copy
|
||||
from collections.abc import Iterator
|
||||
from pathlib import Path
|
||||
from typing import Iterator
|
||||
|
||||
from pluggy import PluginManager
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from ocrmypdf._options import OcrOptions
|
||||
from ocrmypdf.pdfinfo import PdfInfo
|
||||
from ocrmypdf.pdfinfo.info import PageInfo
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from ocrmypdf._plugin_manager import OcrmypdfPluginManager
|
||||
|
||||
|
||||
class PdfContext:
|
||||
"""Holds the context for a particular run of the pipeline."""
|
||||
|
||||
options: Namespace #: The specified options for processing this PDF.
|
||||
options: OcrOptions #: The specified options for processing this PDF.
|
||||
origin: Path #: The filename of the original input file.
|
||||
pdfinfo: PdfInfo #: Detailed data for this PDF.
|
||||
plugin_manager: PluginManager #: PluginManager for processing the current PDF.
|
||||
plugin_manager: (
|
||||
OcrmypdfPluginManager #: PluginManager for processing the current PDF.
|
||||
)
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
options: Namespace,
|
||||
options: OcrOptions,
|
||||
work_folder: Path,
|
||||
origin: Path,
|
||||
pdfinfo: PdfInfo,
|
||||
@@ -55,27 +55,39 @@ class PdfContext:
|
||||
for n in range(npages):
|
||||
yield PageContext(self, n)
|
||||
|
||||
def get_page_context_args(self) -> Iterator[tuple[PageContext]]:
|
||||
"""Get all ``PageContext`` for this PDF packaged in tuple for args-splatting."""
|
||||
npages = len(self.pdfinfo)
|
||||
for n in range(npages):
|
||||
yield (PageContext(self, n),)
|
||||
|
||||
|
||||
class PageContext:
|
||||
"""Holds our context for a page.
|
||||
|
||||
Must be pickable, so stores only intrinsic/simple data elements or those
|
||||
Must be pickle-able, so stores only intrinsic/simple data elements or those
|
||||
capable of their serializing themselves via ``__getstate__``.
|
||||
|
||||
Note: Uses OcrOptions with JSON serialization for multiprocessing compatibility.
|
||||
"""
|
||||
|
||||
options: Namespace #: The specified options for processing this PDF.
|
||||
origin: Path #: The filename of the original input file.
|
||||
pageno: int #: This page number (zero-based).
|
||||
pageinfo: PageInfo #: Information on this page.
|
||||
plugin_manager: PluginManager #: PluginManager for processing the current PDF.
|
||||
plugin_manager: (
|
||||
OcrmypdfPluginManager #: PluginManager for processing the current PDF.
|
||||
)
|
||||
|
||||
def __init__(self, pdf_context: PdfContext, pageno):
|
||||
self.work_folder = pdf_context.work_folder
|
||||
self.origin = pdf_context.origin
|
||||
# Store OcrOptions directly instead of Namespace
|
||||
self.options = pdf_context.options
|
||||
self.pageno = pageno
|
||||
self.pageinfo = pdf_context.pdfinfo[pageno]
|
||||
self.plugin_manager = pdf_context.plugin_manager
|
||||
# Ensure no reference to PdfContext which contains OcrOptions
|
||||
self._pdf_context = None
|
||||
|
||||
def get_path(self, name: str) -> Path:
|
||||
"""Generate a ``Path`` for a file that is part of processing this page.
|
||||
@@ -88,16 +100,22 @@ class PageContext:
|
||||
def __getstate__(self):
|
||||
state = self.__dict__.copy()
|
||||
|
||||
state['options'] = copy(self.options)
|
||||
if not isinstance(state['options'].input_file, (str, bytes, os.PathLike)):
|
||||
state['options'].input_file = 'stream'
|
||||
if not isinstance(state['options'].output_file, (str, bytes, os.PathLike)):
|
||||
state['options'].output_file = 'stream'
|
||||
options_json = self.options.model_dump_json_safe()
|
||||
state['options_json'] = options_json
|
||||
# Remove the OcrOptions object to avoid pickle issues
|
||||
del state['options']
|
||||
|
||||
# Remove any potential references to Pydantic objects
|
||||
state.pop('_pdf_context', None)
|
||||
return state
|
||||
|
||||
def __setstate__(self, state):
|
||||
self.__dict__.update(state)
|
||||
|
||||
def cleanup_working_files(work_folder: Path, options: Namespace):
|
||||
if options.keep_temporary_files:
|
||||
print(f"Temporary working files retained at:\n{work_folder}", file=sys.stderr)
|
||||
else:
|
||||
shutil.rmtree(work_folder, ignore_errors=True)
|
||||
# Reconstruct OcrOptions from JSON if available
|
||||
if 'options_json' in state:
|
||||
from ocrmypdf._options import OcrOptions
|
||||
|
||||
self.options = OcrOptions.model_validate_json_safe(state['options_json'])
|
||||
# Otherwise, we have a fallback Namespace (shouldn't happen in normal operation)
|
||||
# Leave it as-is for compatibility
|
||||
|
||||
@@ -6,9 +6,9 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from contextlib import suppress
|
||||
|
||||
from tqdm import tqdm
|
||||
from rich.console import Console
|
||||
from rich.logging import RichHandler
|
||||
|
||||
|
||||
class PageNumberFilter(logging.Filter):
|
||||
@@ -23,21 +23,8 @@ class PageNumberFilter(logging.Filter):
|
||||
return True
|
||||
|
||||
|
||||
class TqdmConsole:
|
||||
"""Wrapper to log messages in a way that is compatible with tqdm progress bar
|
||||
|
||||
This routes log messages through tqdm so that it can print them above the
|
||||
progress bar, and then refresh the progress bar, rather than overwriting
|
||||
it which looks messy.
|
||||
"""
|
||||
|
||||
def __init__(self, file):
|
||||
self.file = file
|
||||
|
||||
def write(self, msg):
|
||||
# When no progress bar is active, tqdm.write() routes to print()
|
||||
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
||||
|
||||
def flush(self):
|
||||
with suppress(AttributeError):
|
||||
self.file.flush()
|
||||
class RichLoggingHandler(RichHandler):
|
||||
def __init__(self, console: Console, **kwargs):
|
||||
super().__init__(
|
||||
console=console, show_level=False, show_time=False, markup=False, **kwargs
|
||||
)
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""OCRmyPDF page processing pipeline functions."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime as dt
|
||||
import logging
|
||||
import os
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from pikepdf import Dictionary, Name, Pdf
|
||||
from pikepdf import __version__ as PIKEPDF_VERSION
|
||||
from pikepdf.models.metadata import PdfMetadata, encode_pdf_date
|
||||
|
||||
from ocrmypdf._defaults import PROGRAM_NAME
|
||||
from ocrmypdf._jobcontext import PdfContext
|
||||
from ocrmypdf._version import __version__ as OCRMYPF_VERSION
|
||||
from ocrmypdf.languages import iso_639_2_from_3
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def get_docinfo(base_pdf: Pdf, context: PdfContext) -> dict[str, str]:
|
||||
"""Read the document info and store it in a dictionary."""
|
||||
options = context.options
|
||||
|
||||
def from_document_info(key):
|
||||
try:
|
||||
s = base_pdf.docinfo[key]
|
||||
return str(s)
|
||||
except (KeyError, TypeError):
|
||||
return ''
|
||||
|
||||
pdfmark = {
|
||||
k: from_document_info(k)
|
||||
for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate')
|
||||
}
|
||||
if options.title:
|
||||
pdfmark['/Title'] = options.title
|
||||
if options.author:
|
||||
pdfmark['/Author'] = options.author
|
||||
if options.keywords:
|
||||
pdfmark['/Keywords'] = options.keywords
|
||||
if options.subject:
|
||||
pdfmark['/Subject'] = options.subject
|
||||
|
||||
creator_tag = context.plugin_manager.get_ocr_engine(options=options).creator_tag(
|
||||
options
|
||||
)
|
||||
|
||||
pdfmark['/Creator'] = f'{PROGRAM_NAME} {OCRMYPF_VERSION} / {creator_tag}'
|
||||
pdfmark['/Producer'] = f'pikepdf {PIKEPDF_VERSION}'
|
||||
pdfmark['/ModDate'] = encode_pdf_date(dt.datetime.now(dt.UTC))
|
||||
return pdfmark
|
||||
|
||||
|
||||
def report_on_metadata(options, missing):
|
||||
if not missing:
|
||||
return
|
||||
if options.output_type.startswith('pdfa'):
|
||||
log.warning(
|
||||
"Some input metadata could not be copied because it is not "
|
||||
"permitted in PDF/A. You may wish to examine the output "
|
||||
"PDF's XMP metadata."
|
||||
)
|
||||
log.debug("The following metadata fields were not copied: %r", missing)
|
||||
else:
|
||||
log.error(
|
||||
"Some input metadata could not be copied."
|
||||
"You may wish to examine the output PDF's XMP metadata."
|
||||
)
|
||||
log.info("The following metadata fields were not copied: %r", missing)
|
||||
|
||||
|
||||
def repair_docinfo_nuls(pdf):
|
||||
"""If the DocumentInfo block contains NUL characters, remove them.
|
||||
|
||||
If the DocumentInfo block is malformed, log an error and continue.
|
||||
"""
|
||||
modified = False
|
||||
try:
|
||||
if not isinstance(pdf.docinfo, Dictionary):
|
||||
raise TypeError("DocumentInfo is not a dictionary")
|
||||
for k, v in pdf.docinfo.items():
|
||||
if isinstance(v, str) and b'\x00' in bytes(v):
|
||||
pdf.docinfo[k] = bytes(v).replace(b'\x00', b'')
|
||||
modified = True
|
||||
except TypeError:
|
||||
# TypeError can also be raised if dictionary items are unexpected types
|
||||
log.error("File contains a malformed DocumentInfo block - continuing anyway.")
|
||||
return modified
|
||||
|
||||
|
||||
def should_linearize(working_file: Path, context: PdfContext) -> bool:
|
||||
"""Determine whether the PDF should be linearized.
|
||||
|
||||
For smaller files, linearization is not worth the effort.
|
||||
"""
|
||||
filesize = os.stat(working_file).st_size
|
||||
return filesize > (context.options.fast_web_view * 1_000_000)
|
||||
|
||||
|
||||
def _fix_metadata(meta_original: PdfMetadata, meta_pdf: PdfMetadata):
|
||||
# If xmp:CreateDate is missing, set it to the modify date to
|
||||
# ensure consistency with Ghostscript.
|
||||
if 'xmp:CreateDate' not in meta_pdf:
|
||||
meta_pdf['xmp:CreateDate'] = meta_pdf.get('xmp:ModifyDate', '')
|
||||
if meta_pdf.get('dc:title') == 'Untitled' and ('dc:title' not in meta_original):
|
||||
# Ghostscript likes to set title to Untitled if omitted from input.
|
||||
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
||||
# and the XMP Spec do not make this recommendation.
|
||||
del meta_pdf['dc:title']
|
||||
|
||||
|
||||
def _unset_empty_metadata(meta: PdfMetadata, options):
|
||||
"""Unset metadata fields that were explicitly set to empty strings.
|
||||
|
||||
If the user explicitly specified an empty string for any of the
|
||||
following, they should be unset and not reported as missing in
|
||||
the output pdf. Note that some metadata fields use differing names
|
||||
between PDF/A and PDF.
|
||||
"""
|
||||
if options.title == '' and 'dc:title' in meta:
|
||||
del meta['dc:title'] # PDF/A and PDF
|
||||
if options.author == '':
|
||||
if 'dc:creator' in meta:
|
||||
del meta['dc:creator'] # PDF/A (Not xmp:CreatorTool)
|
||||
if 'pdf:Author' in meta:
|
||||
del meta['pdf:Author'] # PDF
|
||||
if options.subject == '':
|
||||
if 'dc:description' in meta:
|
||||
del meta['dc:description'] # PDF/A
|
||||
if 'dc:subject' in meta:
|
||||
del meta['dc:subject'] # PDF
|
||||
if options.keywords == '' and 'pdf:Keywords' in meta:
|
||||
del meta['pdf:Keywords'] # PDF/A and PDF
|
||||
|
||||
|
||||
def _set_language(pdf: Pdf, languages: list[str]):
|
||||
"""Set the language of the PDF."""
|
||||
if Name.Lang in pdf.Root or not languages:
|
||||
return # Already set or can't change
|
||||
primary_language_iso639_3 = languages[0]
|
||||
if not primary_language_iso639_3:
|
||||
return
|
||||
iso639_2 = iso_639_2_from_3(primary_language_iso639_3)
|
||||
if not iso639_2:
|
||||
return
|
||||
pdf.Root.Lang = iso639_2
|
||||
|
||||
|
||||
class MetadataProgress:
|
||||
def __init__(self, progressbar_class, enable: bool = True):
|
||||
self.progressbar_class = progressbar_class
|
||||
self.progressbar = self.progressbar_class(
|
||||
total=100, desc="Linearizing", unit='%', disable=not enable
|
||||
)
|
||||
|
||||
def __enter__(self):
|
||||
self.progressbar.__enter__()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
return self.progressbar.__exit__(exc_type, exc_value, traceback)
|
||||
|
||||
def __call__(self, percent: int):
|
||||
if not self.progressbar_class:
|
||||
return
|
||||
self.progressbar.update(completed=percent)
|
||||
|
||||
|
||||
def metadata_fixup(
|
||||
working_file: Path, context: PdfContext, pdf_save_settings: dict[str, Any]
|
||||
) -> Path:
|
||||
"""Fix certain metadata fields whether PDF or PDF/A.
|
||||
|
||||
Override some of Ghostscript's metadata choices.
|
||||
|
||||
Also report on metadata in the input file that was not retained during
|
||||
conversion.
|
||||
"""
|
||||
output_file = context.get_path('metafix.pdf')
|
||||
options = context.options
|
||||
|
||||
pbar_class = context.plugin_manager.get_progressbar_class()
|
||||
with (
|
||||
Pdf.open(context.origin) as original,
|
||||
Pdf.open(working_file) as pdf,
|
||||
MetadataProgress(pbar_class, options.progress_bar) as pbar,
|
||||
):
|
||||
docinfo = get_docinfo(original, context)
|
||||
with (
|
||||
original.open_metadata(
|
||||
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
||||
) as meta_original,
|
||||
pdf.open_metadata() as meta_pdf,
|
||||
):
|
||||
meta_pdf.load_from_docinfo(
|
||||
docinfo, delete_missing=False, raise_failure=False
|
||||
)
|
||||
_fix_metadata(meta_original, meta_pdf)
|
||||
_unset_empty_metadata(meta_original, options)
|
||||
_unset_empty_metadata(meta_pdf, options)
|
||||
meta_missing = set(meta_original.keys()) - set(meta_pdf.keys())
|
||||
report_on_metadata(options, meta_missing)
|
||||
|
||||
_set_language(pdf, options.languages)
|
||||
pdf.save(output_file, progress=pbar, **pdf_save_settings)
|
||||
|
||||
return output_file
|
||||
@@ -0,0 +1,613 @@
|
||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Internal options model for OCRmyPDF."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import unicodedata
|
||||
from collections.abc import Sequence
|
||||
from enum import StrEnum
|
||||
from io import IOBase
|
||||
from pathlib import Path
|
||||
from typing import Any, BinaryIO
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
|
||||
|
||||
from ocrmypdf._defaults import DEFAULT_LANGUAGE, DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
from ocrmypdf.exceptions import BadArgsError
|
||||
from ocrmypdf.helpers import monotonic
|
||||
|
||||
# Import plugin option models - these will be available after plugins are loaded
|
||||
# We'll use forward references and handle imports dynamically
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
# Module-level registry for plugin option models
|
||||
# This is populated by setup_plugin_infrastructure() after plugins are loaded
|
||||
_plugin_option_models: dict[str, type] = {}
|
||||
|
||||
PathOrIO = BinaryIO | IOBase | Path | str | bytes
|
||||
|
||||
|
||||
class ProcessingMode(StrEnum):
|
||||
"""OCR processing mode for handling pages with existing text.
|
||||
|
||||
This enum controls how OCRmyPDF handles pages that already contain text:
|
||||
|
||||
- ``default``: Error if text is found (standard OCR behavior)
|
||||
- ``force``: Rasterize all content and run OCR regardless of existing text
|
||||
- ``skip``: Skip OCR on pages that already have text
|
||||
- ``redo``: Re-OCR pages, stripping old invisible text layer
|
||||
"""
|
||||
|
||||
default = 'default'
|
||||
force = 'force'
|
||||
skip = 'skip'
|
||||
redo = 'redo'
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges: str) -> set[int]:
|
||||
"""Convert page range string to set of page numbers."""
|
||||
pages: list[int] = []
|
||||
page_groups = ranges.replace(' ', '').split(',')
|
||||
for group in page_groups:
|
||||
if not group:
|
||||
continue
|
||||
try:
|
||||
start, end = group.split('-')
|
||||
except ValueError:
|
||||
pages.append(int(group) - 1)
|
||||
else:
|
||||
try:
|
||||
new_pages = list(range(int(start) - 1, int(end)))
|
||||
if not new_pages:
|
||||
raise BadArgsError(
|
||||
f"invalid page subrange '{start}-{end}'"
|
||||
) from None
|
||||
pages.extend(new_pages)
|
||||
except ValueError:
|
||||
raise BadArgsError(f"invalid page subrange '{group}'") from None
|
||||
|
||||
if not pages:
|
||||
raise BadArgsError(
|
||||
f"The string of page ranges '{ranges}' did not contain any recognizable "
|
||||
f"page ranges."
|
||||
)
|
||||
|
||||
if not monotonic(pages):
|
||||
log.warning(
|
||||
"List of pages to process contains duplicate pages, or pages that are "
|
||||
"out of order"
|
||||
)
|
||||
if any(page < 0 for page in pages):
|
||||
raise BadArgsError("pages refers to a page number less than 1")
|
||||
|
||||
log.debug("OCRing only these pages: %s", pages)
|
||||
return set(pages)
|
||||
|
||||
|
||||
class OcrOptions(BaseModel):
|
||||
"""Internal options model that can masquerade as argparse.Namespace.
|
||||
|
||||
This model provides proper typing and validation while maintaining
|
||||
compatibility with existing code that expects argparse.Namespace behavior.
|
||||
"""
|
||||
|
||||
# I/O options
|
||||
input_file: PathOrIO
|
||||
output_file: PathOrIO
|
||||
sidecar: PathOrIO | None = None
|
||||
output_folder: Path | None = None
|
||||
work_folder: Path | None = None
|
||||
|
||||
# Core OCR options
|
||||
languages: list[str] = Field(default_factory=lambda: [DEFAULT_LANGUAGE])
|
||||
output_type: str = 'auto'
|
||||
mode: ProcessingMode = ProcessingMode.default
|
||||
|
||||
# Backward compatibility properties for force_ocr, skip_text, redo_ocr
|
||||
@property
|
||||
def force_ocr(self) -> bool:
|
||||
"""Backward compatibility alias for mode == ProcessingMode.force."""
|
||||
return self.mode == ProcessingMode.force
|
||||
|
||||
@property
|
||||
def skip_text(self) -> bool:
|
||||
"""Backward compatibility alias for mode == ProcessingMode.skip."""
|
||||
return self.mode == ProcessingMode.skip
|
||||
|
||||
@property
|
||||
def redo_ocr(self) -> bool:
|
||||
"""Backward compatibility alias for mode == ProcessingMode.redo."""
|
||||
return self.mode == ProcessingMode.redo
|
||||
|
||||
# Job control
|
||||
jobs: int | None = None
|
||||
use_threads: bool = True
|
||||
progress_bar: bool = True
|
||||
quiet: bool = False
|
||||
verbose: int = 0
|
||||
keep_temporary_files: bool = False
|
||||
|
||||
# Image processing
|
||||
image_dpi: int | None = None
|
||||
deskew: bool = False
|
||||
clean: bool = False
|
||||
clean_final: bool = False
|
||||
rotate_pages: bool = False
|
||||
remove_background: bool = False
|
||||
remove_vectors: bool = False
|
||||
oversample: int = 0
|
||||
unpaper_args: str | list[str] | None = (
|
||||
None # Can be string or list after validation
|
||||
)
|
||||
|
||||
# OCR behavior
|
||||
skip_big: float | None = None
|
||||
pages: str | set[int] | None = None # Can be string or set after validation
|
||||
invalidate_digital_signatures: bool = False
|
||||
|
||||
# Metadata
|
||||
title: str | None = None
|
||||
author: str | None = None
|
||||
subject: str | None = None
|
||||
keywords: str | None = None
|
||||
|
||||
# Optimization
|
||||
optimize: int = 1
|
||||
jpg_quality: int | None = None
|
||||
png_quality: int | None = None
|
||||
jbig2_threshold: float = 0.85
|
||||
|
||||
# Compatibility alias for plugins that expect jpeg_quality
|
||||
@property
|
||||
def jpeg_quality(self):
|
||||
"""Compatibility alias for jpg_quality."""
|
||||
return self.jpg_quality
|
||||
|
||||
@jpeg_quality.setter
|
||||
def jpeg_quality(self, value):
|
||||
"""Compatibility alias for jpg_quality."""
|
||||
self.jpg_quality = value
|
||||
|
||||
# Advanced options
|
||||
max_image_mpixels: float = 250.0
|
||||
pdf_renderer: str = 'auto'
|
||||
ocr_engine: str = 'auto'
|
||||
rasterizer: str = 'auto'
|
||||
rotate_pages_threshold: float = DEFAULT_ROTATE_PAGES_THRESHOLD
|
||||
user_words: os.PathLike | None = None
|
||||
user_patterns: os.PathLike | None = None
|
||||
fast_web_view: float = 1.0
|
||||
continue_on_soft_render_error: bool | None = None
|
||||
|
||||
# Tesseract options - also accessible via options.tesseract.<field>
|
||||
tesseract_config: list[str] = []
|
||||
tesseract_pagesegmode: int | None = None
|
||||
tesseract_oem: int | None = None
|
||||
tesseract_thresholding: int | None = None
|
||||
tesseract_timeout: float = 0.0
|
||||
tesseract_non_ocr_timeout: float | None = None
|
||||
tesseract_downsample_above: int = 32767
|
||||
tesseract_downsample_large_images: bool | None = None
|
||||
|
||||
# Ghostscript options - also accessible via options.ghostscript.<field>
|
||||
pdfa_image_compression: str | None = None
|
||||
color_conversion_strategy: str = "LeaveColorUnchanged"
|
||||
|
||||
# Optimize/JBIG2 options - also accessible via options.optimize.<field>
|
||||
jbig2_threshold: float = 0.85
|
||||
|
||||
# Plugin system
|
||||
plugins: Sequence[Path | str] | None = None
|
||||
|
||||
# Store any extra attributes (for plugins and dynamic options)
|
||||
extra_attrs: dict[str, Any] = Field(
|
||||
default_factory=dict, exclude=True, alias='_extra_attrs'
|
||||
)
|
||||
|
||||
@field_validator('languages')
|
||||
@classmethod
|
||||
def validate_languages(cls, v):
|
||||
"""Ensure languages list is not empty."""
|
||||
if not v:
|
||||
return [DEFAULT_LANGUAGE]
|
||||
return v
|
||||
|
||||
@field_validator('output_type')
|
||||
@classmethod
|
||||
def validate_output_type(cls, v):
|
||||
"""Validate output type is one of the allowed values."""
|
||||
valid_types = {'auto', 'pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3', 'none'}
|
||||
if v not in valid_types:
|
||||
raise ValueError(f"output_type must be one of {valid_types}")
|
||||
return v
|
||||
|
||||
@field_validator('pdf_renderer')
|
||||
@classmethod
|
||||
def validate_pdf_renderer(cls, v):
|
||||
"""Validate PDF renderer is one of the allowed values."""
|
||||
valid_renderers = {'auto', 'sandwich', 'fpdf2'}
|
||||
# Legacy hocr/hocrdebug are accepted but redirected to fpdf2
|
||||
legacy_renderers = {'hocr', 'hocrdebug'}
|
||||
all_accepted = valid_renderers | legacy_renderers
|
||||
if v not in all_accepted:
|
||||
raise ValueError(f"pdf_renderer must be one of {all_accepted}")
|
||||
return v
|
||||
|
||||
@field_validator('rasterizer')
|
||||
@classmethod
|
||||
def validate_rasterizer(cls, v):
|
||||
"""Validate rasterizer is one of the allowed values."""
|
||||
valid_rasterizers = {'auto', 'ghostscript', 'pypdfium'}
|
||||
if v not in valid_rasterizers:
|
||||
raise ValueError(f"rasterizer must be one of {valid_rasterizers}")
|
||||
return v
|
||||
|
||||
@field_validator('clean_final')
|
||||
@classmethod
|
||||
def validate_clean_final(cls, v, info):
|
||||
"""If clean_final is True, also set clean to True."""
|
||||
if v and hasattr(info, 'data') and 'clean' in info.data:
|
||||
info.data['clean'] = True
|
||||
return v
|
||||
|
||||
@field_validator('jobs')
|
||||
@classmethod
|
||||
def validate_jobs(cls, v):
|
||||
"""Validate jobs is a reasonable number."""
|
||||
if v is not None and (v < 0 or v > 256):
|
||||
raise ValueError("jobs must be between 0 and 256")
|
||||
return v
|
||||
|
||||
@field_validator('verbose')
|
||||
@classmethod
|
||||
def validate_verbose(cls, v):
|
||||
"""Validate verbose level."""
|
||||
if v < 0 or v > 2:
|
||||
raise ValueError("verbose must be between 0 and 2")
|
||||
return v
|
||||
|
||||
@field_validator('oversample')
|
||||
@classmethod
|
||||
def validate_oversample(cls, v):
|
||||
"""Validate oversample DPI."""
|
||||
if v < 0 or v > 5000:
|
||||
raise ValueError("oversample must be between 0 and 5000")
|
||||
return v
|
||||
|
||||
@field_validator('max_image_mpixels')
|
||||
@classmethod
|
||||
def validate_max_image_mpixels(cls, v):
|
||||
"""Validate max image megapixels."""
|
||||
if v < 0:
|
||||
raise ValueError("max_image_mpixels must be non-negative")
|
||||
return v
|
||||
|
||||
@field_validator('rotate_pages_threshold')
|
||||
@classmethod
|
||||
def validate_rotate_pages_threshold(cls, v):
|
||||
"""Validate rotate pages threshold."""
|
||||
if v < 0 or v > 1000:
|
||||
raise ValueError("rotate_pages_threshold must be between 0 and 1000")
|
||||
return v
|
||||
|
||||
@field_validator('title', 'author', 'keywords', 'subject')
|
||||
@classmethod
|
||||
def validate_metadata_unicode(cls, v):
|
||||
"""Validate metadata strings don't contain unsupported Unicode characters."""
|
||||
if v is None:
|
||||
return v
|
||||
|
||||
for char in v:
|
||||
if unicodedata.category(char) == 'Co' or ord(char) >= 0x10000:
|
||||
hexchar = hex(ord(char))[2:].upper()
|
||||
raise ValueError(
|
||||
f"Metadata string contains unsupported Unicode character: "
|
||||
f"{char} (U+{hexchar})"
|
||||
)
|
||||
return v
|
||||
|
||||
@field_validator('pages')
|
||||
@classmethod
|
||||
def validate_pages_format(cls, v):
|
||||
"""Convert page ranges string to set of page numbers."""
|
||||
if v is None:
|
||||
return v
|
||||
if isinstance(v, set):
|
||||
return v # Already processed
|
||||
|
||||
# Convert string ranges to set of page numbers
|
||||
return _pages_from_ranges(v)
|
||||
|
||||
@model_validator(mode='before')
|
||||
@classmethod
|
||||
def handle_special_cases(cls, data):
|
||||
"""Handle special cases for API compatibility and legacy options."""
|
||||
if isinstance(data, dict):
|
||||
# For hOCR API, output_file might not be present
|
||||
if 'output_folder' in data and 'output_file' not in data:
|
||||
data['output_file'] = '/dev/null' # Placeholder
|
||||
|
||||
# Convert legacy boolean options (force_ocr, skip_text, redo_ocr) to mode
|
||||
force = data.pop('force_ocr', None)
|
||||
skip = data.pop('skip_text', None)
|
||||
redo = data.pop('redo_ocr', None)
|
||||
|
||||
# Count how many legacy options are set to True
|
||||
legacy_set = [
|
||||
(force, ProcessingMode.force),
|
||||
(skip, ProcessingMode.skip),
|
||||
(redo, ProcessingMode.redo),
|
||||
]
|
||||
legacy_true = [(val, mode) for val, mode in legacy_set if val]
|
||||
legacy_count = len(legacy_true)
|
||||
|
||||
# Get current mode value (may be string or enum)
|
||||
current_mode = data.get('mode', ProcessingMode.default)
|
||||
if isinstance(current_mode, str):
|
||||
current_mode = ProcessingMode(current_mode)
|
||||
mode_is_set = current_mode != ProcessingMode.default
|
||||
|
||||
if legacy_count > 1:
|
||||
raise ValueError(
|
||||
"Choose only one of --force-ocr, --skip-text, --redo-ocr."
|
||||
)
|
||||
|
||||
if legacy_count == 1:
|
||||
expected_mode = legacy_true[0][1]
|
||||
if mode_is_set and current_mode != expected_mode:
|
||||
legacy_flag = f"--{expected_mode.value.replace('_', '-')}-ocr"
|
||||
raise ValueError(
|
||||
f"Conflicting options: --mode {current_mode.value} "
|
||||
f"cannot be used with {legacy_flag} or similar legacy flag."
|
||||
)
|
||||
# Set mode from legacy option
|
||||
data['mode'] = expected_mode
|
||||
|
||||
return data
|
||||
|
||||
@model_validator(mode='after')
|
||||
def validate_redo_ocr_options(self):
|
||||
"""Validate options compatible with redo mode."""
|
||||
if self.mode == ProcessingMode.redo and (
|
||||
self.deskew or self.clean_final or self.remove_background
|
||||
):
|
||||
raise ValueError(
|
||||
"--redo-ocr (or --mode redo) is not currently compatible with "
|
||||
"--deskew, --clean-final, and --remove-background"
|
||||
)
|
||||
return self
|
||||
|
||||
@model_validator(mode='after')
|
||||
def validate_output_type_compatibility(self):
|
||||
"""Validate output type is compatible with output file."""
|
||||
if self.output_type == 'none' and str(self.output_file) not in (
|
||||
os.devnull,
|
||||
'-',
|
||||
):
|
||||
raise ValueError(
|
||||
"Since you specified `--output-type none`, the output file "
|
||||
f"{self.output_file} cannot be produced. Set the output file to "
|
||||
f"`-` to suppress this message."
|
||||
)
|
||||
return self
|
||||
|
||||
@property
|
||||
def lossless_reconstruction(self):
|
||||
"""Determine lossless_reconstruction based on other options."""
|
||||
lossless = not any(
|
||||
[
|
||||
self.deskew,
|
||||
self.clean_final,
|
||||
self.mode == ProcessingMode.force,
|
||||
self.remove_background,
|
||||
]
|
||||
)
|
||||
return lossless
|
||||
|
||||
def model_dump_json_safe(self) -> str:
|
||||
"""Serialize to JSON with special handling for non-serializable types."""
|
||||
# Create a copy of the model data for serialization
|
||||
data = self.model_dump()
|
||||
|
||||
# Handle special types that don't serialize to JSON directly
|
||||
def _serialize_value(value):
|
||||
if isinstance(value, Path):
|
||||
return {'__type__': 'Path', 'value': str(value)}
|
||||
elif (
|
||||
isinstance(value, BinaryIO | IOBase)
|
||||
or hasattr(value, 'read')
|
||||
or hasattr(value, 'write')
|
||||
):
|
||||
# Stream object - replace with placeholder
|
||||
return {'__type__': 'Stream', 'value': 'stream'}
|
||||
elif hasattr(value, '__class__') and 'Iterator' in value.__class__.__name__:
|
||||
# Handle Pydantic serialization iterators
|
||||
return {'__type__': 'Stream', 'value': 'stream'}
|
||||
elif isinstance(value, property):
|
||||
# Handle property objects that shouldn't be serialized
|
||||
return None
|
||||
elif isinstance(value, list | tuple):
|
||||
return [_serialize_value(item) for item in value]
|
||||
elif isinstance(value, dict):
|
||||
return {k: _serialize_value(v) for k, v in value.items()}
|
||||
else:
|
||||
return value
|
||||
|
||||
# Process all fields
|
||||
serializable_data = {}
|
||||
for key, value in data.items():
|
||||
serialized_value = _serialize_value(value)
|
||||
if serialized_value is not None: # Skip None values from properties
|
||||
serializable_data[key] = serialized_value
|
||||
|
||||
# Add extra_attrs, excluding plugin cache entries (they'll be recreated lazily)
|
||||
if self.extra_attrs:
|
||||
filtered_extra = {
|
||||
k: v
|
||||
for k, v in self.extra_attrs.items()
|
||||
if not k.startswith('_plugin_cache_')
|
||||
}
|
||||
if filtered_extra:
|
||||
serializable_data['_extra_attrs'] = _serialize_value(filtered_extra)
|
||||
|
||||
return json.dumps(serializable_data)
|
||||
|
||||
@classmethod
|
||||
def model_validate_json_safe(cls, json_str: str) -> OcrOptions:
|
||||
"""Reconstruct from JSON with special handling for non-serializable types."""
|
||||
data = json.loads(json_str)
|
||||
|
||||
# Handle special types during deserialization
|
||||
def _deserialize_value(value):
|
||||
if isinstance(value, dict) and '__type__' in value:
|
||||
if value['__type__'] == 'Path':
|
||||
return Path(value['value'])
|
||||
elif value['__type__'] == 'Stream':
|
||||
# For streams, we'll use a placeholder string
|
||||
return value['value']
|
||||
else:
|
||||
return value['value']
|
||||
elif isinstance(value, list):
|
||||
return [_deserialize_value(item) for item in value]
|
||||
elif isinstance(value, dict):
|
||||
return {k: _deserialize_value(v) for k, v in value.items()}
|
||||
else:
|
||||
return value
|
||||
|
||||
# Process all fields
|
||||
deserialized_data = {}
|
||||
extra_attrs = {}
|
||||
|
||||
for key, value in data.items():
|
||||
if key == '_extra_attrs':
|
||||
extra_attrs = _deserialize_value(value)
|
||||
else:
|
||||
deserialized_data[key] = _deserialize_value(value)
|
||||
|
||||
# Create instance
|
||||
instance = cls(**deserialized_data)
|
||||
instance.extra_attrs = extra_attrs
|
||||
|
||||
return instance
|
||||
|
||||
model_config = ConfigDict(
|
||||
extra="forbid", # Force use of extra_attrs for unknown fields
|
||||
arbitrary_types_allowed=True, # Allow BinaryIO, Path, etc.
|
||||
validate_assignment=True, # Validate on attribute assignment
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def register_plugin_models(cls, models: dict[str, type]) -> None:
|
||||
"""Register plugin option model classes for nested access.
|
||||
|
||||
Args:
|
||||
models: Dictionary mapping namespace to model class
|
||||
"""
|
||||
global _plugin_option_models
|
||||
_plugin_option_models.update(models)
|
||||
|
||||
def _get_plugin_options(self, namespace: str) -> Any:
|
||||
"""Get or create a plugin options instance for the given namespace.
|
||||
|
||||
This method creates plugin option instances lazily from flat field values.
|
||||
|
||||
Args:
|
||||
namespace: The plugin namespace (e.g., 'tesseract', 'optimize')
|
||||
|
||||
Returns:
|
||||
An instance of the plugin's option model, or None if not registered
|
||||
"""
|
||||
# Use extra_attrs to cache plugin option instances
|
||||
cache_key = f'_plugin_cache_{namespace}'
|
||||
if cache_key in self.extra_attrs:
|
||||
return self.extra_attrs[cache_key]
|
||||
|
||||
if namespace not in _plugin_option_models:
|
||||
raise AttributeError(
|
||||
f"Plugin namespace '{namespace}' is not registered. "
|
||||
f"Ensure setup_plugin_infrastructure() was called."
|
||||
)
|
||||
|
||||
model_class = _plugin_option_models[namespace]
|
||||
|
||||
def _convert_value(value):
|
||||
"""Convert value to be compatible with plugin model fields."""
|
||||
if isinstance(value, os.PathLike):
|
||||
return os.fspath(value)
|
||||
return value
|
||||
|
||||
# Build kwargs from flat fields
|
||||
kwargs = {}
|
||||
for field_name in model_class.model_fields:
|
||||
# Try namespace_field pattern first (e.g., tesseract_timeout)
|
||||
flat_name = f"{namespace}_{field_name}"
|
||||
if flat_name in OcrOptions.model_fields:
|
||||
value = getattr(self, flat_name)
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
# Also check direct field name (for fields like jbig2_lossy)
|
||||
elif field_name in OcrOptions.model_fields:
|
||||
value = getattr(self, field_name)
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
# Check for special mappings
|
||||
elif namespace == 'optimize' and field_name == 'level':
|
||||
# 'optimize' field maps to 'level' in OptimizeOptions
|
||||
if 'optimize' in OcrOptions.model_fields:
|
||||
value = self.optimize
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
elif namespace == 'optimize' and field_name == 'jpeg_quality':
|
||||
# jpg_quality maps to jpeg_quality
|
||||
if 'jpg_quality' in OcrOptions.model_fields:
|
||||
value = self.jpg_quality
|
||||
if value is not None:
|
||||
kwargs[field_name] = _convert_value(value)
|
||||
|
||||
# Create and cache the plugin options instance
|
||||
instance = model_class(**kwargs)
|
||||
self.extra_attrs[cache_key] = instance
|
||||
return instance
|
||||
|
||||
def __getattr__(self, name: str) -> Any:
|
||||
"""Support dynamic access to plugin option namespaces.
|
||||
|
||||
This allows accessing plugin options like:
|
||||
options.tesseract.timeout
|
||||
options.optimize.level
|
||||
|
||||
Plugin models must be registered via register_plugin_models() for
|
||||
namespace access to work. Built-in plugins register their models
|
||||
during initialization.
|
||||
|
||||
Args:
|
||||
name: Attribute name
|
||||
|
||||
Returns:
|
||||
Plugin options instance if name is a registered namespace,
|
||||
otherwise raises AttributeError
|
||||
"""
|
||||
# Check if this is a plugin namespace
|
||||
if name.startswith('_'):
|
||||
# Private attributes should not trigger plugin lookup
|
||||
raise AttributeError(
|
||||
f"'{type(self).__name__}' object has no attribute '{name}'"
|
||||
)
|
||||
|
||||
# Try to get plugin options for this namespace
|
||||
if name in _plugin_option_models:
|
||||
return self._get_plugin_options(name)
|
||||
|
||||
# Check extra_attrs
|
||||
if 'extra_attrs' in self.__dict__ and name in self.extra_attrs:
|
||||
return self.extra_attrs[name]
|
||||
|
||||
raise AttributeError(
|
||||
f"'{type(self).__name__}' object has no attribute '{name}'"
|
||||
)
|
||||
+627
-283
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,5 @@
|
||||
# SPDX-FileCopyrightText: 2023 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
|
||||
from __future__ import annotations
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user