Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
58abb5785c | ||
|
|
509e75eaff | ||
|
|
0c50eedb2a | ||
|
|
c38ff90081 | ||
|
|
4c029e973f | ||
|
|
21cf9029e8 | ||
|
|
4a640b8dcd | ||
|
|
9471bc8921 | ||
|
|
7fe06c64fc | ||
|
|
d13d70fd56 | ||
|
|
58ec56180a | ||
|
|
32a88f1bad | ||
|
|
99ef42940c | ||
|
|
c152710617 | ||
|
|
8de0f9b86f | ||
|
|
23bc3d3a29 | ||
|
|
8307832ce9 | ||
|
|
dd1cf567db | ||
|
|
2490be8490 | ||
|
|
85e6c6669a | ||
|
|
00498282f5 | ||
|
|
e4cc9fcba7 | ||
|
|
a4555b1dae | ||
|
|
f35a2303bb | ||
|
|
82142fe5ef | ||
|
|
9be533b5f4 | ||
|
|
99653fcd32 | ||
|
|
5442c97ed8 | ||
|
|
0165255bd9 | ||
|
|
378e4dae3b | ||
|
|
cdf5afa753 | ||
|
|
a2deee4920 | ||
|
|
1efa79cce2 | ||
|
|
b3b61c152c | ||
|
|
e429c3d729 | ||
|
|
8308b20096 | ||
|
|
8b41f60b6e | ||
|
|
d56f749017 | ||
|
|
9f31774aa9 | ||
|
|
7d55f6e01f | ||
|
|
c3bd2f296d | ||
|
|
e40c60d4d8 | ||
|
|
3960232ae0 | ||
|
|
5fbb3fc6ac | ||
|
|
0b1db8fccd | ||
|
|
0417610f9b | ||
|
|
43a23e3695 | ||
|
|
c4ca572afd | ||
|
|
e04e4565a9 | ||
|
|
2391fb0be0 | ||
|
|
c16f79d51b | ||
|
|
e3e888efde | ||
|
|
84f7e34ace | ||
|
|
32e2175891 | ||
|
|
975abfde9a | ||
|
|
d929ed76c2 | ||
|
|
4a27124eab | ||
|
|
683ffb84e8 | ||
|
|
2f2602357b | ||
|
|
a5f429f499 | ||
|
|
bdb7f92131 | ||
|
|
09f15ac4c0 | ||
|
|
4fdbf55c11 | ||
|
|
fe2b07652b | ||
|
|
f6d7aa6e33 | ||
|
|
a005d14f91 | ||
|
|
6f66232d44 | ||
|
|
b8a780d684 | ||
|
|
82f393dd09 | ||
|
|
4952af1604 | ||
|
|
bcf77375c0 | ||
|
|
3eab161771 | ||
|
|
b7f38e976b | ||
|
|
a6567f2ae4 | ||
|
|
e860c56b75 | ||
|
|
2e15d52895 | ||
|
|
ce97af5a79 | ||
|
|
3831c4cd4d | ||
|
|
61a2674317 | ||
|
|
9ad8cbf1f6 | ||
|
|
123fde174d | ||
|
|
fd991a2380 | ||
|
|
6f5d77d930 | ||
|
|
5169ac633b | ||
|
|
5b6ab1e003 | ||
|
|
8f984bf958 | ||
|
|
9c5f0d0ec6 | ||
|
|
32041c43e1 | ||
|
|
599028bebb | ||
|
|
6faa8f7221 | ||
|
|
a4dc5e365f | ||
|
|
e2a563cc76 | ||
|
|
1037d73efb | ||
|
|
aeb7b142a9 | ||
|
|
422ea9777e | ||
|
|
2f1c743227 | ||
|
|
96ee21aee9 | ||
|
|
4b759af6ff | ||
|
|
25d2b0cda4 | ||
|
|
16dd8b54a8 | ||
|
|
c4dc5269d2 | ||
|
|
c36e9950ae | ||
|
|
0c0d53b10f | ||
|
|
63de7e1677 | ||
|
|
b0e92760a2 | ||
|
|
054c0773a3 | ||
|
|
89aa78b724 | ||
|
|
708113a514 | ||
|
|
95ef5410c2 | ||
|
|
868b3b4abd | ||
|
|
045bdff95a | ||
|
|
d12b27ac1d | ||
|
|
e4e00de79f | ||
|
|
a53a3937c2 | ||
|
|
343424b4d2 | ||
|
|
c5edff2c2f | ||
|
|
8c5f8b8ddd | ||
|
|
39da931a56 | ||
|
|
9fe354359b | ||
|
|
facc4750bc | ||
|
|
437c235738 | ||
|
|
9559b0b186 | ||
|
|
91456e19a4 | ||
|
|
a2d89f67c4 | ||
|
|
f34130d193 | ||
|
|
c5571388e2 | ||
|
|
9af59c0d6d | ||
|
|
55ae838cb7 | ||
|
|
c434b97f55 | ||
|
|
607eee198d | ||
|
|
5e2a7f8a56 | ||
|
|
fd9550acda | ||
|
|
b354511ac9 | ||
|
|
7be293f628 | ||
|
|
65855dc14c | ||
|
|
cac4a8b9b6 | ||
|
|
17d97b354a | ||
|
|
1c1b60fa9f | ||
|
|
6b745d892f | ||
|
|
fbf271a3ec | ||
|
|
8077718804 | ||
|
|
66bda3420a | ||
|
|
f6510e2b15 | ||
|
|
51abd79136 | ||
|
|
5607429d9a | ||
|
|
b8b7ecfe7f | ||
|
|
d4abe88452 | ||
|
|
cb3cfaa055 | ||
|
|
9db01c7ff5 | ||
|
|
d0301813cc | ||
|
|
cff37bf681 | ||
|
|
66d04dd6e3 | ||
|
|
06a1f987d4 | ||
|
|
e51e21c6b6 | ||
|
|
c5fa72bd4e | ||
|
|
bf99587aa1 | ||
|
|
d249aef57d | ||
|
|
43ab7c88d7 | ||
|
|
ca9669742d | ||
|
|
8a1dddc3ee | ||
|
|
0cd424ffcb | ||
|
|
fde550f9a7 | ||
|
|
a3726e4ce3 | ||
|
|
4ab0a8ff35 | ||
|
|
37f6f72df3 | ||
|
|
3f92867ae6 | ||
|
|
e63503d64b | ||
|
|
17d20309c7 | ||
|
|
fe7c69ce95 | ||
|
|
9baccee8c5 | ||
|
|
d5bb9929f3 | ||
|
|
72d3ee3a87 | ||
|
|
17c419dfcb | ||
|
|
84cc49b14b | ||
|
|
b7f63bc93d | ||
|
|
ad9a3b5302 | ||
|
|
4e4bcaf243 | ||
|
|
11afe3507f | ||
|
|
7691ba8535 | ||
|
|
b787a369ee | ||
|
|
9fb8b267af | ||
|
|
0a08d6ce1f | ||
|
|
f517efe819 | ||
|
|
5f5421f23d | ||
|
|
703b6db95c | ||
|
|
000040d497 | ||
|
|
5bd6665b49 | ||
|
|
1c303afe21 | ||
|
|
11a5c80917 | ||
|
|
9b2ab92913 | ||
|
|
0c4b69ec5a | ||
|
|
45bea1c0e0 | ||
|
|
db914d4cd1 | ||
|
|
df4a8faecd | ||
|
|
1273e7aeda | ||
|
|
e13a673b1a | ||
|
|
979b0bcaed | ||
|
|
3438afaffe | ||
|
|
681fa039cc | ||
|
|
69e80f1545 | ||
|
|
983835cce4 | ||
|
|
6c23b137e2 | ||
|
|
d656b2b3f2 | ||
|
|
031b800aac | ||
|
|
05eb85ee77 | ||
|
|
4da5214ca9 | ||
|
|
1ee829dd59 | ||
|
|
99db5d91ae | ||
|
|
3a4490ee36 | ||
|
|
a492e3b472 | ||
|
|
c3719d3b72 | ||
|
|
ad48fc6415 | ||
|
|
7f8018ffde | ||
|
|
80651fe12c | ||
|
|
a58209e895 | ||
|
|
775b958c55 | ||
|
|
cdcdd16865 | ||
|
|
b332d76782 | ||
|
|
3660007fc8 | ||
|
|
b55d7e57af | ||
|
|
6e99e7b346 | ||
|
|
4d26867dee | ||
|
|
78e8bf9cbf | ||
|
|
de61530d4d | ||
|
|
c149f860b5 | ||
|
|
68c852acec | ||
|
|
a8565bac6e | ||
|
|
6e8b0c3194 | ||
|
|
ff860e8362 | ||
|
|
cf4b04c5d1 | ||
|
|
078bc2abe9 |
+3
-3
@@ -1,5 +1,3 @@
|
||||
# Coverage isn't really compatible with subprocesses so results are unreliable
|
||||
|
||||
[paths]
|
||||
source =
|
||||
src
|
||||
@@ -8,9 +6,11 @@ source =
|
||||
[run]
|
||||
branch = true
|
||||
parallel = true
|
||||
concurrency =
|
||||
thread
|
||||
multiprocessing
|
||||
source =
|
||||
src/ocrmypdf
|
||||
tests
|
||||
omit =
|
||||
tests/spoof/*
|
||||
|
||||
|
||||
+23
-22
@@ -1,6 +1,6 @@
|
||||
# OCRmyPDF
|
||||
#
|
||||
FROM ubuntu:19.04 as base
|
||||
FROM ubuntu:19.10 as base
|
||||
|
||||
FROM base as builder
|
||||
|
||||
@@ -10,36 +10,37 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
build-essential autoconf automake libtool \
|
||||
libleptonica-dev \
|
||||
zlib1g-dev \
|
||||
ocrmypdf \
|
||||
pngquant \
|
||||
python3-pip \
|
||||
python3-venv \
|
||||
tesseract-ocr \
|
||||
unpaper \
|
||||
wget \
|
||||
python3 \
|
||||
python3-distutils \
|
||||
ca-certificates \
|
||||
curl \
|
||||
git
|
||||
|
||||
# Get the latest pip (Ubuntu version doesn't support manylinux2010)
|
||||
RUN \
|
||||
curl https://bootstrap.pypa.io/get-pip.py | python3
|
||||
|
||||
# Compile and install jbig2
|
||||
# Needs libleptonica-dev, zlib1g-dev
|
||||
RUN \
|
||||
mkdir jbig2 \
|
||||
&& wget -q https://github.com/agl/jbig2enc/archive/0.29.tar.gz -O - | \
|
||||
&& curl -L https://github.com/agl/jbig2enc/archive/0.29.tar.gz | \
|
||||
tar xz -C jbig2 --strip-components=1 \
|
||||
&& cd jbig2 \
|
||||
&& ./autogen.sh && ./configure && make && make install \
|
||||
&& cd .. \
|
||||
&& rm -rf jbig2
|
||||
|
||||
RUN python3 -m venv /appenv
|
||||
|
||||
COPY . /app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
RUN . /appenv/bin/activate; \
|
||||
pip install --upgrade pip \
|
||||
&& pip install .
|
||||
RUN pip3 install --no-cache-dir \
|
||||
-r requirements/main.txt \
|
||||
-r requirements/webservice.txt \
|
||||
-r requirements/test.txt \
|
||||
-r requirements/watcher.txt \
|
||||
.
|
||||
|
||||
FROM base
|
||||
|
||||
@@ -53,7 +54,6 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
zlib1g \
|
||||
pngquant \
|
||||
python3 \
|
||||
python3-venv \
|
||||
qpdf \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-chi-sim \
|
||||
@@ -62,11 +62,15 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
tesseract-ocr-fra \
|
||||
tesseract-ocr-por \
|
||||
tesseract-ocr-spa \
|
||||
unpaper \
|
||||
wget
|
||||
unpaper
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
COPY --from=builder /usr/local/lib/ /usr/local/lib/
|
||||
COPY --from=builder /usr/local/bin/ /usr/local/bin/
|
||||
|
||||
# Copy
|
||||
COPY --from=builder /app/misc/webservice.py /app/
|
||||
COPY --from=builder /app/misc/watcher.py /app/
|
||||
|
||||
# Copy minimal project files to get the test suite.
|
||||
COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/
|
||||
@@ -74,7 +78,4 @@ COPY --from=builder /app/requirements /app/requirements
|
||||
COPY --from=builder /app/tests /app/tests
|
||||
COPY --from=builder /app/src /app/src
|
||||
|
||||
COPY --from=builder /appenv /appenv
|
||||
COPY --from=builder /usr/local /usr/local
|
||||
|
||||
ENTRYPOINT ["/appenv/bin/ocrmypdf"]
|
||||
ENTRYPOINT ["/usr/local/bin/ocrmypdf"]
|
||||
|
||||
@@ -1,91 +0,0 @@
|
||||
FROM alpine:3.9 as base
|
||||
|
||||
FROM base as builder
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
|
||||
# Normally:
|
||||
# echo '@testing http://nl.alpinelinux.org/alpine/edge/testing' >> /etc/apk/repositories
|
||||
|
||||
RUN \
|
||||
echo -e '@testing http://nl.alpinelinux.org/alpine/edge/testing\n@community http://nl.alpinelinux.org/alpine/edge/community'\
|
||||
>> /etc/apk/repositories \
|
||||
# Add runtime dependencies
|
||||
&& apk add --update \
|
||||
python3-dev \
|
||||
py3-setuptools \
|
||||
jbig2enc@community \
|
||||
ghostscript \
|
||||
qpdf@community \
|
||||
qpdf-dev@community \
|
||||
tesseract-ocr \
|
||||
unpaper \
|
||||
pngquant \
|
||||
libxml2-dev \
|
||||
libxslt-dev \
|
||||
zlib-dev \
|
||||
libffi-dev \
|
||||
leptonica-dev \
|
||||
binutils \
|
||||
&& pip3 install --upgrade pip \
|
||||
# Install pybind11 for pikepdf
|
||||
&& pip3 install pybind11 \
|
||||
# Install flask for the webservice
|
||||
&& pip3 install flask \
|
||||
# Add build dependencies
|
||||
&& apk add --virtual build-dependencies \
|
||||
build-base \
|
||||
git
|
||||
|
||||
COPY . /app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
RUN pip3 install .
|
||||
|
||||
FROM base
|
||||
|
||||
ENV LANG=C.UTF-8
|
||||
|
||||
# Normally:
|
||||
# echo '@testing http://nl.alpinelinux.org/alpine/edge/testing' >> /etc/apk/repositories
|
||||
|
||||
RUN \
|
||||
echo -e '@testing http://nl.alpinelinux.org/alpine/edge/testing\n@community http://nl.alpinelinux.org/alpine/edge/community'\
|
||||
>> /etc/apk/repositories \
|
||||
# Add runtime dependencies
|
||||
&& apk add --update \
|
||||
python3 \
|
||||
jbig2enc@community \
|
||||
ghostscript \
|
||||
qpdf@community \
|
||||
qpdf-dev@community \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-data-deu \
|
||||
tesseract-ocr-data-chi_sim \
|
||||
unpaper \
|
||||
pngquant \
|
||||
libxml2 \
|
||||
libxslt \
|
||||
zlib \
|
||||
libffi \
|
||||
leptonica-dev \
|
||||
binutils \
|
||||
&& mkdir /app
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Copy build artifacts (python site-packages)
|
||||
COPY --from=builder /usr/lib/python3.6/site-packages /usr/lib/python3.6/site-packages
|
||||
COPY --from=builder /usr/bin/ocrmypdf /usr/bin/dumppdf.py /usr/bin/latin2ascii.py /usr/bin/pdf2txt.py /usr/bin/img2pdf /usr/bin/chardetect /usr/bin/
|
||||
|
||||
# Copy
|
||||
COPY --from=builder /app/misc/webservice.py /app/
|
||||
|
||||
# Copy minimal project files to get the test suite.
|
||||
COPY --from=builder /app/setup.cfg /app/setup.py /app/README.md /app/
|
||||
COPY --from=builder /app/requirements /app/requirements
|
||||
COPY --from=builder /app/tests /app/tests
|
||||
COPY --from=builder /app/src /app/src
|
||||
|
||||
ENTRYPOINT ["/usr/bin/ocrmypdf"]
|
||||
@@ -0,0 +1,36 @@
|
||||
---
|
||||
name: Bug report
|
||||
about: Create a report to help us improve
|
||||
title: ''
|
||||
labels: ''
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
**To Reproduce**
|
||||
What command line or API call were you trying to run?
|
||||
|
||||
```bash
|
||||
ocrmypdf ...arguments... input.pdf output.pdf
|
||||
```
|
||||
|
||||
Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful.
|
||||
|
||||
**Example file**
|
||||
Please include an example *input* PDF (or image). The input file is more helpful.
|
||||
|
||||
If possible, use an input file with no personal or confidential information. At your option you may GPG-encrypt the file for OCRmyPDF's author only.
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Screenshots**
|
||||
If applicable, add screenshots to help explain your problem.
|
||||
|
||||
**System**
|
||||
- OS: [e.g. Linux, Windows, macOS]
|
||||
- OCRmyPDF Version: ``ocrmypdf --version``
|
||||
- How did you install ocrmypdf? Did you use a system package manager, `pip`, or a Docker image?
|
||||
@@ -0,0 +1,17 @@
|
||||
---
|
||||
name: Feature request
|
||||
about: Suggest an idea for this project
|
||||
title: ''
|
||||
labels: enhancement
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
**Is your feature request related to a problem? Please describe.**
|
||||
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
||||
|
||||
**Describe the solution you'd like**
|
||||
A clear and concise description of what you want to happen.
|
||||
|
||||
**Additional context**
|
||||
Add any other context or screenshots about the feature request here.
|
||||
@@ -7,6 +7,7 @@
|
||||
*.pyc
|
||||
*.sublime-*
|
||||
*.DS_Store
|
||||
.mypy_cache/
|
||||
|
||||
# Package building
|
||||
.eggs/
|
||||
|
||||
+20
-3
@@ -1,6 +1,23 @@
|
||||
repos:
|
||||
- repo: https://github.com/ambv/black
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v2.4.0
|
||||
hooks:
|
||||
- id: check-case-conflict
|
||||
- id: check-merge-conflict
|
||||
- id: check-toml
|
||||
- id: check-yaml
|
||||
- id: debug-statements
|
||||
- repo: https://github.com/asottile/seed-isort-config
|
||||
rev: v1.9.3
|
||||
hooks:
|
||||
- id: seed-isort-config
|
||||
- repo: https://github.com/pre-commit/mirrors-isort
|
||||
rev: v4.3.21 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases
|
||||
hooks:
|
||||
- id: isort
|
||||
- repo: https://github.com/psf/black
|
||||
rev: stable
|
||||
hooks:
|
||||
- id: black
|
||||
language_version: python3.7
|
||||
- id: black
|
||||
language_version: python3.7
|
||||
exclude: ^src/ocrmypdf/lib/_leptonica.py
|
||||
|
||||
-143
@@ -1,143 +0,0 @@
|
||||
cache:
|
||||
pip: true
|
||||
directories:
|
||||
- $HOME/Library/Caches/Homebrew
|
||||
|
||||
matrix:
|
||||
include:
|
||||
- os: linux
|
||||
dist: trusty
|
||||
sudo: required
|
||||
language: python
|
||||
python: "3.6"
|
||||
env:
|
||||
- DIST=trusty
|
||||
- MINIMAL=true
|
||||
addons:
|
||||
apt:
|
||||
update: true
|
||||
sources:
|
||||
- sourceline: "ppa:alex-p/tesseract-ocr"
|
||||
- sourceline: "ppa:vshn/ghostscript"
|
||||
packages:
|
||||
- ghostscript
|
||||
- libffi-dev
|
||||
- qpdf
|
||||
- tesseract-ocr
|
||||
- tesseract-ocr-deu
|
||||
- tesseract-ocr-eng
|
||||
- tesseract-ocr-fra
|
||||
before_install: |
|
||||
pip3 install --upgrade pip
|
||||
pip3 install --upgrade wheel
|
||||
- os: linux
|
||||
dist: trusty
|
||||
sudo: required
|
||||
language: python
|
||||
python: "3.6"
|
||||
env:
|
||||
- DIST=trusty
|
||||
addons:
|
||||
apt:
|
||||
update: true
|
||||
sources:
|
||||
- sourceline: "ppa:alex-p/tesseract-ocr"
|
||||
- sourceline: "ppa:heyarje/libav-11"
|
||||
- sourceline: "ppa:vshn/ghostscript"
|
||||
packages:
|
||||
- ghostscript
|
||||
- libavcodec56
|
||||
- libavformat56
|
||||
- libavutil54
|
||||
- libffi-dev
|
||||
- qpdf
|
||||
- tesseract-ocr
|
||||
- tesseract-ocr-deu
|
||||
- tesseract-ocr-eng
|
||||
- tesseract-ocr-fra
|
||||
- libexempi3 # --- optional extras from here ---
|
||||
- pngquant
|
||||
- poppler-utils
|
||||
before_install: |
|
||||
mkdir -p bin packages
|
||||
pip3 install --upgrade pip
|
||||
pip3 install --upgrade wheel
|
||||
wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O packages/unpaper_6.1-1.deb
|
||||
sudo dpkg -i packages/unpaper_6.1-1.deb
|
||||
- os: linux
|
||||
dist: xenial
|
||||
sudo: required
|
||||
language: python
|
||||
python: "3.7"
|
||||
env:
|
||||
- DIST=xenial
|
||||
addons:
|
||||
apt:
|
||||
update: true
|
||||
sources:
|
||||
- sourceline: "ppa:alex-p/tesseract-ocr"
|
||||
packages:
|
||||
- ghostscript
|
||||
- libexempi3
|
||||
- libffi-dev
|
||||
- pngquant
|
||||
- poppler-utils
|
||||
- qpdf
|
||||
- tesseract-ocr
|
||||
- tesseract-ocr-deu
|
||||
- tesseract-ocr-eng
|
||||
- tesseract-ocr-fra
|
||||
- unpaper
|
||||
- os: osx
|
||||
osx_image: xcode9.2
|
||||
language: generic
|
||||
addons:
|
||||
homebrew:
|
||||
update: true
|
||||
packages:
|
||||
- exempi
|
||||
- ghostscript
|
||||
- jbig2enc
|
||||
- leptonica
|
||||
- openjpeg
|
||||
- pngquant
|
||||
- python
|
||||
- qpdf
|
||||
- tesseract
|
||||
- unpaper
|
||||
before_install: |
|
||||
pip3 install --upgrade pip
|
||||
pip3 install wheel
|
||||
|
||||
before_cache:
|
||||
- rm -f $HOME/.cache/pip/log/debug.log
|
||||
|
||||
install:
|
||||
- mkdir -p bin
|
||||
- export PATH=$PWD/bin:$PATH
|
||||
- pip3 install pycparser # py3.7 workaround for https://github.com/eliben/pycparser/issues/251
|
||||
- pip3 install -r requirements/main.txt
|
||||
- pip3 install --no-deps .
|
||||
- pip3 install -r requirements/test.txt
|
||||
|
||||
script:
|
||||
- tesseract --version
|
||||
- qpdf --version
|
||||
- pytest -n auto
|
||||
|
||||
deploy:
|
||||
# release for main pypi
|
||||
# 3.7 is considered the build leader and does the deploy, otherwise there is
|
||||
# a race and all versions will try to deploy
|
||||
# OTOH if we ever need separate binary wheels then each version needs its
|
||||
# own deploy
|
||||
- provider: pypi
|
||||
user: ocrmypdf-travis
|
||||
password:
|
||||
secure: "DTFOmmNL6olA0+yXvp4u9jXZlZeqrJsJ0526jzqf4a3gZ6jnGTq5UI6WzRsslSyoMMfXKtHQebqHM6ogSgCZinyZ3ufHJo8fn9brxbEc2gsiWkbj5o3bGwdWMT1vNNE7XW0VCpw87rZ1EEwjl4FJHFudMlPR1yfU5+uq0k0PACo="
|
||||
distributions: "sdist bdist_wheel"
|
||||
on:
|
||||
branch: master
|
||||
tags: true
|
||||
condition: $TRAVIS_PYTHON_VERSION == "3.7" && $TRAVIS_OS_NAME == "linux"
|
||||
skip_upload_docs: true
|
||||
-43
@@ -1,43 +0,0 @@
|
||||
# requirements
|
||||
recursive-include requirements *
|
||||
|
||||
# git
|
||||
include .git_archival.txt
|
||||
|
||||
# docker
|
||||
include .dockerignore
|
||||
recursive-include .docker *
|
||||
|
||||
# tests
|
||||
include .coveragerc
|
||||
recursive-include tests *.bin
|
||||
recursive-include tests *.jpg
|
||||
recursive-include tests *.jsonl
|
||||
recursive-include tests *.png
|
||||
recursive-include tests *.pdf
|
||||
recursive-include tests *.py
|
||||
recursive-include tests *.rst
|
||||
recursive-include tests *.txt
|
||||
recursive-exclude tests/resources/private *
|
||||
|
||||
# documentation
|
||||
include LICENSE
|
||||
include *.rst
|
||||
recursive-exclude .github *
|
||||
recursive-include docs *.py
|
||||
recursive-include docs *.rst
|
||||
recursive-include docs *.svg
|
||||
recursive-exclude docs/_build *
|
||||
|
||||
|
||||
# support files
|
||||
recursive-include src/ocrmypdf/data *
|
||||
include *.py
|
||||
exclude tasks.py
|
||||
recursive-exclude .travis *
|
||||
exclude .travis*
|
||||
|
||||
|
||||
# code
|
||||
exclude src/ocrmypdf/lib/_leptonica.py
|
||||
exclude scratch.py
|
||||
@@ -1,6 +1,8 @@
|
||||
<img src="docs/images/logo.svg" width="240" alt="OCRmyPDF">
|
||||
|
||||
[![Travis build status][travis]](https://travis-ci.org/jbarlow83/OCRmyPDF) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs]
|
||||
[![Build Status][azure]](https://dev.azure.com/jim0585/ocrmypdf/_build/latest?definitionId=2&branchName=master) [![PyPI version][pypi]](https://pypi.org/project/ocrmypdf/) ![Homebrew version][homebrew] ![ReadTheDocs][docs] ![Python versions][pyversions]
|
||||
|
||||
[azure]: https://dev.azure.com/jim0585/ocrmypdf/_apis/build/status/jbarlow83.OCRmyPDF?branchName=master
|
||||
|
||||
[travis]: https://travis-ci.org/jbarlow83/OCRmyPDF.svg?branch=master "Travis build status"
|
||||
|
||||
@@ -10,6 +12,8 @@
|
||||
|
||||
[docs]: https://readthedocs.org/projects/ocrmypdf/badge/?version=latest "RTD"
|
||||
|
||||
[pyversions]: https://img.shields.io/pypi/pyversions/ocrmypdf "Supported Python versions"
|
||||
|
||||
OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched or copy-pasted.
|
||||
|
||||
```bash
|
||||
@@ -34,7 +38,7 @@ Main features
|
||||
- Keeps the exact resolution of the original embedded images
|
||||
- When possible, inserts OCR information as a "lossless" operation without disrupting any other content
|
||||
- Optimizes PDF images, often producing files smaller than the input file
|
||||
- If requested deskews and/or cleans the image before performing OCR
|
||||
- If requested, deskews and/or cleans the image before performing OCR
|
||||
- Validates input and output files
|
||||
- Distributes work across all available CPU cores
|
||||
- Uses [Tesseract OCR](https://github.com/tesseract-ocr/tesseract) engine to recognize more than [100 languages](https://github.com/tesseract-ocr/tessdata)
|
||||
@@ -46,7 +50,7 @@ For details: please consult the [documentation](https://ocrmypdf.readthedocs.io/
|
||||
Motivation
|
||||
----------
|
||||
|
||||
I searched the web for a free command line tool to OCR PDF files on Linux/UNIX: I found many, but none of them were really satisfying.
|
||||
I searched the web for a free command line tool to OCR PDF files: I found many, but none of them were really satisfying:
|
||||
|
||||
- Either they produced PDF files with misplaced text under the image (making copy/paste impossible)
|
||||
- Or they did not handle accents and multilingual characters
|
||||
@@ -61,7 +65,7 @@ I searched the web for a free command line tool to OCR PDF files on Linux/UNIX:
|
||||
Installation
|
||||
------------
|
||||
|
||||
Linux, UNIX, and macOS are supported. Windows is not directly supported but there is a Docker image available that runs on Windows.
|
||||
Linux, Windows, macOS and FreeBSD are supported. Docker images are also available.
|
||||
|
||||
Users of Debian 9 or later or Ubuntu 16.10 or later may simply
|
||||
|
||||
@@ -75,7 +79,7 @@ and users of Fedora 29 or later may simply
|
||||
dnf install ocrmypdf
|
||||
```
|
||||
|
||||
and macOS users with Homebrew may simply
|
||||
and Homebrew users (macOS, Linux, Windows Subsystem for Linux) may simply
|
||||
|
||||
```bash
|
||||
brew install ocrmypdf
|
||||
@@ -93,7 +97,10 @@ OCRmyPDF uses Tesseract for OCR, and relies on its language packs. For Linux use
|
||||
apt-cache search tesseract-ocr
|
||||
|
||||
# Debian/Ubuntu users
|
||||
apt-get install tesseract-ocr-chi-sim # Example: Install Chinese Simplified language back
|
||||
apt-get install tesseract-ocr-chi-sim # Example: Install Chinese Simplified language pack
|
||||
|
||||
# Arch Linux users
|
||||
pacman -S tesseract-data-eng tesseract-data-deu # Example: Install the English and German language packs
|
||||
```
|
||||
|
||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested.
|
||||
@@ -101,7 +108,7 @@ You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what l
|
||||
Documentation and support
|
||||
-------------------------
|
||||
|
||||
Once ocrmypdf is installed, the built-in help which explains the command syntax and options can be accessed via:
|
||||
Once OCRmyPDF is installed, the built-in help which explains the command syntax and options can be accessed via:
|
||||
|
||||
```bash
|
||||
ocrmypdf --help
|
||||
@@ -109,31 +116,26 @@ ocrmypdf --help
|
||||
|
||||
Our [documentation is served on Read the Docs](https://ocrmypdf.readthedocs.io/en/latest/index.html).
|
||||
|
||||
If you detect an issue, please:
|
||||
|
||||
- Check whether your issue is already known
|
||||
- If no problem report exists on github, please create one here: <https://github.com/jbarlow83/OCRmyPDF/issues>
|
||||
- Describe your problem thoroughly
|
||||
- Append the console output of the script when running the debug mode (`-v 1` option)
|
||||
- If possible provide your input PDF file as well as the content of the temporary folder (using a file sharing service like Dropbox)
|
||||
Please report issues on our [GitHub issues](https://github.com/jbarlow83/OCRmyPDF/issues) page, and follow the issue template for quick response.
|
||||
|
||||
Requirements
|
||||
------------
|
||||
|
||||
Runs on CPython 3.5, 3.6 and 3.7. Requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings.
|
||||
In addition to the required Python version (3.6+), OCRmyPDF requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. OCRmyPDF is pure Python, but uses CFFI to portably generate library bindings. OCRmyPDF works on pretty much everything: Linux, macOS, Windows and FreeBSD.
|
||||
|
||||
Press & Media
|
||||
-------------
|
||||
|
||||
- [Going paperless with OCRmyPDF](https://medium.com/@ikirichenko/going-paperless-with-ocrmypdf-e2f36143f46a)
|
||||
- [Converting a scanned document into a compressed searchable PDF with redactions](https://medium.com/@treyharris/converting-a-scanned-document-into-a-compressed-searchable-pdf-with-redactions-63f61c34fe4c)
|
||||
- [c't 1-2014, page 59](http://heise.de/-2279695): Detailed presentation of OCRmyPDF v1.0 in the leading German IT magazine c't
|
||||
- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](http://heise.de/-2356670)
|
||||
- [c't 1-2014, page 59](https://heise.de/-2279695): Detailed presentation of OCRmyPDF v1.0 in the leading German IT magazine c't
|
||||
- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670)
|
||||
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
|
||||
Business enquiries
|
||||
------------------
|
||||
|
||||
OCRmyPDF would not be the software that it is today is without companies and users choosing to provide support for feature development and consulting enquiries. We are happy to discuss all enquiries, whether for extending the existing feature set, or integrating OCRmyPDF into a larger system.
|
||||
OCRmyPDF would not be the software that it is today without companies and users choosing to provide support for feature development and consulting enquiries. We are happy to discuss all enquiries, whether for extending the existing feature set, or integrating OCRmyPDF into a larger system.
|
||||
|
||||
License
|
||||
-------
|
||||
|
||||
@@ -0,0 +1,251 @@
|
||||
trigger:
|
||||
tags:
|
||||
include:
|
||||
- v*
|
||||
branches:
|
||||
include:
|
||||
- "*"
|
||||
exclude:
|
||||
- "travis"
|
||||
|
||||
stages:
|
||||
- stage: "Test"
|
||||
jobs:
|
||||
- job: Windows
|
||||
pool:
|
||||
vmImage: "vs2017-win2016"
|
||||
strategy:
|
||||
matrix:
|
||||
Python36:
|
||||
python.version: "3.6"
|
||||
Python37:
|
||||
python.version: "3.7"
|
||||
Python38:
|
||||
python.version: "3.8"
|
||||
steps:
|
||||
- task: UsePythonVersion@0
|
||||
inputs:
|
||||
versionSpec: "$(python.version)"
|
||||
- pwsh: |
|
||||
choco install --yes --no-progress --pre tesseract
|
||||
choco install --yes --no-progress python3
|
||||
choco install --yes --no-progress ghostscript
|
||||
choco install --yes --no-progress pngquant
|
||||
displayName: "Install system packages"
|
||||
- pwsh: |
|
||||
refreshenv
|
||||
python -m pip install --upgrade pip wheel
|
||||
python -m pip install -r requirements/main.txt -r requirements/test.txt .
|
||||
displayName: "Install Python packages"
|
||||
- pwsh: |
|
||||
refreshenv
|
||||
$env:pathext += ';.py'
|
||||
# -n auto helps Windows
|
||||
python -m pytest -n auto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml
|
||||
displayName: "Test"
|
||||
- task: PublishTestResults@2
|
||||
inputs:
|
||||
testResultsFiles: "test.xml"
|
||||
testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)"
|
||||
condition: succeededOrFailed()
|
||||
- job: "Ubuntu_1804"
|
||||
pool:
|
||||
vmImage: "ubuntu-18.04"
|
||||
strategy:
|
||||
matrix:
|
||||
Python36:
|
||||
python.version: "3.6"
|
||||
Python37:
|
||||
python.version: "3.7"
|
||||
Python38:
|
||||
python.version: "3.8"
|
||||
steps:
|
||||
- task: UsePythonVersion@0
|
||||
inputs:
|
||||
versionSpec: "$(python.version)"
|
||||
- bash: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
python3-software-properties \
|
||||
curl \
|
||||
ghostscript \
|
||||
img2pdf \
|
||||
libexempi3 \
|
||||
libffi-dev \
|
||||
liblept5 \
|
||||
libsm6 libxext6 libxrender-dev \
|
||||
pngquant \
|
||||
poppler-utils \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-deu \
|
||||
tesseract-ocr-eng \
|
||||
unpaper \
|
||||
zlib1g
|
||||
displayName: "Install system packages"
|
||||
- bash: |
|
||||
curl https://bootstrap.pypa.io/get-pip.py | python3
|
||||
pip3 install -r requirements/main.txt -r requirements/test.txt .
|
||||
displayName: "Install Python packages"
|
||||
- bash: |
|
||||
tesseract --version
|
||||
displayName: "Record versions"
|
||||
- bash: |
|
||||
# -n auto is slower on Linux and breaks on Python 3.8
|
||||
pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml
|
||||
displayName: "Test"
|
||||
- task: PublishTestResults@2
|
||||
inputs:
|
||||
testResultsFiles: "test.xml"
|
||||
testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)"
|
||||
condition: succeededOrFailed()
|
||||
- job: "Ubuntu_1604"
|
||||
pool:
|
||||
vmImage: "ubuntu-16.04"
|
||||
strategy:
|
||||
matrix:
|
||||
Python36:
|
||||
python.version: "3.6"
|
||||
steps:
|
||||
- task: UsePythonVersion@0
|
||||
inputs:
|
||||
versionSpec: "$(python.version)"
|
||||
- bash: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
software-properties-common
|
||||
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
ghostscript \
|
||||
img2pdf \
|
||||
libexempi3 \
|
||||
libffi-dev \
|
||||
liblept5 \
|
||||
libsm6 libxext6 libxrender-dev \
|
||||
pngquant \
|
||||
poppler-utils \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-deu \
|
||||
tesseract-ocr-eng \
|
||||
unpaper \
|
||||
zlib1g
|
||||
displayName: "Install system packages"
|
||||
- bash: |
|
||||
curl https://bootstrap.pypa.io/get-pip.py | python3
|
||||
pip3 install -r requirements/main.txt -r requirements/test.txt .
|
||||
displayName: "Install Python packages"
|
||||
- bash: |
|
||||
tesseract --version
|
||||
displayName: "Record versions"
|
||||
- bash: |
|
||||
# -n auto is slower on Linux and breaks on Python 3.8
|
||||
pytest -n0 --junitxml=test.xml --cov=ocrmypdf --cov-report=xml
|
||||
displayName: "Test"
|
||||
- task: PublishTestResults@2
|
||||
inputs:
|
||||
testResultsFiles: "test.xml"
|
||||
testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)"
|
||||
condition: succeededOrFailed()
|
||||
- job: "macOS_Mojave"
|
||||
pool:
|
||||
vmImage: "macos-10.14"
|
||||
strategy:
|
||||
matrix:
|
||||
Python37:
|
||||
python.version: "3.7"
|
||||
Python38:
|
||||
python.version: "3.8"
|
||||
steps:
|
||||
- task: UsePythonVersion@0
|
||||
inputs:
|
||||
versionSpec: "$(python.version)"
|
||||
- bash: |
|
||||
brew update
|
||||
brew unlink python@2
|
||||
brew install \
|
||||
exempi \
|
||||
ghostscript \
|
||||
jbig2enc \
|
||||
leptonica \
|
||||
openjpeg \
|
||||
pngquant \
|
||||
tesseract \
|
||||
unpaper
|
||||
displayName: "Install system packages"
|
||||
- bash: |
|
||||
pip3 install --upgrade pip
|
||||
pip3 install -r requirements/main.txt -r requirements/test.txt .
|
||||
displayName: "Install Python packages"
|
||||
- bash: |
|
||||
tesseract --version
|
||||
displayName: "Record versions"
|
||||
- bash: pytest -nauto --junitxml=test.xml --cov=ocrmypdf --cov-report=xml
|
||||
displayName: "Test"
|
||||
- task: PublishTestResults@2
|
||||
inputs:
|
||||
testResultsFiles: "test.xml"
|
||||
testRunTitle: "$(Agent.OS) - $(Build.DefinitionName) - Python $(python.version)"
|
||||
condition: succeededOrFailed()
|
||||
- task: PublishCodeCoverageResults@1
|
||||
inputs:
|
||||
codeCoverageTool: Cobertura
|
||||
summaryFileLocation: "$(System.DefaultWorkingDirectory)/**/coverage.xml"
|
||||
|
||||
- stage: "Artifacts"
|
||||
jobs:
|
||||
- job: "sdist_wheel"
|
||||
pool:
|
||||
vmImage: "ubuntu-18.04"
|
||||
steps:
|
||||
- task: UsePythonVersion@0
|
||||
inputs:
|
||||
versionSpec: "3.7"
|
||||
- bash: |
|
||||
python -m pip install --upgrade pip wheel
|
||||
python setup.py sdist bdist_wheel
|
||||
- publish: dist
|
||||
artifact: sdist_wheel
|
||||
|
||||
- stage: "Deploy"
|
||||
jobs:
|
||||
- deployment: "PyPI"
|
||||
pool:
|
||||
vmImage: "ubuntu-18.04"
|
||||
environment: "deploy"
|
||||
strategy:
|
||||
runOnce:
|
||||
deploy:
|
||||
steps:
|
||||
- download: current
|
||||
artifact: sdist_wheel
|
||||
- script: |
|
||||
mkdir -p dist
|
||||
mv $(Pipeline.Workspace)/sdist_wheel/* dist
|
||||
displayName: "Move dist files"
|
||||
- task: UsePythonVersion@0
|
||||
inputs:
|
||||
versionSpec: "3.8"
|
||||
architecture: x64
|
||||
- script: |
|
||||
pip install --upgrade twine
|
||||
displayName: "Generate artifacts"
|
||||
- script: |
|
||||
cat <<FILE >.pypirc
|
||||
[distutils]
|
||||
index-servers =
|
||||
pypi
|
||||
|
||||
[pypi]
|
||||
username: __token__
|
||||
password: $(TOKEN_PYPI)
|
||||
|
||||
FILE
|
||||
displayName: "Generate PyPI auth file"
|
||||
- script: |
|
||||
python -m twine upload --config-file .pypirc dist/*
|
||||
displayName: "Upload to PyPI"
|
||||
condition: and(succeeded(), startsWith(variables['Build.SourceBranch'], 'refs/tags/'))
|
||||
- script: |
|
||||
curl -X POST -d "token=$(TOKEN_RTD)" https://readthedocs.org/api/v2/webhook/pikepdf/39557/
|
||||
displayName: "Trigger ReadTheDocs"
|
||||
condition: and(succeeded(), or(startsWith(variables['Build.SourceBranch'], 'refs/tags/'), startsWith(variables['Build.SourceBranch'], 'refs/heads/master')))
|
||||
Vendored
+2
-8
@@ -79,7 +79,7 @@ Copyright: held by the contributors to the Wikipedia article "Optical character
|
||||
(epson.pdf generated from Wikipedia article as of 2016-09-14)
|
||||
License: CC-BY-SA-3.0
|
||||
|
||||
Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf
|
||||
Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf tests/resources/3small.pdf
|
||||
Copyright: (C) 2005 Ellywa
|
||||
License: GFDL-1.2+ or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0
|
||||
|
||||
@@ -87,7 +87,7 @@ Files: tests/resources/overlay.pdf
|
||||
Copyright: (C) 2017 Max Anderson
|
||||
License: Expat
|
||||
|
||||
Files: tests/resources/baiona*.png
|
||||
Files: tests/resources/baiona*.png tests/resources/3small.pdf
|
||||
Copyright: (C) 2014 Euskaldunaa
|
||||
License: CC-BY-SA-4.0
|
||||
|
||||
@@ -95,12 +95,6 @@ Files: tests/resources/vector.pdf
|
||||
Copyright: (C) 2018 Catscratch
|
||||
License: Expat
|
||||
|
||||
Files: test/resources/enron*.pdf
|
||||
Copyright: EnronData.org
|
||||
License: CC-BY-3.0
|
||||
See: https://enrondata.readthedocs.io/en/latest/data/edo-enron-email-pst-dataset/
|
||||
Comment: Unprocessed.
|
||||
|
||||
Files: src/ocrmypdf/data/sRGB.icc
|
||||
Copyright: Kai-Uwe Behrmann <www.behrmann.name>
|
||||
Marti Maria <www.littlecms.com>
|
||||
|
||||
+17
-9
@@ -322,15 +322,23 @@ working files on a per page basis have the page number as a prefix
|
||||
(starting with page 1), an infix indicates the processing stage, and a
|
||||
suffix indicates the file type. Some important files include:
|
||||
|
||||
- ``.page.png`` - what the input page looks like
|
||||
- ``.image`` - the image we will show the user if we are in a mode that
|
||||
changes the final appearance; may be in one of several image formats
|
||||
- ``.text.pdf`` - the OCR file; this will load as a blank page but
|
||||
should have visible text if checked with a tool like pdftotext or
|
||||
pdfminder.six
|
||||
- ``.ocr.png`` - the file that is sent to Tesseract for OCR; depending
|
||||
- ``_rasterize.png`` - what the input page looks like
|
||||
- ``_ocr.png`` - the file that is sent to Tesseract for OCR; depending
|
||||
on arguments this may differ from the presentation image
|
||||
- ``layers.rendered.pdf`` - the composite PDF, before metadata repair
|
||||
and optimization
|
||||
- ``_pp_deskew.png`` - the image, after deskewing
|
||||
- ``_pp_clean.png`` - the image, after cleaning with unpaper
|
||||
- ``_ocr_tess.pdf`` - the OCR file; appears as a blank page with invisible
|
||||
text embedded
|
||||
- ``_ocr_tess.txt`` - the OCR text (not necessarily all text on the page,
|
||||
if the page is mixed format)
|
||||
- ``fix_docinfo.pdf`` - a temporary file created to fix the PDF DocumentInfo
|
||||
data structure
|
||||
- ``graft_layers.pdf`` - the rendered PDF with OCR layers grafted on
|
||||
- ``pdfa.pdf`` - ``graft_layers.pdf`` after conversion to PDF/A
|
||||
- ``pdfa.ps`` - a PostScript file used by Ghostscript for PDF/A conversion
|
||||
- ``optimize.pdf`` - the PDF generated before optimization
|
||||
- ``optimize.out.pdf`` - the PDF generated by optimization
|
||||
- ``origin`` - the input file
|
||||
- ``origin.pdf`` - the input file or the input image converted to PDF
|
||||
- ``images/*`` - images extracted during the optimization process; here
|
||||
the prefix indicates a PDF object ID not a page number
|
||||
|
||||
+10
-1
@@ -51,6 +51,14 @@ Forking a child process to call ``ocrmypdf.ocr()`` is suggested. That
|
||||
way your application will survive and remain interactive even if
|
||||
OCRmyPDF does not.
|
||||
|
||||
.. warning::
|
||||
|
||||
On Windows, the script that calls ``ocrmypdf.ocr()`` must be protected
|
||||
by an "ifmain" guard (``if __name__ == '__main__'``) or you must use
|
||||
``ocrmypdf.ocr(...use_threads=True)``. If you do not take at least one
|
||||
of these steps, Windows fork semantics will prevent OCRmyPDF from working
|
||||
correct.
|
||||
|
||||
Logging
|
||||
-------
|
||||
|
||||
@@ -71,7 +79,8 @@ Progress monitoring
|
||||
OCRmyPDF uses the ``tqdm`` package to implement its progress bars.
|
||||
:func:`ocrmypdf.configure_logging` will set up logging output to
|
||||
``sys.stderr`` in a way that is compatible with the display of the
|
||||
progress bar.
|
||||
progress bar. Use ``ocrmypdf.ocr(...progress_bar=False)`` to disable
|
||||
the progress bar.
|
||||
|
||||
Exceptions
|
||||
----------
|
||||
|
||||
+90
-129
@@ -47,7 +47,7 @@ where the PDFs are stored):
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
find . -printf '%p' -name '*.pdf' -exec docker run --rm -v <host dir>:<container dir> jbarlow83/ocrmypdf-alpine '<container dir>/{}' '<container dir>/{}' \;
|
||||
find . -printf '%p' -name '*.pdf' -exec docker run --rm -v <host dir>:<container dir> jbarlow83/ocrmypdf '<container dir>/{}' '<container dir>/{}' \;
|
||||
|
||||
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
||||
``find`` to create a directory list and ``parallel`` to parallelize runs
|
||||
@@ -57,59 +57,20 @@ of ``ocrmypdf``, again updating files in place.
|
||||
|
||||
find . -name '*.pdf' | parallel --tag -j 2 ocrmypdf '{}' '{}'
|
||||
|
||||
In a Windows batch file, use
|
||||
|
||||
.. code-block:: bat
|
||||
|
||||
for /r %%f in (*.pdf) do ocrmypdf %%f %%f
|
||||
|
||||
Sample script
|
||||
-------------
|
||||
|
||||
This user contributed script also provides an example of batch
|
||||
processing.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
#!/usr/bin/env python3
|
||||
# Walk through directory tree, replacing all files with OCR'd version
|
||||
# Contributed by DeliciousPickle@github
|
||||
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
||||
print(script_dir + '/ocr-tree.py: Start')
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
start_dir = sys.argv[1]
|
||||
else:
|
||||
start_dir = '.'
|
||||
|
||||
if len(sys.argv) > 2:
|
||||
log_file = sys.argv[2]
|
||||
else:
|
||||
log_file = script_dir + '/ocr-tree.log'
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO, format='%(asctime)s %(message)s',
|
||||
filename=log_file, filemode='w')
|
||||
|
||||
for dir_name, subdirs, file_list in os.walk(start_dir):
|
||||
logging.info('\n')
|
||||
logging.info(dir_name + '\n')
|
||||
os.chdir(dir_name)
|
||||
for filename in file_list:
|
||||
file_ext = os.path.splitext(filename)[1]
|
||||
if file_ext == '.pdf':
|
||||
full_path = dir_name + '/' + filename
|
||||
print(full_path)
|
||||
cmd = ["ocrmypdf", "--deskew", filename, filename]
|
||||
logging.info(cmd)
|
||||
proc = subprocess.run(
|
||||
cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT)
|
||||
result = proc.stdout
|
||||
if proc.returncode == 6:
|
||||
print("Skipped document because it already contained text")
|
||||
elif proc.returncode == 0:
|
||||
print("OCR complete")
|
||||
logging.info(result)
|
||||
.. literalinclude:: ../misc/batch.py
|
||||
:caption: misc/batch.py
|
||||
|
||||
Synology DiskStations
|
||||
---------------------
|
||||
@@ -125,62 +86,8 @@ products use ARM or Power processors and do not support Docker. Further
|
||||
adjustments might be needed to deal with the Synology's relatively
|
||||
limited CPU and RAM.
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
#!/bin/env python3
|
||||
# Contributed by github.com/Enantiomerie
|
||||
|
||||
# script needs 2 arguments
|
||||
# 1. source dir with *.pdf - default is location of script
|
||||
# 2. move dir where *.pdf and *_OCR.pdf are moved to
|
||||
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import shutil
|
||||
|
||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
||||
timestamp = time.strftime("%Y-%m-%d-%H%M_")
|
||||
log_file = script_dir + '/' + timestamp + 'ocrmypdf.log'
|
||||
logging.basicConfig(level=logging.INFO, format='%(asctime)s %(message)s', filename=log_file, filemode='w')
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
start_dir = sys.argv[1]
|
||||
else:
|
||||
start_dir = '.'
|
||||
|
||||
for dir_name, subdirs, file_list in os.walk(start_dir):
|
||||
logging.info('\n')
|
||||
logging.info(dir_name + '\n')
|
||||
os.chdir(dir_name)
|
||||
for filename in file_list:
|
||||
file_ext = os.path.splitext(filename)[1]
|
||||
if file_ext == '.pdf':
|
||||
full_path = dir_name + '/' + filename
|
||||
file_noext = os.path.splitext(filename)[0]
|
||||
timestamp_OCR = time.strftime("%Y-%m-%d-%H%M_OCR_")
|
||||
filename_OCR = timestamp_OCR + file_noext + '.pdf'
|
||||
docker_mount = dir_name + ':/home/docker'
|
||||
# create string for pdf processing
|
||||
# diskstation needs a user:group docker:docker. find uid:gid of your diskstation docker:docker with id docker.
|
||||
# use this uid:gid in -u flag
|
||||
# rw rights for docker:docker at source dir are also necessary
|
||||
# the script is processed as root user via chron
|
||||
cmd = ['docker', 'run', '--rm', '-v', docker_mount, '-u=1030:65538', 'jbarlow83/ocrmypdf', , '--deskew' , filename, filename_OCR]
|
||||
logging.info(cmd)
|
||||
proc = subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.STDOUT)
|
||||
result = proc.stdout.read()
|
||||
logging.info(result)
|
||||
full_path_OCR = dir_name + '/' + filename_OCR
|
||||
os.chmod(full_path_OCR, 0o666)
|
||||
os.chmod(full_path, 0o666)
|
||||
full_path_OCR_archive = sys.argv[2]
|
||||
full_path_archive = sys.argv[2] + '/no_ocr'
|
||||
shutil.move(full_path_OCR,full_path_OCR_archive)
|
||||
shutil.move(full_path, full_path_archive)
|
||||
logging.info('Finished.\n')
|
||||
.. literalinclude:: ../misc/synology.py
|
||||
:caption: misc/synology.py - Sample script for Synology DiskStations
|
||||
|
||||
Huge batch jobs
|
||||
---------------
|
||||
@@ -192,38 +99,83 @@ and all inquiries are appreciated.
|
||||
Hot (watched) folders
|
||||
=====================
|
||||
|
||||
To set up a "hot folder" that will trigger OCR for every file inserted,
|
||||
use a program like Python
|
||||
`watchdog <https://pypi.python.org/pypi/watchdog>`__ (supports all major
|
||||
OS).
|
||||
Watched folders with watcher.py
|
||||
-------------------------------
|
||||
|
||||
One could then configure a scanner to automatically place scanned files
|
||||
in a hot folder, so that they will be queued for OCR and copied to the
|
||||
destination.
|
||||
OCRmyPDF has a folder watcher called watcher.py, which is currently included in source
|
||||
distributions but not part of the main program. It may be used natively or may run
|
||||
in a Docker container. Native instances tend to give better performance. watcher.py
|
||||
works on all platforms.
|
||||
|
||||
Users may need to customize the script to meet their requirements.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pip install watchdog
|
||||
pip3 install -r requirements/watcher.txt
|
||||
|
||||
watchdog installs the command line program ``watchmedo``, which can be
|
||||
told to run ``ocrmypdf`` on any .pdf added to the current directory
|
||||
(``.``) and place the result in the previously created ``out/`` folder.
|
||||
env OCR_INPUT_DIRECTORY=/mnt/input-pdfs \
|
||||
OCR_OUTPUT_DIRECTORY=/mnt/output-pdfs \
|
||||
OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
||||
python3 watcher.py
|
||||
|
||||
.. csv-table:: watcher.py environment variables
|
||||
:header: "Environment variable", "Description"
|
||||
:widths: 50, 50
|
||||
|
||||
"OCR_INPUT_DIRECTORY", "Set input directory to monitor (recursive)"
|
||||
"OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)"
|
||||
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
|
||||
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
|
||||
"OCR_DESKEW", "Apply deskew to crooked input PDFs"
|
||||
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``"
|
||||
"OCR_POLL_NEW_FILE_SECONDS", "Polling interval"
|
||||
"OCR_LOGLEVEL", "Level of log messages to report"
|
||||
|
||||
One could configure a networked scanner or scanning computer to drop files in the
|
||||
watched folder.
|
||||
|
||||
Watched folders with Docker
|
||||
---------------------------
|
||||
|
||||
The watcher service is included in the OCRmyPDF Docker image. To run it:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
cd hot-folder
|
||||
mkdir out
|
||||
watchmedo shell-command \
|
||||
--patterns="*.pdf" \
|
||||
--ignore-directories \
|
||||
--command='ocrmypdf "${watch_src_path}" "out/${watch_src_path}" ' \
|
||||
. # don't forget the final dot
|
||||
docker run \
|
||||
-v <path to files to convert>:/input \
|
||||
-v <path to store results>:/output \
|
||||
-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
|
||||
-e OCR_ON_SUCCESS_DELETE=1 \
|
||||
-e OCR_DESKEW=1 \
|
||||
-e PYTHONUNBUFFERED=1 \
|
||||
-it --entrypoint python3 \
|
||||
jbarlow83/ocrmypdf \
|
||||
watcher.py
|
||||
|
||||
For more complex behavior you can write a Python script around to use
|
||||
the watchdog API.
|
||||
This service will watch for a file that matches ``/input/\*.pdf`` and will
|
||||
convert it to a OCRed PDF in ``/output/``. The parameters to this image are:
|
||||
|
||||
On file servers, you could configure watchmedo as a system service so it
|
||||
will run all the time.
|
||||
.. csv-table:: watcher.py parameters for Docker
|
||||
:header: "Parameter", "Description"
|
||||
:widths: 50, 50
|
||||
|
||||
"``-v <path to files to convert>:/input``", "Files placed in this location will be OCRed"
|
||||
"``-v <path to store results>:/output``", "This is where OCRed files will be stored"
|
||||
"``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1"
|
||||
"``-e OCR_ON_SUCCESS_DELETE=1``", "Define environment variable"
|
||||
"``-e OCR_DESKEW=1``", "Define environment variable"
|
||||
"``-e PYTHONBUFFERED=1``", "This will force STDOUT to be unbuffered and allow you to see messages in docker logs"
|
||||
|
||||
This service relies on polling to check for changes to the filesystem. It
|
||||
may not be suitable for some environments, such as filesystems shared on a
|
||||
slow network.
|
||||
|
||||
A configuration manager such as Docker Compose could be used to ensure that the
|
||||
service is always available.
|
||||
|
||||
.. literalinclude:: ../misc/docker-compose.example.yml
|
||||
:language: yaml
|
||||
:caption: misc/docker-compose.example.yml
|
||||
|
||||
Caveats
|
||||
-------
|
||||
@@ -244,9 +196,19 @@ Caveats
|
||||
Alternatives
|
||||
------------
|
||||
|
||||
- On Linux, `systemd user services <https://wiki.archlinux.org/index.php/Systemd/User>`__
|
||||
can be configured to automatically perform OCR on a collection of files.
|
||||
|
||||
- `Watchman <https://facebook.github.io/watchman/>`__ is a more
|
||||
powerful alternative to ``watchmedo``.
|
||||
|
||||
AWS Lambda is not viable
|
||||
------------------------
|
||||
|
||||
AWS Lambda and its equivalents have low limits on execution time and payload
|
||||
size, relative to OCRmyPDF's needs. As of this writing, the request/response
|
||||
payload for AWS Lambda was 6 MB, which means many PDFs will not fit.
|
||||
|
||||
macOS Automator
|
||||
===============
|
||||
|
||||
@@ -256,8 +218,7 @@ of Automator, the ``PATH`` may be set differently your Terminal's
|
||||
``PATH``; you may need to explicitly set the PATH to include
|
||||
``ocrmypdf``. The following example may serve as a starting point:
|
||||
|
||||
|Example macOS Automator script|
|
||||
.. figure:: images/macos-workflow.png
|
||||
:alt: Example macOS Automator workflow
|
||||
|
||||
You may customize the command sent to ocrmypdf.
|
||||
|
||||
.. |Example macOS Automator script| image:: images/macos-workflow.png
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
=======================
|
||||
Contributing guidelines
|
||||
=======================
|
||||
|
||||
Contributions are welcome!
|
||||
|
||||
Big changes
|
||||
===========
|
||||
|
||||
Please open a new issue to discuss or propose a major change. Not only is it fun
|
||||
to discuss big ideas, but we might save each other's time too. Perhaps some of the
|
||||
work you're contemplating is already half-done in a development branch.
|
||||
|
||||
Code style
|
||||
==========
|
||||
|
||||
We use PEP8, ``black`` for code formatting and ``isort`` for import sorting. The
|
||||
settings for these programs are in ``pyproject.toml`` and ``setup.cfg``. Pull
|
||||
requests should follow the style guide. One difference we use from "black" style
|
||||
is that strings shown to the user are always in double quotes (``"``) and strings
|
||||
for internal uses are in single quotes (``'``).
|
||||
|
||||
Tests
|
||||
=====
|
||||
|
||||
New features should come with tests that confirm their correctness.
|
||||
|
||||
New Python dependencies
|
||||
=======================
|
||||
|
||||
If you are proposing a change that will require a new Python dependency, we
|
||||
prefer dependencies that are already packaged by Debian or Red Hat. This makes
|
||||
life much easier for our downstream package maintainers.
|
||||
|
||||
Python dependencies must also be GPLv3 compatible.
|
||||
|
||||
New non-Python dependencies
|
||||
===========================
|
||||
|
||||
OCRmyPDF uses several external programs (Tesseract, Ghostscript and others) for
|
||||
its functionality. In general we prefer to avoid adding new external programs.
|
||||
|
||||
Style guide: Is it OCRmyPDF or ocrmypdf?
|
||||
========================================
|
||||
|
||||
The program/project is OCRmyPDF and the name of the executable or library is ocrmypdf.
|
||||
|
||||
Known ports/packagers
|
||||
=====================
|
||||
|
||||
OCRmyPDF has been ported to many platforms already. If you are interesting in
|
||||
porting to a new platform, check with
|
||||
`Repology <https://repology.org/projects/?search=ocrmypdf>`__ to see the status
|
||||
of that platform.
|
||||
|
||||
Packager maintainers, please ensure that the command line completion scripts in
|
||||
``misc/`` are installed.
|
||||
@@ -89,6 +89,18 @@ This produces a file named "output.pdf" and a companion text file named
|
||||
|
||||
ocrmypdf --sidecar output.txt input.pdf output.pdf
|
||||
|
||||
.. note::
|
||||
|
||||
The sidecar file contains the **OCR text** found by OCRmyPDF. If the document
|
||||
contains pages that already have text, that text will not appear in the
|
||||
sidecar. If the option ``--pages`` is used, only those pages on which OCR
|
||||
was performed will be included in the sidecar. If certain pages were skipped
|
||||
because of options like ``--skip-big`` or ``--tesseract-timeout``, those pages
|
||||
will not be in the sidecar.
|
||||
|
||||
To extract all text from a PDF, whether generated from OCR or otherwise,
|
||||
use a program like Poppler's ``pdftotext`` or ``pdfgrep``.
|
||||
|
||||
OCR images, not PDFs
|
||||
--------------------
|
||||
|
||||
@@ -216,6 +228,41 @@ processing or PDF/A conversion.
|
||||
|
||||
ocrmypdf --tesseract-timeout=0 --remove-background input.pdf output.pdf
|
||||
|
||||
Optimize images without performing OCR
|
||||
--------------------------------------
|
||||
|
||||
You can also optimize all images without performing any OCR:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --tesseract-timeout=0 --optimize 3 --skip-text input.pdf output.pdf
|
||||
|
||||
Perform OCR only certain pages
|
||||
------------------------------
|
||||
|
||||
You can ask OCRmyPDF to only apply OCR to certain pages.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --pages 2,3,13-17 input.pdf output.pdf
|
||||
|
||||
Hyphens denote a range of pages and commas separate page numbers. If you prefer
|
||||
to use spaces, quote all of the page numbers: ``--pages '2, 3, 5, 7'``.
|
||||
|
||||
OCRmyPDF will warn if your list of page numbers contains duplicates or
|
||||
overlap pages. OCRmyPDF does not currently account for document page numbers,
|
||||
such as an introduction section of a book that uses Roman numerals. It simply
|
||||
counts the number of virtual pieces of paper since the start.
|
||||
|
||||
Regardless of the argument to ``--pages``, OCRmyPDF will optimize all pages in
|
||||
the file and convert it to PDF/A, unless you disable those options. In this
|
||||
example, we want to OCR only the title and otherwise change the PDF as little
|
||||
as possible:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
ocrmypdf --pages 1 --output-type pdf --optimize 0 input.pdf output.pdf
|
||||
|
||||
Redo existing OCR
|
||||
=================
|
||||
|
||||
|
||||
+42
-47
@@ -22,21 +22,20 @@ Installing the Docker image
|
||||
If you have `Docker <https://docs.docker.com/>`__ installed on your
|
||||
system, you can install a Docker image of the latest release.
|
||||
|
||||
The recommended OCRmyPDF Docker image is currently named
|
||||
``ocrmypdf-alpine``:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker pull jbarlow83/ocrmypdf-alpine
|
||||
|
||||
Follow the Docker installation instructions for your platform. If you
|
||||
can run this command successfully, your system is ready to download and
|
||||
If you can run this command successfully, your system is ready to download and
|
||||
execute the image:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run hello-world
|
||||
|
||||
The recommended OCRmyPDF Docker image is currently named ``ocrmypdf``:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker pull jbarlow83/ocrmypdf
|
||||
|
||||
|
||||
OCRmyPDF will use all available CPU cores. By default, the VirtualBox
|
||||
machine instance on Windows and macOS has only a single CPU core
|
||||
enabled. Use the VirtualBox Manager to determine the name of your Docker
|
||||
@@ -51,64 +50,67 @@ CPUs:
|
||||
docker-machine start "yourVM"
|
||||
eval $(docker-machine env "yourVM")
|
||||
|
||||
See the Docker documentation for
|
||||
`adjusting memory and CPU on other platforms <https://docs.docker.com/config/containers/resource_constraints/>`__.
|
||||
|
||||
Using the Docker image on the command line
|
||||
==========================================
|
||||
|
||||
**Unlike typical Docker containers**, in this mode we are using the
|
||||
OCRmyPDF Docker container is intended to be emphemeral – it runs for one
|
||||
OCR job and then terminates, just like a command line program. We are
|
||||
using Docker as a way of delivering an application, not a server.
|
||||
**Unlike typical Docker containers**, in this section the OCRmyPDF Docker
|
||||
container is emphemeral – it runs for one OCR job and terminates, just like a
|
||||
command line program. We are using Docker to deliver an application (as opposed
|
||||
to the more conventional case, where a Docker container runs as a server).
|
||||
|
||||
To start a Docker container (instance of the image):
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker tag jbarlow83/ocrmypdf-alpine ocrmypdf
|
||||
docker run --rm -i ocrmypdf (... all other arguments here...)
|
||||
docker tag jbarlow83/ocrmypdf ocrmypdf
|
||||
docker run --rm -i ocrmypdf (... all other arguments here...) - -
|
||||
|
||||
For convenience, create a shell alias to hide the Docker command. It is
|
||||
easier to send the input file to file stdin and read the output from
|
||||
stdout – this avoids the occasionally messy permission issues with
|
||||
Docker entirely.
|
||||
easier to send the input file as stdin and read the output from
|
||||
stdout – **this avoids the messy permission issues with Docker entirely**.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
alias ocrmypdf='docker run --rm -i ocrmypdf'
|
||||
ocrmypdf --version # runs docker version
|
||||
ocrmypdf <input.pdf >output.pdf
|
||||
alias docker_ocrmypdf='docker run --rm -i ocrmypdf'
|
||||
docker_ocrmypdf --version # runs docker version
|
||||
docker_ocrmypdf - - <input.pdf >output.pdf
|
||||
|
||||
Or in the wonderful `fish shell <https://fishshell.com/>`__:
|
||||
|
||||
.. code-block:: fish
|
||||
|
||||
alias ocrmypdf 'docker run --rm ocrmypdf'
|
||||
funcsave ocrmypdf
|
||||
alias docker_ocrmypdf 'docker run --rm ocrmypdf'
|
||||
funcsave docker_ocrmypdf
|
||||
|
||||
Alternately, you could mount the local current working directory as a
|
||||
Docker volume:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run --rm -v $(pwd):/data ocrmypdf /data/input.pdf /data/output.pdf
|
||||
alias docker_ocrmypdf='docker run --rm -i --user "$(id -u):$(id -g)" --workdir /data -v "$PWD:/data" ocrmypdf'
|
||||
docker_ocrmypdf /data/input.pdf /data/output.pdf
|
||||
|
||||
.. _docker-lang-packs:
|
||||
|
||||
Adding languages to the Docker image
|
||||
====================================
|
||||
|
||||
By default the Docker image includes English, German and Simplified
|
||||
Chinese, the most popular languages for OCRmyPDF users based on
|
||||
feedback. You may add other languages by creating a new Dockerfile based
|
||||
on the public one:
|
||||
By default the Docker image includes English, German, Simplified Chinese,
|
||||
French, Portuguese and Spanish, the most popular languages for OCRmyPDF
|
||||
users based on feedback. You may add other languages by creating a new
|
||||
Dockerfile based on the public one:
|
||||
|
||||
.. code-block:: dockerfile
|
||||
|
||||
FROM jbarlow83/ocrmypdf-alpine
|
||||
FROM jbarlow83/ocrmypdf
|
||||
|
||||
# Add French
|
||||
RUN apk add tesseract-ocr-data-fra
|
||||
RUN apt install tesseract-ocr-fra
|
||||
|
||||
You can also copy training data to ``/usr/share/tessdata``.
|
||||
You can also copy training data to ``/usr/share/tesseract-ocr/<tesseract version>/tessdata``.
|
||||
|
||||
Executing the test suite
|
||||
========================
|
||||
@@ -117,17 +119,16 @@ The OCRmyPDF test suite is installed with image. To run it:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run --entrypoint python3 jbarlow83/ocrmypdf-alpine setup.py test
|
||||
docker run --entrypoint python3 jbarlow83/ocrmypdf -m pytest
|
||||
|
||||
Accessing the shell
|
||||
===================
|
||||
|
||||
``bash`` is not installed in the image. To use the busybox shell in the
|
||||
Docker image:
|
||||
To use the bash shell in the Docker image:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run -it --entrypoint busybox jbarlow83/ocrmypdf-alpine sh
|
||||
docker run -it --entrypoint bash jbarlow83/ocrmypdf
|
||||
|
||||
Using the OCRmyPDF web service wrapper
|
||||
======================================
|
||||
@@ -137,7 +138,12 @@ service. The webservice may be launched as follows:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker run --entrypoint python3 -p 5000:5000 jbarlow83/ocrmypdf-alpine webservice.py
|
||||
docker run --entrypoint python3 -p 5000:5000 jbarlow83/ocrmypdf webservice.py
|
||||
|
||||
This will configure the machine to listen on port 5000. On Linux machines
|
||||
this is port 5000 of localhost. On macOS or Windows machines running
|
||||
Docker, this is port 5000 of the virtual machine that runs your Docker
|
||||
images. You can find its IP address using the command ``docker-machine ip``.
|
||||
|
||||
Unlike command line usage this program will open a socket and wait for
|
||||
connections.
|
||||
@@ -162,14 +168,3 @@ also licensed in this way.
|
||||
|
||||
In addition to the above, please read our
|
||||
:ref:`general remarks on using OCRmyPDF as a service <ocr-service>`.
|
||||
|
||||
Ubuntu-based Docker image
|
||||
=========================
|
||||
|
||||
A Ubuntu-based OCRmyPDF image is also available. The main advantage this
|
||||
image offers is that it supports manylinux Python wheels (which are not
|
||||
supported on Alpine Linux). This may be useful for plugins.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
docker pull jbarlow83/ocrmypdf
|
||||
|
||||
+12
-4
@@ -1,10 +1,12 @@
|
||||
OCRmyPDF documentation
|
||||
======================
|
||||
|
||||
OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to
|
||||
be searched.
|
||||
OCRmyPDF adds an optical charcter recognition (OCR) text layer to scanned PDF
|
||||
files, allowing them to be searched.
|
||||
|
||||
PDF is the best format for storing and exchanging scanned documents. Unfortunately, PDFs can be difficult to modify. OCRmyPDF makes it easy to apply image processing and OCR to existing PDFs.
|
||||
PDF is the best format for storing and exchanging scanned documents.
|
||||
Unfortunately, PDFs can be difficult to modify. OCRmyPDF makes it easy to apply
|
||||
image processing and OCR to existing PDFs.
|
||||
|
||||
.. toctree::
|
||||
:maxdepth: 1
|
||||
@@ -12,6 +14,7 @@ PDF is the best format for storing and exchanging scanned documents. Unfortunat
|
||||
introduction
|
||||
release_notes
|
||||
installation
|
||||
optimizer
|
||||
languages
|
||||
jbig2
|
||||
|
||||
@@ -22,11 +25,16 @@ PDF is the best format for storing and exchanging scanned documents. Unfortunat
|
||||
cookbook
|
||||
docker
|
||||
advanced
|
||||
api
|
||||
batch
|
||||
security
|
||||
errors
|
||||
|
||||
.. toctree::
|
||||
:caption: Developers
|
||||
:maxdepth: 2
|
||||
|
||||
api
|
||||
contributing
|
||||
|
||||
Indices and tables
|
||||
==================
|
||||
|
||||
+181
-59
@@ -8,11 +8,20 @@ Installing OCRmyPDF
|
||||
|latest|
|
||||
|
||||
The easiest way to install OCRmyPDF is to follow the steps for your operating
|
||||
system/platform, although sometimes this version may be out of date.
|
||||
system/platform, although sometimes this version may be out of date. This
|
||||
installation guide provides information allowing you to compare the current
|
||||
version to the one provided by your platform.
|
||||
|
||||
If you want to use the latest version of OCRmyPDF, your best bet is to install
|
||||
the most recent version your platform provides, and then upgrade that version by
|
||||
installing the Python binary wheels.
|
||||
If you want to use the latest version of OCRmyPDF and all of its optional
|
||||
dependencies, the easiest way to get that is install the Homebrew package. Homebrew
|
||||
is best known as a macOS package manger, but also works for
|
||||
`Linux and Windows Subsystem for Linux <https://docs.brew.sh/Homebrew-on-Linux>`__.
|
||||
After Homebrew is installed, simply run ``brew install ocrmypdf``.
|
||||
|
||||
You can also use the more detailed procedures here to manually install OCRmyPDF
|
||||
from source or with the ``pip`` package manager for binary wheels. The reason
|
||||
for these varied steps is that OCRmyPDF requires third-party executables that are
|
||||
not part of Python.
|
||||
|
||||
.. contents:: Platform-specific steps
|
||||
:depth: 2
|
||||
@@ -55,8 +64,8 @@ Debian and Ubuntu 18.04 or newer
|
||||
| |ubu-1804| |ubu-1810| |ubu-1904| |ubu-1910| |
|
||||
+-----------------------------------------------+
|
||||
|
||||
Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later may
|
||||
simply
|
||||
Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later, including users
|
||||
of Windows Subsystem for Linux, may simply
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
@@ -101,7 +110,7 @@ Fedora 29 or newer
|
||||
| |fedora-29| |fedora-30| |fedora-rawhide| |
|
||||
+-----------------------------------------------+
|
||||
|
||||
Users of Fedora 29 later may simply
|
||||
Users of Fedora 29 or later may simply
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
@@ -127,24 +136,32 @@ from sources <#installing-head-revision-from-sources>`__.
|
||||
Installing the latest version on Ubuntu 18.04 LTS
|
||||
-------------------------------------------------
|
||||
|
||||
Ubuntu 18.04 includes ocrmypdf 6.1.2. To install a more recent version,
|
||||
first install the system version to get most of the dependencies:
|
||||
Ubuntu 18.04 includes ocrmypdf 6.1.2 - you can install that with ``apt``, but
|
||||
it is quite old now. To install a more recent version, first install several
|
||||
system dependencies:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo apt-get update
|
||||
sudo apt-get install \
|
||||
ocrmypdf \
|
||||
python3-pip
|
||||
|
||||
There are a few system dependency changes since ocrmypdf 6.1.2. Let's
|
||||
get these, too.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo apt-get install \
|
||||
sudo apt-get -y update
|
||||
sudo apt-get -y install \
|
||||
ghostscript \
|
||||
icc-profiles-free \
|
||||
liblept5 \
|
||||
libxml2 \
|
||||
pngquant
|
||||
pngquant \
|
||||
python3-cffi \
|
||||
python3-distutils \
|
||||
python3-pkg-resources \
|
||||
python3-reportlab \
|
||||
qpdf \
|
||||
tesseract-ocr \
|
||||
zlib1g
|
||||
|
||||
We will need a newer version of ``pip`` then was available for Ubuntu 18.04:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
wget https://bootstrap.pypa.io/get-pip.py && python3 get-pip.py
|
||||
|
||||
Then install the most recent ocrmypdf for the local user and set the
|
||||
user's ``PATH`` to check for the user's Python packages.
|
||||
@@ -152,7 +169,7 @@ user's ``PATH`` to check for the user's Python packages.
|
||||
.. code-block:: bash
|
||||
|
||||
export PATH=$HOME/.local/bin:$PATH
|
||||
pip3 install --user ocrmypdf
|
||||
python3 -m pip install --user ocrmypdf
|
||||
|
||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||
|
||||
@@ -200,7 +217,8 @@ of ``pip`` at ``/usr/local/bin/pip``.
|
||||
**Install OCRmyPDF**
|
||||
|
||||
OCRmyPDF requires the locale to be set for UTF-8. **On some minimal
|
||||
Ubuntu installations systems**, it may be necessary to set the locale.
|
||||
Ubuntu installations**, such as the Ubuntu 16.04 Docker images it may be
|
||||
necessary to set the locale.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
@@ -214,7 +232,7 @@ environment variable contains ``$HOME/.local/bin``.
|
||||
.. code-block:: bash
|
||||
|
||||
export PATH=$HOME/.local/bin:$PATH
|
||||
pip3 install --user ocrmypdf
|
||||
pip3.6 install --user ocrmypdf
|
||||
|
||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||
|
||||
@@ -272,35 +290,107 @@ Now we need to install ``pip`` and let it install ocrmypdf:
|
||||
curl https://bootstrap.pypa.io/ez_setup.py -o - | python3.6 && python3.6 -m easy_install pip
|
||||
pip3.6 install ocrmypdf
|
||||
|
||||
These installation instructions omit the optional dependency
|
||||
``unpaper``, which is only available at version 0.4.2 in Ubuntu 14.04.
|
||||
The author could not find a backport of ``unpaper``, and created a .deb
|
||||
package to do the job of installing unpaper 6.1 (for x86 64-bit only):
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O unpaper_6.1-1.deb
|
||||
sudo dpkg -i unpaper_6.1-1.deb
|
||||
The optional dependency ``unpaper`` is only available at 0.4.2 in Ubuntu 14.04,
|
||||
and no backports are available. Previously the author maintained a backported
|
||||
.deb package for unpaper 6.1, but since Ubuntu 14.04 is now end of life, this is
|
||||
not supported. As such, ``unpaper`` is not available on Ubuntu 14.04 or must by
|
||||
compiled by hand.
|
||||
|
||||
To add JBIG2 encoding, see :ref:`jbig2`.
|
||||
|
||||
ArchLinux (AUR)
|
||||
---------------
|
||||
Arch Linux (AUR)
|
||||
----------------
|
||||
|
||||
.. image:: https://repology.org/badge/version-for-repo/aur/ocrmypdf.svg
|
||||
:alt: ArchLinux
|
||||
:target: https://repology.org/metapackage/ocrmypdf
|
||||
|
||||
There is an `ArchLinux User Repository package for
|
||||
ocrmypdf <https://aur.archlinux.org/packages/ocrmypdf/>`__. You can use
|
||||
the following command.
|
||||
There is an `Arch User Repository (AUR) package for OCRmyPDF
|
||||
<https://aur.archlinux.org/packages/ocrmypdf/>`__.
|
||||
|
||||
Installing AUR packages as root is not allowed, so you must first `setup a
|
||||
non-root user
|
||||
<https://wiki.archlinux.org/index.php/Users_and_groups#User_management>`__ and
|
||||
`configure sudo <https://wiki.archlinux.org/index.php/Sudo#Configuration>`__.
|
||||
The standard Docker image, ``archlinux/base:latest``, does **not** have a
|
||||
non-root user configured, so users of that image must follow these guides. If
|
||||
you are using a VM image, such as `the official Vagrant image
|
||||
<https://app.vagrantup.com/archlinux/boxes/archlinux>`__, this work may already
|
||||
be completed for you.
|
||||
|
||||
Next you should install the `base-devel package group
|
||||
<https://www.archlinux.org/groups/x86_64/base-devel/>`__. This includes the
|
||||
standard tooling needed to build packages, such as a compiler and binary tools.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
yaourt -S ocrmypdf
|
||||
sudo pacman -S base-devel
|
||||
|
||||
If you have any difficulties with installation, check the repository
|
||||
package page.
|
||||
The OCRmyPDF package depends on `the python-pdfminer.six AUR package
|
||||
<https://aur.archlinux.org/packages/python-pdfminer.six/>`__. Dependencies on
|
||||
AUR packages are not automatically resolved, so this package must be manually
|
||||
installed first.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
curl -O https://aur.archlinux.org/cgit/aur.git/snapshot/python-pdfminer.six.tar.gz
|
||||
tar xvzf python-pdfminer.six.tar.gz
|
||||
cd python-pdfminer.six
|
||||
makepkg -sri
|
||||
|
||||
With that complete you can then repeat the same series of steps for the
|
||||
OCRmyPDF package.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
curl -O https://aur.archlinux.org/cgit/aur.git/snapshot/ocrmypdf.tar.gz
|
||||
tar xvzf ocrmypdf.tar.gz
|
||||
cd ocrmypdf
|
||||
makepkg -sri
|
||||
|
||||
At this point you will have a working install of OCRmyPDF, but the Tesseract
|
||||
install won’t include any OCR language data. You can install `the
|
||||
tesseract-data package group
|
||||
<https://www.archlinux.org/groups/any/tesseract-data/>`__ to add all supported
|
||||
languages, or use that package listing to identify the appropriate package for
|
||||
your desired language.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
sudo pacman -S tesseract-data-eng
|
||||
|
||||
As an alternative to this manual procedure, consider using an `AUR helper
|
||||
<https://wiki.archlinux.org/index.php/AUR_helpers>`__. Such a tool will
|
||||
automatically fetch, build and install the AUR package, resolve dependencies
|
||||
(including dependencies on AUR packages), and ease the upgrade procedure.
|
||||
|
||||
If you have any difficulties with installation, check the repository package
|
||||
page.
|
||||
|
||||
.. note::
|
||||
|
||||
The OCRmyPDF AUR package currently omits the JBIG2 encoder. OCRmyPDF works
|
||||
fine without it but will produce larger output files. The encoder is
|
||||
available from `the jbig2enc-git AUR package
|
||||
<https://aur.archlinux.org/packages/jbig2enc-git/>`__ and may be installed
|
||||
using the same series of steps as for the installation of the pdfminer.six
|
||||
and OCRmyPDF AUR packages. Alternatively, it may be built manually from
|
||||
source following the instructions in `Installing the JBIG2 encoder
|
||||
<jbig2>`__. If JBIG2 is installed, OCRmyPDF 7.0.0 and later will
|
||||
automatically detect it.
|
||||
|
||||
Alpine Linux
|
||||
------------
|
||||
|
||||
.. image:: https://repology.org/badge/version-for-repo/alpine_edge/ocrmypdf.svg
|
||||
:alt: Alpine Linux
|
||||
:target: https://repology.org/metapackage/ocrmypdf
|
||||
|
||||
To install OCRmyPDF for Alpine Linux:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
apk add ocrmypdf
|
||||
|
||||
Other Linux packages
|
||||
--------------------
|
||||
@@ -379,7 +469,7 @@ packs. If you need other languages you can optionally install them all:
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
brew install tesseract --with-all-languages # Option 2: for all language packs
|
||||
brew install tesseract-lang # Option 2: for all language packs
|
||||
|
||||
Update the homebrew pip:
|
||||
|
||||
@@ -412,12 +502,12 @@ Installing on FreeBSD
|
||||
:alt: FreeBSD
|
||||
:target: https://repology.org/project/python:ocrmypdf/versions
|
||||
|
||||
FreeBSD 11.2, 11.3, 12.0-RELEASE and 13.0-CURRENT are supported. Other
|
||||
FreeBSD 11.3, 12.0, 12.1-RELEASE and 13.0-CURRENT are supported. Other
|
||||
versions likely work but have not been tested.
|
||||
|
||||
.. code-block:: bash
|
||||
|
||||
pkg install py36-ocrmypdf
|
||||
pkg install py37-ocrmypdf
|
||||
|
||||
To install a more recent version, you could attempt to first install the system
|
||||
version with ``pkg``, then use ``pip install --user ocrmypdf``.
|
||||
@@ -426,16 +516,57 @@ Installing the Docker image
|
||||
===========================
|
||||
|
||||
For some users, installing the Docker image will be easier than
|
||||
installing all of OCRmyPDF's dependencies. For Windows, it is the only
|
||||
option.
|
||||
installing all of OCRmyPDF's dependencies.
|
||||
|
||||
See `OCRmyPDF Docker Image <docker>`__ for more information.
|
||||
|
||||
Installing on Windows
|
||||
=====================
|
||||
|
||||
Direct installation on Windows is not currently possible, but it works well in
|
||||
Windows Subsystem for Linux:
|
||||
.. warning::
|
||||
|
||||
Native Windows support is new. Consider it "beta" software. Some
|
||||
functionality is missing or may be more difficult to enable. If you need a
|
||||
production-ready solution, use Windows Subsystem for Linux or a Docker
|
||||
image.
|
||||
|
||||
.. note::
|
||||
|
||||
Administrator privileges will be required for some of these steps.
|
||||
|
||||
You must install the following for Windows:
|
||||
|
||||
* Python 3.7 (64-bit)
|
||||
* Tesseract 4.0 or later
|
||||
* Ghostscript 9.50 or later
|
||||
|
||||
You can install these with the Chocolatey package manager:
|
||||
|
||||
* ``choco install python3``
|
||||
* ``choco install --pre tesseract``
|
||||
* ``choco install ghostscript``
|
||||
|
||||
Also consider adding:
|
||||
|
||||
* ``choco install pngquant``
|
||||
|
||||
Windows 10 64-bit and 64-bit versions of applications are recommended. Earlier
|
||||
versions of Windows and 32-bit versions of these programs are not tested, and not
|
||||
supported at this time.
|
||||
|
||||
OCRmyPDF will check for Tesseract-OCR and Ghostscript in your Program Files folder.
|
||||
If they are in some other location, you may need to modify the ``PATH``
|
||||
environment variable so Tesseract, Ghostscript, and other any optional executables can
|
||||
be found. You can enter it in the command line or
|
||||
`follow these directions <https://www.computerhope.com/issues/ch000549.htm#dospath>`_
|
||||
to make the change persistent and system-wide.
|
||||
|
||||
You may then use pip to install ocrmypdf:
|
||||
|
||||
* ``pip install ocrmypdf``
|
||||
|
||||
Installing on Windows Subsystem for Linux
|
||||
=========================================
|
||||
|
||||
#. Install Ubuntu 18.04 for Windows Subsystem for Linux, if not already installed.
|
||||
#. Follow the procedure to install :ref:`OCRmyPDF on Ubuntu 18.04 <ubuntu-lts-latest>`.
|
||||
@@ -443,7 +574,7 @@ Windows Subsystem for Linux:
|
||||
|
||||
.. code-block:: powershell
|
||||
|
||||
wsl sudo ln -s /home/user/.local/bin/ocrmypdf /usr/local/bin/ocrmypdf
|
||||
wsl sudo ln -s /home/$USER/.local/bin/ocrmypdf /usr/local/bin/ocrmypdf
|
||||
|
||||
Then confirm that the expected version from PyPI (|latest|) is installed:
|
||||
|
||||
@@ -454,15 +585,6 @@ Then confirm that the expected version from PyPI (|latest|) is installed:
|
||||
You can then run OCRmyPDF in the Windows command prompt or Powershell, prefixing
|
||||
``wsl``, and call it from Windows programs or batch files.
|
||||
|
||||
Why no native Windows?
|
||||
^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
It would probably not be too difficult to port on Windows. The main
|
||||
reason this has been avoided is the difficulty of packaging and
|
||||
installing the various non-Python dependencies: Tesseract, QPDF,
|
||||
Ghostscript, Leptonica. Pull requests to add or improve Windows support
|
||||
would be quite welcome.
|
||||
|
||||
Docker
|
||||
^^^^^^
|
||||
|
||||
@@ -517,7 +639,7 @@ manager. ``pip`` cannot provide them.
|
||||
|
||||
As of ocrmypdf 7.2.1, the following versions are recommended:
|
||||
|
||||
- Python 3.7
|
||||
- Python 3.7 or 3.8
|
||||
- Ghostscript 9.23 or newer
|
||||
- qpdf 8.2.1
|
||||
- Tesseract 4.0.0 or newer
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
Introduction
|
||||
============
|
||||
|
||||
OCRmyPDF is a Python 3 package that adds OCR layers to PDFs.
|
||||
OCRmyPDF is a Python 3 application and library that adds OCR layers to PDFs.
|
||||
|
||||
About OCR
|
||||
=========
|
||||
@@ -10,7 +10,7 @@ About OCR
|
||||
`Optical character
|
||||
recognition <https://en.wikipedia.org/wiki/Optical_character_recognition>`__
|
||||
is technology that converts images of typed or handwritten text, such as
|
||||
in a scanned document, to computer text that can be searched and copied.
|
||||
in a scanned document, to computer text that can be selected, searched and copied.
|
||||
|
||||
OCRmyPDF uses
|
||||
`Tesseract <https://github.com/tesseract-ocr/tesseract>`__, the best
|
||||
@@ -208,7 +208,7 @@ consider one of these similar open source programs:
|
||||
Web front-ends
|
||||
==============
|
||||
|
||||
The Docker image ``ocrmypdf-alpine`` provides a web service front-end
|
||||
The Docker image ``ocrmypdf`` provides a web service front-end
|
||||
that allows files to submitted over HTTP and the results "downloaded".
|
||||
This is an HTTP server intended to simplify web services deployments; it
|
||||
is not intended to be deployed on the public internet and no real
|
||||
|
||||
+16
-2
@@ -4,11 +4,20 @@
|
||||
Installing additional language packs
|
||||
====================================
|
||||
|
||||
OCRmyPDF uses Tesseract for OCR, and relies on its language packs for
|
||||
languages other than English.
|
||||
OCRmyPDF uses Tesseract for OCR, and relies on its language packs for all languages.
|
||||
On most platforms, English is installed with Tesseract by default, but not always.
|
||||
|
||||
Tesseract supports `most
|
||||
languages <https://github.com/tesseract-ocr/tesseract/blob/master/doc/tesseract.1.asc#languages>`__.
|
||||
Languages are identified by standardized three-letter codes (called ISO 639-2 Alpha-3).
|
||||
Tesseract's documentation also lists the three-letter code for your language.
|
||||
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
||||
are not, e.g. German is ``deu``.
|
||||
|
||||
After you have installed a language pack, you can use it ``ocrmypdf -l <language>``,
|
||||
for example ``ocrmypdf -l spa``. For multilingual documents, you can specify
|
||||
all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French.
|
||||
English is assumed by default unless other language(s) are specified.
|
||||
|
||||
For Linux users, you can often find packages that provide language
|
||||
packs:
|
||||
@@ -57,3 +66,8 @@ Docker users
|
||||
Users of the OCRmyPDF Docker image should install language packs into a
|
||||
derived Docker image as
|
||||
:ref:`described in that section <docker-lang-packs>`.
|
||||
|
||||
Windows users
|
||||
=============
|
||||
|
||||
The Tesseract installer provided by Chocolatey already includes 100 languages.
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
================
|
||||
PDF optimization
|
||||
================
|
||||
|
||||
OCRmyPDF includes an image-oriented PDF optimizer. By default, the optimizer
|
||||
runs with safe settings with the goal of improving compression at no loss of
|
||||
quality. At higher optimization levels, lossy optimizations may be applied and
|
||||
tuned. Optimization occurs after OCR, and only if OCR succeeded. It does not
|
||||
perform other possible optimizations such as deduplicating resources,
|
||||
consolidating fonts, simplifying vector drawings, or anything of that nature.
|
||||
|
||||
Optimization ranges from ``-O0`` through ``-O3``, where ``0`` disables
|
||||
optimization and ``3`` implements all options. ``1``, the default, performs only
|
||||
safe and lossless optimizations. (This is similar to GCC's optimization
|
||||
parameter.) The exact type of optimizations performed will vary over time.
|
||||
|
||||
PDF optimization requires third-party, optional tools for certain optimizations.
|
||||
If these are not installed or cannot be found by OCRmyPDF, optimization will not
|
||||
be as good.
|
||||
|
||||
Optimizations that always occurs
|
||||
================================
|
||||
|
||||
OCRmyPDF will automatically replace obsolete or inferior compression schemes
|
||||
such as RLE or LZW with superior schemes such as Deflate and converting
|
||||
monochrome images to CCITT G4. Since this is harmless it always occurs and there
|
||||
is no way to disable it. Other non-image compressed objects are compressed as
|
||||
well.
|
||||
|
||||
Fast web view
|
||||
=============
|
||||
|
||||
OCRmyPDF automatically optimizes PDFs for "fast web view" in Adobe Acrobat's
|
||||
parlance, or equivalently, linearizes PDFs so that the resources they reference
|
||||
are presented in the order a viewer needs them for sequential display. This
|
||||
reduces the latency of viewing a PDF both online and from local storage. This
|
||||
actually slightly increases the file size.
|
||||
|
||||
To disable this optimization and all others, use ``ocrmypdf --optimize 0 ...``
|
||||
or the shorthand ``-O0``.
|
||||
|
||||
Lossless optimizations
|
||||
======================
|
||||
|
||||
At optimization level ``-O1`` (the default), OCRmyPDF will also attempt lossless
|
||||
image optimization.
|
||||
|
||||
If a JBIG2 encoder is available, then monochrome images will be converted to
|
||||
JBIG2, with the potential for huge savings on large black and white images,
|
||||
since JBIG2 is far more efficient than any other monochrome (bi-level)
|
||||
compression. (All known US patents related to JBIG2 have probably expired, but
|
||||
it remains the responsibility of the user to supply a JBIG2 encoder such as
|
||||
`jbig2enc <https://github.com/agl/jbig2enc>`__. OCRmyPDF does not implement
|
||||
JBIG2 encoding on its own.)
|
||||
|
||||
OCRmyPDF currently does not attempt to recompress losslessly compressed objects
|
||||
more aggressively.
|
||||
|
||||
Lossy optimizations
|
||||
===================
|
||||
|
||||
At optimization level ``-O2`` and ``-O3``, OCRmyPDF will some attempt lossy
|
||||
image optimization.
|
||||
|
||||
If ``pngquant`` is installed, OCRmyPDF will use it to perform quantize paletted
|
||||
images to reduce their size.
|
||||
|
||||
The quality of JPEGs may be lowered, on the assumption that a lower quality
|
||||
image may be suitable for storage after OCR.
|
||||
|
||||
It is not possible to optimize all image types. Uncommon image types may be
|
||||
skipped by the optimizer.
|
||||
|
||||
OCRmyPDF provides :ref:`lossy mode JBIG2 <jbig2-lossy>` as an advanced feature
|
||||
that additional requires the argument ``--jbig2-lossy``.
|
||||
@@ -13,6 +13,157 @@ Note that it is licensed under GPLv3, so scripts that
|
||||
``import ocrmypdf`` and are released publicly should probably also be
|
||||
licensed under GPLv3.
|
||||
|
||||
v9.7.2
|
||||
======
|
||||
|
||||
- Fixed an issue with ``ocrmypdf.ocr(...language=)`` not accepting a list of
|
||||
languages as documented.
|
||||
- Updated setup.py to confirm that pdfminer.six version 20200402 is supported.
|
||||
|
||||
v9.7.1
|
||||
======
|
||||
|
||||
- Fixed version check failing when used with qpdf 10.0.0.
|
||||
- Added some missing type annotations.
|
||||
- Updated documentation to warn about need for "ifmain" guard and Windows.
|
||||
|
||||
v9.7.0
|
||||
======
|
||||
|
||||
- Fixed an error in watcher.py if ``OCR_JSON_SETTINGS`` was not defined.
|
||||
- Ghostscript 9.51 is now blacklisted, due to numerous problems with this version.
|
||||
- Added a workaround for a problem with "txtwrite" in Ghostscript 9.52.
|
||||
- Fixed an issue where the incorrect number of threads used was shown when
|
||||
``OMP_THREAD_LIMIT`` was manipulated.
|
||||
- Removed a possible performance bottlenecks for files that use hundreds to
|
||||
thousands of images on the same page.
|
||||
- Documentation improvements.
|
||||
- Optimization will now be applied to some monochrome images that have a color
|
||||
profile defined instead of only black and white.
|
||||
- ICC profiles are consulted when determining the simplified colorspace of an
|
||||
image.
|
||||
|
||||
v9.6.1
|
||||
======
|
||||
|
||||
- Documentation improvements - thanks to many users for their contributions!
|
||||
|
||||
- Fixed installation instructions for ArchLinux (@pigmonkey)
|
||||
- Updated installation instructions for FreeBSD and other OSes (@knobix)
|
||||
- Added instructions for using Docker Compose with watchdog (@ianalexander,
|
||||
@deisi)
|
||||
- Other miscellany (@mb720, @toy, @caiofacchinato)
|
||||
- Some scripts provided in the documentation have been migrated out so that
|
||||
they can be copied out as whole files, and to ensure syntax checking
|
||||
is maintained.
|
||||
|
||||
- Fixed an error that caused bash completions to fail on macOS. (#502, #504;
|
||||
@AlexanderWillner)
|
||||
- Fixed a rare case where OCRmyPDF threw an exception while processing a PDF
|
||||
with the wrong object type in its ``/Trailer /Info``. The error is now logged
|
||||
and incorrect object is ignored. (#497)
|
||||
- Removed potentially non-free file ``enron1.pdf`` and simplified the test that
|
||||
used it.
|
||||
- Removed potentially non-free file ``misc/media/logo.afdesign``.
|
||||
|
||||
v9.6.0
|
||||
======
|
||||
|
||||
- Fixed a regression with transferring metadata from the input PDF to the output
|
||||
PDF in certain situations.
|
||||
- pdfminer.six is now supported up to version 2020-01-24.
|
||||
- Messages are explaining page rotation decisions are now shown at the standard
|
||||
verbosity level again when ``--rotate-pages``. In some previous version they
|
||||
were set to debug level messages that only appeared with the parameter ``-v1``.
|
||||
- Improvements to ``misc/watcher.py``. Thanks to @ianalexander and @svenihoney.
|
||||
- Documentation improvements.
|
||||
|
||||
v9.5.0
|
||||
======
|
||||
|
||||
- Added API functions to measure OCR quality.
|
||||
- Modest improvements to handling PDFs with difficult/non compliant metadata.
|
||||
|
||||
v9.4.0
|
||||
======
|
||||
|
||||
- Updated recommended dependency versions.
|
||||
- Improvements to test coverage and changes to facilitate better measurement of
|
||||
test coverage, such as when tests run in subprocesses.
|
||||
- Improvements to error messages when Leptonica is not installed correctly.
|
||||
- Fixed use of pytest "session scope" that may have caused some intermittent
|
||||
CI failures.
|
||||
- When the argument ``--keep-temporary-files`` or verbosity is set to ``-v1``,
|
||||
a debug log file is generated in the working temporary folder.
|
||||
|
||||
v9.3.0
|
||||
======
|
||||
|
||||
- Improved native Windows support: we now check in the obvious places in
|
||||
the "Program Files" folders installations of Tesseract and Ghostscript,
|
||||
rather than relying on the user to edit ``PATH`` to specify their location.
|
||||
The ``PATH`` environment variable can still be used to differentiate when
|
||||
multiple installations are present or the programs are installed to non-
|
||||
standard locations.
|
||||
- Fixed an exception on parsing Ghostscript error messages.
|
||||
- Added an improved example demonstrating how to set up a watched folder
|
||||
for automated OCR processing (thanks to @ianalexander for the contribution).
|
||||
|
||||
v9.2.0
|
||||
======
|
||||
|
||||
- Native Windows is now supported.
|
||||
- Continuous integration moved to Azure Pipelines.
|
||||
- Improved test coverage and speed of tests.
|
||||
- Fixed an issue where a page that was originally a JPEG would be saved as a
|
||||
PNG, increasing file size. This occurred only when a preprocessing option
|
||||
was selected along with ``--output-type=pdf`` and all images on the original
|
||||
page were JPEGs. Regression since v7.0.0.
|
||||
- OCRmyPDF no longer depends on the QPDF executable ``qpdf`` or ``libqpdf``.
|
||||
It uses pikepdf (which in turn depends on ``libqpdf``). Package maintainers
|
||||
should adjust dependencies so that OCRmyPDF no longer calls for libqpdf on
|
||||
its own. For users of Python binary wheels, this change means a separate
|
||||
installation of QPDF is no longer necessary. This change is mainly to
|
||||
simplify installation on Windows.
|
||||
- Fixed a rare case where log messages from Tesseract would be discarded.
|
||||
- Fixed incorrect function signature for pixFindPageForeground, causing
|
||||
exceptions on certain platforms/Leptonica versions.
|
||||
|
||||
v9.1.1
|
||||
======
|
||||
|
||||
- Expand the range of pdfminer.six versions that are supported.
|
||||
- Fixed Docker build when using pikepdf 1.7.0.
|
||||
- Fixed documentation to recommend using pip from get-pip.py.
|
||||
|
||||
v9.1.0
|
||||
======
|
||||
|
||||
- Improved diagnostics when file size increases at output. Now warns if JBIG2
|
||||
or pngquant were not available.
|
||||
- pikepdf 1.7.0 is now required, to pick up changes that remove the need for
|
||||
a source install on Linux systems running Python 3.8.
|
||||
|
||||
v9.0.5
|
||||
======
|
||||
|
||||
- The Alpine Docker image (jbarlow83/ocrmypdf-alpine) has been dropped due to
|
||||
the difficulties of supporting Alpine Linux.
|
||||
- The primary Docker image (jbarlow83/ocrmypdf) has been improved to take on
|
||||
the extra features that used to be exclusive to the Alpine image.
|
||||
- No changes to application code.
|
||||
- pdfminer.six version 20191020 is now supported.
|
||||
|
||||
v9.0.4
|
||||
======
|
||||
|
||||
- Fixed compatibility with Python 3.8 (but requires source install for the moment).
|
||||
- Fixed Tesseract settings for ``--user-words`` and ``--user-patterns``.
|
||||
- Changed to pikepdf 1.6.5 (for Python 3.8).
|
||||
- Changed to Pillow 6.2.0 (to mitigate a security vulnerability in earlier Pillow).
|
||||
- A debug message now mentions when English is automatically selected if the locale
|
||||
is not English.
|
||||
|
||||
v9.0.3
|
||||
======
|
||||
|
||||
|
||||
@@ -0,0 +1,50 @@
|
||||
#!/usr/bin/env python3
|
||||
# Original version by DeliciousPickle@github; modified
|
||||
|
||||
# This script must be edited to meet your needs.
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
# pylint: disable=logging-not-lazy
|
||||
|
||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
||||
print(script_dir + '/batch.py: Start')
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
start_dir = sys.argv[1]
|
||||
else:
|
||||
start_dir = '.'
|
||||
|
||||
if len(sys.argv) > 2:
|
||||
log_file = sys.argv[2]
|
||||
else:
|
||||
log_file = script_dir + '/ocr-tree.log'
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s %(message)s',
|
||||
filename=log_file,
|
||||
filemode='w',
|
||||
)
|
||||
|
||||
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
||||
|
||||
for dir_name, subdirs, file_list in os.walk(start_dir):
|
||||
logging.info(dir_name + '\n')
|
||||
os.chdir(dir_name)
|
||||
for filename in file_list:
|
||||
file_ext = os.path.splitext(filename)[1]
|
||||
if file_ext == '.pdf':
|
||||
full_path = dir_name + '/' + filename
|
||||
print(full_path)
|
||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||
if result == ocrmypdf.ExitCode.already_done_ocr:
|
||||
print("Skipped document because it already contained text")
|
||||
elif result == ocrmypdf.ExitCode.ok:
|
||||
print("OCR complete")
|
||||
logging.info(result)
|
||||
@@ -5,7 +5,33 @@ set -o errexit
|
||||
_ocrmypdf()
|
||||
{
|
||||
local cur prev cword words split
|
||||
_init_completion -s || return
|
||||
|
||||
# Homebrew on Macs have version 1.3 of bash-completion which doesn't include - see #502
|
||||
if declare -F _init_completions >/dev/null 2>&1; then
|
||||
_init_completion -s || return
|
||||
else
|
||||
COMPREPLY=()
|
||||
_get_comp_words_by_ref cur prev words cword
|
||||
fi
|
||||
|
||||
if [[ $cur == -* ]]; then
|
||||
COMPREPLY=( $( compgen -W '--language --image-dpi --output-type
|
||||
--sidecar --version --jobs --quiet --verbose --title --author
|
||||
--subject --keywords --rotate-pages --remove-background --deskew
|
||||
--clean --clean-final --unpaper-args --oversample --remove-vectors
|
||||
--threshold --force-ocr --skip-text --redo-ocr
|
||||
--skip-big --jpeg-quality --png-quality --jbig2-lossy
|
||||
--max-image-mpixels --tesseract-config --tesseract-pagesegmode
|
||||
--help --tesseract-oem --pdf-renderer --tesseract-timeout
|
||||
--rotate-pages-threshold --pdfa-image-compression --user-words
|
||||
--user-patterns --keep-temporary-files --output-type
|
||||
--no-progress-bar --pages --fast-web-view' \
|
||||
-- "$cur" ) )
|
||||
return
|
||||
else
|
||||
_filedir
|
||||
return
|
||||
fi
|
||||
|
||||
case $prev in
|
||||
--version|-h|--help)
|
||||
@@ -58,31 +84,13 @@ _ocrmypdf()
|
||||
COMPREPLY=( $( compgen -W '{1..13}' -- "$cur" ) )
|
||||
return
|
||||
;;
|
||||
--sidecar|--title|--author|--subject|--keywords|--unpaper-args|--pages)
|
||||
--sidecar|--title|--author|--subject|--keywords|--unpaper-args|--pages|--fast-web-view)
|
||||
# argument required but no completions available
|
||||
return
|
||||
;;
|
||||
esac
|
||||
|
||||
$split && return
|
||||
|
||||
if [[ $cur == -* ]]; then
|
||||
COMPREPLY=( $( compgen -W '--language --image-dpi --output-type
|
||||
--sidecar --version --jobs --quiet --verbose --title --author
|
||||
--subject --keywords --rotate-pages --remove-background --deskew
|
||||
--clean --clean-final --unpaper-args --oversample --remove-vectors
|
||||
--threshold --force-ocr --skip-text --redo-ocr
|
||||
--skip-big --jpeg-quality --png-quality --jbig2-lossy
|
||||
--max-image-mpixels --tesseract-config --tesseract-pagesegmode
|
||||
--help --tesseract-oem --pdf-renderer --tesseract-timeout
|
||||
--rotate-pages-threshold --pdfa-image-compression --user-words
|
||||
--user-patterns --keep-temporary-files --output-type' \
|
||||
-- "$cur" ) )
|
||||
return
|
||||
else
|
||||
_filedir
|
||||
return
|
||||
fi
|
||||
} &&
|
||||
complete -F _ocrmypdf ocrmypdf
|
||||
|
||||
|
||||
@@ -59,6 +59,8 @@ function __fish_ocrmypdf_verbose
|
||||
end
|
||||
complete -c ocrmypdf -x -s v -l verbose -a '(__fish_ocrmypdf_verbose)' -d "set verbosity level"
|
||||
|
||||
complete -c ocrmypdf -x -l no-progress-bar -d "disable the progress bar"
|
||||
|
||||
function __fish_ocrmypdf_pdfa_compression
|
||||
echo -e "auto\t"(_ "let Ghostscript decide how to compress images")
|
||||
echo -e "jpeg\t"(_ "convert color and grayscale images to JPEG")
|
||||
@@ -111,5 +113,6 @@ complete -c ocrmypdf -x -l rotate-pages-threshold -d "page rotation confidence"
|
||||
|
||||
complete -c ocrmypdf -r -l user-words -d "specify location of user words file"
|
||||
complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file"
|
||||
complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF"
|
||||
|
||||
complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf)"
|
||||
|
||||
@@ -0,0 +1,15 @@
|
||||
---
|
||||
version: "3.3"
|
||||
services:
|
||||
ocrmypdf:
|
||||
restart: always
|
||||
container_name: ocrmypdf
|
||||
image: jbarlow83/ocrmypdf
|
||||
volumes:
|
||||
- "/media/scan:/input"
|
||||
- "/mnt/scan:/output"
|
||||
environment:
|
||||
- OCR_OUTPUT_DIRECTORY_YEAR_MONT=0
|
||||
user: "<SET TO YOUR USER ID>:<SET TO YOUR GROUP ID>"
|
||||
entrypoint: python3
|
||||
command: watcher.py
|
||||
Binary file not shown.
@@ -0,0 +1,72 @@
|
||||
#!/bin/env python3
|
||||
# Contributed by github.com/Enantiomerie
|
||||
|
||||
# This script must be edited to meet your needs.
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
# pylint: disable=logging-not-lazy
|
||||
|
||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
||||
timestamp = time.strftime("%Y-%m-%d-%H%M_")
|
||||
log_file = script_dir + '/' + timestamp + 'ocrmypdf.log'
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s %(message)s',
|
||||
filename=log_file,
|
||||
filemode='w',
|
||||
)
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
start_dir = sys.argv[1]
|
||||
else:
|
||||
start_dir = '.'
|
||||
|
||||
for dir_name, subdirs, file_list in os.walk(start_dir):
|
||||
logging.info(dir_name)
|
||||
os.chdir(dir_name)
|
||||
for filename in file_list:
|
||||
file_stem, file_ext = os.path.splitext(filename)
|
||||
if file_ext != '.pdf':
|
||||
continue
|
||||
full_path = os.path.join(dir_name, filename)
|
||||
timestamp_ocr = time.strftime("%Y-%m-%d-%H%M_OCR_")
|
||||
filename_ocr = timestamp_ocr + file_stem + '.pdf'
|
||||
# create string for pdf processing
|
||||
# the script is processed as root user via chron
|
||||
cmd = [
|
||||
'docker',
|
||||
'run',
|
||||
'--rm',
|
||||
'-i',
|
||||
'jbarlow83/ocrmypdf',
|
||||
'--deskew',
|
||||
'-',
|
||||
'-',
|
||||
]
|
||||
logging.info(cmd)
|
||||
full_path_ocr = os.path.join(dir_name, filename_ocr)
|
||||
with open(filename, 'rb') as input_file, open(
|
||||
full_path_ocr, 'wb'
|
||||
) as output_file:
|
||||
proc = subprocess.run(
|
||||
cmd,
|
||||
stdin=input_file,
|
||||
stdout=output_file,
|
||||
stderr=subprocess.PIPE,
|
||||
check=False,
|
||||
)
|
||||
logging.info(proc.stderr.read())
|
||||
os.chmod(full_path_ocr, 0o664)
|
||||
os.chmod(full_path, 0o664)
|
||||
full_path_ocr_archive = sys.argv[2]
|
||||
full_path_archive = sys.argv[2] + '/no_ocr'
|
||||
shutil.move(full_path_ocr, full_path_ocr_archive)
|
||||
shutil.move(full_path, full_path_archive)
|
||||
logging.info('Finished.\n')
|
||||
+149
@@ -0,0 +1,149 @@
|
||||
# Copyright (C) 2019 Ian Alexander: https://github.com/ianalexander
|
||||
# Copyright (C) 2020 James R Barlow: https://github.com/jbarlow83
|
||||
#
|
||||
# This program is free software: you can redistribute it and/or modify
|
||||
# it under the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation, either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# This program is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
from watchdog.events import PatternMatchingEventHandler
|
||||
from watchdog.observers import Observer
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
|
||||
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
|
||||
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
|
||||
OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False))
|
||||
ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', False))
|
||||
DESKEW = bool(os.getenv('OCR_DESKEW', False))
|
||||
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
||||
POLL_NEW_FILE_SECONDS = os.getenv('OCR_POLL_NEW_FILE_SECONDS', 1)
|
||||
LOGLEVEL = os.environ.get('OCR_LOGLEVEL', 'INFO').upper()
|
||||
PATTERNS = ['*.pdf']
|
||||
|
||||
log = logging.getLogger('ocrmypdf-watcher')
|
||||
|
||||
|
||||
def get_output_dir(root, basename):
|
||||
if OUTPUT_DIRECTORY_YEAR_MONTH:
|
||||
today = datetime.today()
|
||||
output_directory_year_month = (
|
||||
Path(root) / str(today.year) / f'{today.month:02d}'
|
||||
)
|
||||
if not output_directory_year_month.exists():
|
||||
output_directory_year_month.mkdir(parents=True, exist_ok=True)
|
||||
output_path = Path(output_directory_year_month) / basename
|
||||
else:
|
||||
output_path = Path(OUTPUT_DIRECTORY) / basename
|
||||
return output_path
|
||||
|
||||
|
||||
def wait_for_file_ready(file_path):
|
||||
# This loop waits to make sure that the file is completely loaded on
|
||||
# disk before attempting to read. Docker sometimes will publish the
|
||||
# watchdog event before the file is actually fully on disk, causing
|
||||
# pikepdf to fail.
|
||||
|
||||
retries = 5
|
||||
while retries:
|
||||
try:
|
||||
pdf = pikepdf.open(file_path)
|
||||
except (FileNotFoundError, pikepdf.PdfError) as e:
|
||||
log.info(f"File {file_path} is not ready yet")
|
||||
log.debug("Exception was", exc_info=e)
|
||||
time.sleep(POLL_NEW_FILE_SECONDS)
|
||||
retries -= 1
|
||||
else:
|
||||
pdf.close()
|
||||
return True
|
||||
|
||||
return False
|
||||
|
||||
|
||||
def execute_ocrmypdf(file_path):
|
||||
file_path = Path(file_path)
|
||||
output_path = get_output_dir(OUTPUT_DIRECTORY, file_path.name)
|
||||
|
||||
log.info("-" * 20)
|
||||
log.info(f'New file: {file_path}. Waiting until fully loaded...')
|
||||
if not wait_for_file_ready(file_path):
|
||||
log.info(f"Gave up waiting for {file_path} to become ready")
|
||||
return
|
||||
log.info(f'Attempting to OCRmyPDF to: {output_path}')
|
||||
exit_code = ocrmypdf.ocr(
|
||||
input_file=file_path,
|
||||
output_file=output_path,
|
||||
deskew=DESKEW,
|
||||
**OCR_JSON_SETTINGS,
|
||||
)
|
||||
if exit_code == 0 and ON_SUCCESS_DELETE:
|
||||
log.info(f'OCR is done. Deleting: {file_path}')
|
||||
file_path.unlink()
|
||||
else:
|
||||
log.info('OCR is done')
|
||||
|
||||
|
||||
class HandleObserverEvent(PatternMatchingEventHandler):
|
||||
def on_any_event(self, event):
|
||||
if event.event_type in ['created']:
|
||||
execute_ocrmypdf(event.src_path)
|
||||
|
||||
|
||||
def main():
|
||||
ocrmypdf.configure_logging(
|
||||
verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True
|
||||
)
|
||||
log.info(
|
||||
f"Starting OCRmyPDF watcher with config:\n"
|
||||
f"Input Directory: {INPUT_DIRECTORY}\n"
|
||||
f"Output Directory: {OUTPUT_DIRECTORY}\n"
|
||||
f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}"
|
||||
)
|
||||
log.debug(
|
||||
f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n"
|
||||
f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n"
|
||||
f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n"
|
||||
f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n"
|
||||
f"DESKEW: {DESKEW}\n"
|
||||
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
||||
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
||||
f"LOGLEVEL: {LOGLEVEL}\n"
|
||||
)
|
||||
|
||||
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS:
|
||||
log.error('OCR_JSON_SETTINGS should not specify input file or output file')
|
||||
sys.exit(1)
|
||||
|
||||
handler = HandleObserverEvent(patterns=PATTERNS)
|
||||
observer = Observer()
|
||||
observer.schedule(handler, INPUT_DIRECTORY, recursive=True)
|
||||
observer.start()
|
||||
try:
|
||||
while True:
|
||||
time.sleep(1)
|
||||
except KeyboardInterrupt:
|
||||
observer.stop()
|
||||
observer.join()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+3
-1
@@ -10,7 +10,8 @@ build-backend = "setuptools.build_meta"
|
||||
|
||||
[tool.black]
|
||||
line-length = 88
|
||||
py36 = true
|
||||
target-version = ["py36",
|
||||
"py37", "py38"]
|
||||
skip-string-normalization = true
|
||||
include = '\.pyi?$'
|
||||
exclude = '''
|
||||
@@ -28,5 +29,6 @@ exclude = '''
|
||||
| docs
|
||||
| misc
|
||||
| \.egg-info
|
||||
| src/ocrmypdf/lib/_leptonica.py
|
||||
)/
|
||||
'''
|
||||
|
||||
@@ -1,4 +1,2 @@
|
||||
check-manifest >= 0.35
|
||||
twine >= 1.8.1
|
||||
coverage >= 4.5
|
||||
GitPython == 2.1.3
|
||||
|
||||
@@ -1,13 +1,10 @@
|
||||
# requirements.txt can be used to replicate the developer's build environment
|
||||
# setup.py lists a separate set of requirements that are looser to simplify
|
||||
# installation
|
||||
chardet == 3.0.4
|
||||
cffi == 1.12.2
|
||||
cffi == 1.14.0
|
||||
img2pdf == 0.3.3
|
||||
pdfminer.six == 20181108
|
||||
pikepdf == 1.6.1
|
||||
Pillow >= 5.0.0, != 5.1.0 ; sys_platform == "darwin"
|
||||
pycparser == 2.19
|
||||
python-xmp-toolkit == 2.0.1
|
||||
reportlab == 3.5.13
|
||||
tqdm == 4.32.1
|
||||
pdfminer.six == 20200402
|
||||
pikepdf == 1.10.2
|
||||
Pillow == 7.0.0
|
||||
reportlab == 3.5.34
|
||||
tqdm == 4.42.1
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
pytest >= 5.0.0
|
||||
pytest-helpers-namespace >= 2019.1.8
|
||||
pytest-xdist >= 1.29.0 # For DumpError fix
|
||||
pytest-cov >= 2.6.1
|
||||
python-xmp-toolkit # requires apt-get install libexempi3
|
||||
pytest-xdist >= 1.31.0
|
||||
pytest-cov >= 2.8.0
|
||||
python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3
|
||||
# or brew install exempi
|
||||
PyPDF2 >= 1.26.0
|
||||
#PyMuPDF == 1.13.4 # optional
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
watchdog >= 0.8.2, < 1.0
|
||||
@@ -0,0 +1 @@
|
||||
Flask >= 1, < 2
|
||||
@@ -22,6 +22,8 @@ include_trailing_comma=True
|
||||
force_grid_wrap=0
|
||||
use_parentheses=True
|
||||
line_length=88
|
||||
known_first_party = ocrmypdf
|
||||
known_third_party = PIL,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug
|
||||
|
||||
[metadata]
|
||||
license_file = LICENSE
|
||||
|
||||
@@ -21,11 +21,12 @@ from __future__ import print_function, unicode_literals
|
||||
|
||||
import sys
|
||||
|
||||
from setuptools import find_packages, setup
|
||||
|
||||
if sys.version_info < (3, 6):
|
||||
print("Python 3.6 or newer is required", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
from setuptools import setup, find_packages
|
||||
|
||||
# pylint: disable=w0613
|
||||
|
||||
@@ -68,6 +69,7 @@ setup(
|
||||
classifiers=[
|
||||
"Programming Language :: Python :: 3.6",
|
||||
"Programming Language :: Python :: 3.7",
|
||||
"Programming Language :: Python :: 3.8",
|
||||
"Development Status :: 5 - Production/Stable",
|
||||
"Environment :: Console",
|
||||
"Intended Audience :: End Users/Desktop",
|
||||
@@ -75,6 +77,7 @@ setup(
|
||||
"Intended Audience :: System Administrators",
|
||||
"License :: OSI Approved :: GNU General Public License v3 (GPLv3)",
|
||||
"Operating System :: MacOS :: MacOS X",
|
||||
"Operating System :: Microsoft :: Windows :: Windows 10",
|
||||
"Operating System :: POSIX",
|
||||
"Operating System :: POSIX :: BSD",
|
||||
"Operating System :: POSIX :: Linux",
|
||||
@@ -95,11 +98,9 @@ setup(
|
||||
'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108
|
||||
'cffi >= 1.9.1', # must be a setup and install requirement
|
||||
'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely
|
||||
'pdfminer.six == 20181108',
|
||||
'pikepdf >= 1.6.0, < 2',
|
||||
'Pillow >= 4.0.0, != 5.1.0 ; sys_platform == "darwin"',
|
||||
# Pillow < 4 has BytesIO/TIFF bug w/img2pdf 0.2.3
|
||||
# block 5.1.0, broken wheels
|
||||
'pdfminer.six >= 20181108, <= 20200402',
|
||||
'pikepdf >= 1.8.1, < 2',
|
||||
'Pillow >= 6.2.0',
|
||||
'reportlab >= 3.3.0', # oldest released version with sane image handling
|
||||
'tqdm >= 4',
|
||||
],
|
||||
|
||||
@@ -23,6 +23,7 @@ from .exceptions import (
|
||||
DpiError,
|
||||
EncryptedPdfError,
|
||||
ExitCode,
|
||||
ExitCodeException,
|
||||
InputFileError,
|
||||
MissingDependencyError,
|
||||
OutputFileAccessError,
|
||||
|
||||
@@ -21,7 +21,7 @@ from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
|
||||
MAX_REPLACE_PAGES = int(os.environ.get('_OCRMYPDF_MAX_REPLACE_PAGES', 100))
|
||||
MAX_REPLACE_PAGES = 100
|
||||
|
||||
|
||||
def _update_page_resources(*, page, font, font_key, procset):
|
||||
|
||||
@@ -100,14 +100,14 @@ def cleanup_working_files(work_folder, options):
|
||||
|
||||
class LogNameAdapter(logging.LoggerAdapter):
|
||||
def process(self, msg, kwargs):
|
||||
# return '[%s] %s' % (self.extra['filename'], msg), kwargs
|
||||
# return '[%s] %s' % (self.extra['input_filename'], msg), kwargs
|
||||
return '%s' % (msg,), kwargs
|
||||
|
||||
|
||||
class LogNamePageAdapter(logging.LoggerAdapter):
|
||||
def process(self, msg, kwargs):
|
||||
return (
|
||||
#'[%s:%05u] %s' % (self.extra['filename'], self.extra['page'], msg),
|
||||
#'[%s:%05u] %s' % (self.extra['input_filename'], self.extra['page'], msg),
|
||||
'%4u: %s' % (self.extra['page'], msg),
|
||||
kwargs,
|
||||
)
|
||||
@@ -116,9 +116,9 @@ class LogNamePageAdapter(logging.LoggerAdapter):
|
||||
def make_logger(options=None, prefix='ocrmypdf', filename=None, page=None):
|
||||
log = logging.getLogger(prefix)
|
||||
if filename and page:
|
||||
adapter = LogNamePageAdapter(log, dict(filename=filename, page=page))
|
||||
adapter = LogNamePageAdapter(log, dict(input_filename=filename, page=page))
|
||||
elif filename:
|
||||
adapter = LogNameAdapter(log, dict(filename=filename))
|
||||
adapter = LogNameAdapter(log, dict(input_filename=filename))
|
||||
else:
|
||||
adapter = log
|
||||
return adapter
|
||||
|
||||
+107
-93
@@ -19,6 +19,7 @@ import os
|
||||
import re
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from shutil import copyfileobj
|
||||
|
||||
import img2pdf
|
||||
@@ -37,18 +38,18 @@ from .exceptions import (
|
||||
UnsupportedImageFormatError,
|
||||
)
|
||||
from .exec import ghostscript, tesseract
|
||||
from .helpers import re_symlink
|
||||
from .helpers import safe_symlink
|
||||
from .hocrtransform import HocrTransform
|
||||
from .optimize import optimize
|
||||
from .pdfa import generate_pdfa_ps
|
||||
from .pdfinfo import Colorspace, PdfInfo
|
||||
from .pdfinfo import Colorspace, Encoding, PdfInfo
|
||||
|
||||
VECTOR_PAGE_DPI = 400
|
||||
|
||||
|
||||
def triage_image_file(input_file, output_file, options, log):
|
||||
log.info("Input file is not a PDF, checking if it is an image...")
|
||||
try:
|
||||
log.info("Input file is not a PDF, checking if it is an image...")
|
||||
im = Image.open(input_file)
|
||||
except EnvironmentError as e:
|
||||
# Recover the original filename
|
||||
@@ -85,9 +86,9 @@ def triage_image_file(input_file, output_file, options, log):
|
||||
|
||||
if 'iccprofile' not in im.info:
|
||||
if im.mode == 'RGB':
|
||||
log.info('Input image has no ICC profile, assuming sRGB')
|
||||
log.info("Input image has no ICC profile, assuming sRGB")
|
||||
elif im.mode == 'CMYK':
|
||||
log.info('Input CMYK image has no ICC profile, not usable')
|
||||
log.error("Input CMYK image has no ICC profile, not usable")
|
||||
raise UnsupportedImageFormatError()
|
||||
|
||||
try:
|
||||
@@ -123,7 +124,7 @@ def _pdf_guess_version(input_file, search_window=1024):
|
||||
return ''
|
||||
|
||||
|
||||
def triage(input_file, output_file, options, log):
|
||||
def triage(original_filename, input_file, output_file, options, log):
|
||||
try:
|
||||
if _pdf_guess_version(input_file):
|
||||
if options.image_dpi:
|
||||
@@ -132,11 +133,12 @@ def triage(input_file, output_file, options, log):
|
||||
"input file is a PDF, not an image."
|
||||
)
|
||||
# Origin file is a pdf create a symlink with pdf extension
|
||||
re_symlink(input_file, output_file)
|
||||
safe_symlink(input_file, output_file)
|
||||
return output_file
|
||||
except EnvironmentError as e:
|
||||
log.error(e)
|
||||
raise InputFileError() from e
|
||||
log.debug(f"Temporary file was at: {input_file}")
|
||||
msg = str(e).replace(input_file, original_filename)
|
||||
raise InputFileError(msg) from e
|
||||
|
||||
triage_image_file(input_file, output_file, options, log)
|
||||
return output_file
|
||||
@@ -181,7 +183,7 @@ def validate_pdfinfo_options(context):
|
||||
)
|
||||
raise InputFileError()
|
||||
else:
|
||||
log.warn(
|
||||
log.warning(
|
||||
"This PDF has a fillable form. "
|
||||
"Chances are it is a pure digital "
|
||||
"document that does not need OCR."
|
||||
@@ -327,6 +329,35 @@ def rasterize_preview(input_file, page_context):
|
||||
return output_file
|
||||
|
||||
|
||||
def describe_rotation(page_context, orient_conf, correction):
|
||||
"""
|
||||
Describe the page rotation we are going to perform.
|
||||
"""
|
||||
direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'}
|
||||
turns = {0: ' ', 90: '⬏', 180: '↻', 270: '⬑'}
|
||||
|
||||
existing_rotation = page_context.pageinfo.rotation
|
||||
action = ''
|
||||
if orient_conf.confidence >= page_context.options.rotate_pages_threshold:
|
||||
if correction != 0:
|
||||
action = 'will rotate ' + turns[correction]
|
||||
else:
|
||||
action = 'rotation appears correct'
|
||||
else:
|
||||
if correction != 0:
|
||||
action = 'confidence too low to rotate'
|
||||
else:
|
||||
action = 'no change'
|
||||
|
||||
facing = ''
|
||||
|
||||
if existing_rotation != 0:
|
||||
facing = f"with existing rotation {direction.get(existing_rotation, '?')}, "
|
||||
facing += f"page is facing {direction.get(orient_conf.angle, '?')}"
|
||||
|
||||
return f"{facing}, confidence {orient_conf.confidence:.2f} - {action}"
|
||||
|
||||
|
||||
def get_orientation_correction(preview, page_context):
|
||||
"""
|
||||
Work out orientation correct for each page.
|
||||
@@ -353,44 +384,14 @@ def get_orientation_correction(preview, page_context):
|
||||
tesseract_env=page_context.options.tesseract_env,
|
||||
)
|
||||
|
||||
direction = {0: '⇧', 90: '⇨', 180: '⇩', 270: '⇦'}
|
||||
|
||||
existing_rotation = page_context.pageinfo.rotation
|
||||
|
||||
correction = orient_conf.angle % 360
|
||||
|
||||
apply_correction = False
|
||||
action = ''
|
||||
if orient_conf.confidence >= page_context.options.rotate_pages_threshold:
|
||||
if correction != 0:
|
||||
apply_correction = True
|
||||
action = ' - will rotate'
|
||||
else:
|
||||
action = ' - rotation appears correct'
|
||||
else:
|
||||
if correction != 0:
|
||||
action = ' - confidence too low to rotate'
|
||||
else:
|
||||
action = ' - no change'
|
||||
|
||||
facing = ''
|
||||
if existing_rotation != 0:
|
||||
facing = 'with existing rotation {}, '.format(
|
||||
direction.get(existing_rotation, '?')
|
||||
)
|
||||
facing += 'page is facing {}'.format(direction.get(orient_conf.angle, '?'))
|
||||
|
||||
page_context.log.debug(
|
||||
'{pagenum:4d}: {facing}, confidence {conf:.2f}{action}'.format(
|
||||
pagenum=page_context.pageinfo.pageno,
|
||||
facing=facing,
|
||||
conf=orient_conf.confidence,
|
||||
action=action,
|
||||
)
|
||||
)
|
||||
|
||||
if apply_correction:
|
||||
page_context.log.info(describe_rotation(page_context, orient_conf, correction))
|
||||
if (
|
||||
orient_conf.confidence >= page_context.options.rotate_pages_threshold
|
||||
and correction != 0
|
||||
):
|
||||
return correction
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
@@ -557,7 +558,7 @@ def ocr_tesseract_hocr(input_file, page_context):
|
||||
|
||||
def should_visible_page_image_use_jpg(pageinfo):
|
||||
# If all images were JPEGs originally, produce a JPEG as output
|
||||
return pageinfo.images and all(im.enc == 'jpeg' for im in pageinfo.images)
|
||||
return pageinfo.images and all(im.enc == Encoding.jpeg for im in pageinfo.images)
|
||||
|
||||
|
||||
def create_visible_page_jpg(image, page_context):
|
||||
@@ -693,15 +694,22 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context):
|
||||
# stamping them out as soon as possible.
|
||||
modified = False
|
||||
with pikepdf.open(input_pdf) as pdf_file:
|
||||
if pdf_file.docinfo:
|
||||
for k, v in pdf_file.docinfo.items():
|
||||
if b'\x00' in bytes(v):
|
||||
pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'')
|
||||
modified = True
|
||||
try:
|
||||
len(pdf_file.docinfo)
|
||||
except TypeError:
|
||||
context.log.error(
|
||||
"File contains a malformed DocumentInfo block - continuing anyway"
|
||||
)
|
||||
else:
|
||||
if pdf_file.docinfo:
|
||||
for k, v in pdf_file.docinfo.items():
|
||||
if b'\x00' in bytes(v):
|
||||
pdf_file.docinfo[k] = bytes(v).replace(b'\x00', b'')
|
||||
modified = True
|
||||
if modified:
|
||||
pdf_file.save(fix_docinfo_file)
|
||||
else:
|
||||
os.symlink(input_pdf, fix_docinfo_file)
|
||||
safe_symlink(input_pdf, fix_docinfo_file)
|
||||
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_version=input_pdfinfo.min_version,
|
||||
@@ -725,47 +733,53 @@ def should_linearize(working_file, context):
|
||||
def metadata_fixup(working_file, context):
|
||||
output_file = context.get_path('metafix.pdf')
|
||||
options = context.options
|
||||
original = pikepdf.open(context.origin)
|
||||
docinfo = get_docinfo(original, options)
|
||||
pdf = pikepdf.open(working_file)
|
||||
with pdf.open_metadata() as meta:
|
||||
meta.load_from_docinfo(docinfo, delete_missing=False)
|
||||
# If xmp:CreateDate is missing, set it to the modify date to
|
||||
# match Ghostscript, for consistency
|
||||
if 'xmp:CreateDate' not in meta:
|
||||
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
||||
|
||||
meta_original = original.open_metadata()
|
||||
not_copied = set(meta_original.keys()) - set(meta.keys())
|
||||
if not_copied:
|
||||
if options.output_type.startswith('pdfa'):
|
||||
context.log.warning(
|
||||
"Some input metadata could not be copied because it is not "
|
||||
"permitted in PDF/A. You may wish to examine the output "
|
||||
"PDF's XMP metadata."
|
||||
)
|
||||
context.log.debug(
|
||||
"The following metadata fields were not copied: %r", not_copied
|
||||
)
|
||||
else:
|
||||
context.log.error(
|
||||
"Some input metadata could not be copied."
|
||||
"You may wish to examine the output PDF's XMP metadata."
|
||||
)
|
||||
context.log.info(
|
||||
"The following metadata fields were not copied: %r", not_copied
|
||||
)
|
||||
pdf.save(
|
||||
output_file,
|
||||
compress_streams=True,
|
||||
preserve_pdfa=True,
|
||||
object_stream_mode=pikepdf.ObjectStreamMode.generate,
|
||||
linearize=( # Don't linearize if optimize() will be linearizing too
|
||||
should_linearize(working_file, context) if options.optimize == 0 else False
|
||||
),
|
||||
)
|
||||
original.close()
|
||||
pdf.close()
|
||||
def report_on_metadata(missing):
|
||||
if not missing:
|
||||
return
|
||||
if options.output_type.startswith('pdfa'):
|
||||
context.log.warning(
|
||||
"Some input metadata could not be copied because it is not "
|
||||
"permitted in PDF/A. You may wish to examine the output "
|
||||
"PDF's XMP metadata."
|
||||
)
|
||||
context.log.debug(
|
||||
"The following metadata fields were not copied: %r", missing
|
||||
)
|
||||
else:
|
||||
context.log.error(
|
||||
"Some input metadata could not be copied."
|
||||
"You may wish to examine the output PDF's XMP metadata."
|
||||
)
|
||||
context.log.info(
|
||||
"The following metadata fields were not copied: %r", missing
|
||||
)
|
||||
|
||||
with pikepdf.open(context.origin) as original, pikepdf.open(working_file) as pdf:
|
||||
docinfo = get_docinfo(original, options)
|
||||
with pdf.open_metadata() as meta:
|
||||
meta.load_from_docinfo(docinfo, delete_missing=False, raise_failure=False)
|
||||
# If xmp:CreateDate is missing, set it to the modify date to
|
||||
# match Ghostscript, for consistency
|
||||
if 'xmp:CreateDate' not in meta:
|
||||
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
||||
|
||||
meta_original = original.open_metadata()
|
||||
missing = set(meta_original.keys()) - set(meta.keys())
|
||||
report_on_metadata(missing)
|
||||
|
||||
pdf.save(
|
||||
output_file,
|
||||
compress_streams=True,
|
||||
preserve_pdfa=True,
|
||||
object_stream_mode=pikepdf.ObjectStreamMode.generate,
|
||||
linearize=( # Don't linearize if optimize() will be linearizing too
|
||||
should_linearize(working_file, context)
|
||||
if options.optimize == 0
|
||||
else False
|
||||
),
|
||||
)
|
||||
|
||||
return output_file
|
||||
|
||||
|
||||
|
||||
+96
-32
@@ -23,8 +23,10 @@ import signal
|
||||
import sys
|
||||
import threading
|
||||
from collections import namedtuple
|
||||
from pathlib import Path
|
||||
from tempfile import mkdtemp
|
||||
|
||||
import PIL
|
||||
from tqdm import tqdm
|
||||
|
||||
from ._graft import OcrGrafter
|
||||
@@ -176,7 +178,7 @@ def post_process(pdf_file, context):
|
||||
return optimize_pdf(pdf_out, context)
|
||||
|
||||
|
||||
def worker_init(queue):
|
||||
def worker_init(queue, max_pixels):
|
||||
"""Initialize a process pool worker"""
|
||||
|
||||
# Ignore SIGINT (our parent process will kill us gracefully)
|
||||
@@ -188,9 +190,15 @@ def worker_init(queue):
|
||||
root.handlers = []
|
||||
root.addHandler(h)
|
||||
|
||||
# In Windows, child process will not inherit our change to this value in
|
||||
# the parent process, so ensure workers get it set
|
||||
PIL.Image.MAX_IMAGE_PIXELS = max_pixels
|
||||
|
||||
def worker_thread_init(queue):
|
||||
pass
|
||||
|
||||
def worker_thread_init(_queue, max_pixels):
|
||||
# This is probably not needed since threads should all see the same memory,
|
||||
# but done for consistency.
|
||||
PIL.Image.MAX_IMAGE_PIXELS = max_pixels
|
||||
|
||||
|
||||
def log_listener(queue):
|
||||
@@ -223,20 +231,24 @@ def exec_concurrent(context):
|
||||
# Run exec_page_sync on every page context
|
||||
max_workers = min(len(context.pdfinfo), context.options.jobs)
|
||||
if max_workers > 1:
|
||||
context.log.info("Start processing %d pages concurrent", max_workers)
|
||||
context.log.info("Start processing %d pages concurrently", max_workers)
|
||||
|
||||
# Tesseract 4.0 is multithreaded, and we also run multiple workers. We want to
|
||||
# avoid the situation where we end up trying to run NxN jobs on N CPU cores,
|
||||
# as that gives poor performance. Performance testing shows we're better off
|
||||
# Tesseract 4.x can be multithreaded, and we also run multiple workers. We want
|
||||
# to manage how many threads it uses to avoid creating total threads than cores.
|
||||
# Performance testing shows we're better off
|
||||
# parallelizing ocrmypdf and forcing Tesseract to be single threaded, which we
|
||||
# get by setting the envvar OMP_THREAD_LIMIT to 1. But if the page count of the
|
||||
# input file is small, then we allow Tesseract to use threads, subject to the
|
||||
# constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers and limiting
|
||||
# Tesseract to 4 threads.
|
||||
tess_threads = min(4, context.options.jobs // max_workers)
|
||||
# constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers.
|
||||
# As of Tesseract 4.1, 3 threads is the most effective on a 4 core/8 thread system.
|
||||
tess_threads = min(3, context.options.jobs // max_workers)
|
||||
if context.options.tesseract_env is None:
|
||||
context.options.tesseract_env = os.environ.copy()
|
||||
context.options.tesseract_env.setdefault('OMP_THREAD_LIMIT', str(tess_threads))
|
||||
try:
|
||||
tess_threads = int(context.options.tesseract_env['OMP_THREAD_LIMIT'])
|
||||
except ValueError: # OMP_THREAD_LIMIT initialized to non-numeric
|
||||
context.log.error("Environment variable OMP_THREAD_LIMIT is not numeric")
|
||||
if tess_threads > 1:
|
||||
context.log.info("Using Tesseract OpenMP thread limit %d", tess_threads)
|
||||
|
||||
@@ -260,26 +272,43 @@ def exec_concurrent(context):
|
||||
unit='page',
|
||||
unit_scale=0.5,
|
||||
disable=not context.options.progress_bar,
|
||||
) as pbar, Pool(
|
||||
processes=max_workers, initializer=initializer, initargs=(log_queue,)
|
||||
) as pool:
|
||||
results = pool.imap_unordered(exec_page_sync, context.get_page_contexts())
|
||||
while True:
|
||||
try:
|
||||
page_result = results.next()
|
||||
sidecars[page_result.pageno] = page_result.text
|
||||
pbar.update()
|
||||
ocrgraft.graft_page(page_result)
|
||||
pbar.update()
|
||||
except StopIteration:
|
||||
break
|
||||
except (Exception, KeyboardInterrupt):
|
||||
) as pbar:
|
||||
pool = Pool(
|
||||
processes=max_workers,
|
||||
initializer=initializer,
|
||||
initargs=(log_queue, PIL.Image.MAX_IMAGE_PIXELS),
|
||||
)
|
||||
try:
|
||||
results = pool.imap_unordered(exec_page_sync, context.get_page_contexts())
|
||||
while True:
|
||||
try:
|
||||
page_result = results.next()
|
||||
sidecars[page_result.pageno] = page_result.text
|
||||
pbar.update()
|
||||
ocrgraft.graft_page(page_result)
|
||||
pbar.update()
|
||||
except StopIteration:
|
||||
break
|
||||
except KeyboardInterrupt:
|
||||
# Terminate pool so we exit instantly
|
||||
pool.terminate()
|
||||
# Don't try listener.join() here, will deadlock
|
||||
raise
|
||||
except Exception:
|
||||
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
||||
# Unless inside pytest, exit immediately because no one wants
|
||||
# to wait for child processes to finalize results that will be
|
||||
# thrown away. Inside pytest, we want child processes to exit
|
||||
# cleanly so that they output an error messages or coverage data
|
||||
# we need from them.
|
||||
pool.terminate()
|
||||
log_queue.put_nowait(None) # Terminate log listener
|
||||
# Don't try listener.join() here, will deadlock
|
||||
raise
|
||||
raise
|
||||
finally:
|
||||
# Terminate log listener
|
||||
log_queue.put_nowait(None)
|
||||
pool.close()
|
||||
pool.join()
|
||||
|
||||
log_queue.put_nowait(None)
|
||||
listener.join()
|
||||
|
||||
# Output sidecar text
|
||||
@@ -301,7 +330,25 @@ def exec_concurrent(context):
|
||||
class NeverRaise(Exception):
|
||||
"""An exception that is never raised"""
|
||||
|
||||
pass
|
||||
pass # pylint: disable=unnecessary-pass
|
||||
|
||||
|
||||
def samefile(f1, f2):
|
||||
if os.name == 'nt':
|
||||
return f1 == f2
|
||||
else:
|
||||
return os.path.samefile(f1, f2)
|
||||
|
||||
|
||||
def configure_debug_logging(log_filename, prefix=''):
|
||||
log_file_handler = logging.FileHandler(log_filename, delay=True)
|
||||
log_file_handler.setLevel(logging.DEBUG)
|
||||
formatter = logging.Formatter(
|
||||
'[%(asctime)s] - %(name)s - %(levelname)7s - %(message)s'
|
||||
)
|
||||
log_file_handler.setFormatter(formatter)
|
||||
logging.getLogger(prefix).addHandler(log_file_handler)
|
||||
return log_file_handler
|
||||
|
||||
|
||||
def run_pipeline(options, api=False):
|
||||
@@ -314,13 +361,23 @@ def run_pipeline(options, api=False):
|
||||
options.jobs = available_cpu_count()
|
||||
|
||||
work_folder = mkdtemp(prefix="com.github.ocrmypdf.")
|
||||
debug_log_handler = None
|
||||
if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get(
|
||||
'PYTEST_CURRENT_TEST', ''
|
||||
):
|
||||
debug_log_handler = configure_debug_logging(Path(work_folder) / "debug.log")
|
||||
|
||||
try:
|
||||
check_requested_output_file(options)
|
||||
start_input_file = create_input_file(options, work_folder)
|
||||
start_input_file, original_filename = create_input_file(options, work_folder)
|
||||
|
||||
# Triage image or pdf
|
||||
origin_pdf = triage(
|
||||
start_input_file, os.path.join(work_folder, 'origin.pdf'), options, log
|
||||
original_filename,
|
||||
start_input_file,
|
||||
os.path.join(work_folder, 'origin.pdf'),
|
||||
options,
|
||||
log,
|
||||
)
|
||||
|
||||
# Gather pdfinfo and create context
|
||||
@@ -329,6 +386,7 @@ def run_pipeline(options, api=False):
|
||||
detailed_page_analysis=options.redo_ocr,
|
||||
progbar=options.progress_bar,
|
||||
)
|
||||
|
||||
context = PDFContext(options, work_folder, origin_pdf, pdfinfo)
|
||||
|
||||
# Validate options are okay for this pdf
|
||||
@@ -339,7 +397,7 @@ def run_pipeline(options, api=False):
|
||||
|
||||
if options.output_file == '-':
|
||||
log.info("Output sent to stdout")
|
||||
elif os.path.samefile(options.output_file, os.devnull):
|
||||
elif samefile(options.output_file, os.devnull):
|
||||
pass # Say nothing when sending to dev null
|
||||
else:
|
||||
if options.output_type.startswith('pdfa'):
|
||||
@@ -375,6 +433,12 @@ def run_pipeline(options, api=False):
|
||||
log.exception("An exception occurred while executing the pipeline")
|
||||
return ExitCode.other_error
|
||||
finally:
|
||||
if debug_log_handler:
|
||||
try:
|
||||
debug_log_handler.close()
|
||||
log.removeHandler(debug_log_handler)
|
||||
except EnvironmentError as e:
|
||||
print(e, file=sys.stderr)
|
||||
cleanup_working_files(work_folder, options)
|
||||
|
||||
return ExitCode.ok
|
||||
|
||||
+41
-11
@@ -17,6 +17,7 @@
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
import locale
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
@@ -41,12 +42,13 @@ from .exec import (
|
||||
tesseract,
|
||||
unpaper,
|
||||
)
|
||||
from .helpers import is_file_writable, is_iterable_notstr, monotonic, re_symlink
|
||||
from .helpers import is_file_writable, is_iterable_notstr, monotonic, safe_symlink
|
||||
|
||||
# -------------
|
||||
# External dependencies
|
||||
|
||||
HOCR_OK_LANGS = frozenset(['eng', 'deu', 'spa', 'ita', 'por'])
|
||||
DEFAULT_LANGUAGE = 'eng' # Enforce English hegemony
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -56,9 +58,21 @@ log = logging.getLogger(__name__)
|
||||
verify_python3_env()
|
||||
|
||||
|
||||
def check_platform():
|
||||
if os.name == 'nt' and sys.maxsize <= 2 ** 32: # pragma: no cover
|
||||
# 32-bit interpreter on Windows
|
||||
log.error(
|
||||
"You are running OCRmyPDF in a 32-bit (x86) Python interpreter."
|
||||
"Please use a 64-bit (x86-64) version of Python."
|
||||
)
|
||||
|
||||
|
||||
def check_options_languages(options):
|
||||
if not options.language:
|
||||
options.language = ['eng'] # Enforce English hegemony
|
||||
options.language = [DEFAULT_LANGUAGE]
|
||||
system_lang = locale.getlocale()[0]
|
||||
if system_lang and not system_lang.startswith('en'):
|
||||
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
||||
|
||||
# Support v2.x "eng+deu" language syntax
|
||||
if '+' in options.language[0]:
|
||||
@@ -287,6 +301,7 @@ def check_options_pillow(options):
|
||||
|
||||
|
||||
def check_options(options):
|
||||
check_platform()
|
||||
check_options_languages(options)
|
||||
check_options_metadata(options)
|
||||
check_options_output(options)
|
||||
@@ -299,7 +314,7 @@ def check_options(options):
|
||||
check_dependency_versions(options)
|
||||
|
||||
|
||||
def check_closed_streams(options):
|
||||
def check_closed_streams(options): # pragma: no cover
|
||||
"""Work around Python issue with multiprocessing forking on closed streams
|
||||
|
||||
https://bugs.python.org/issue28326
|
||||
@@ -365,12 +380,12 @@ def create_input_file(options, work_folder):
|
||||
target = os.path.join(work_folder, 'stdin')
|
||||
with open(target, 'wb') as stream_buffer:
|
||||
copyfileobj(sys.stdin.buffer, stream_buffer)
|
||||
return target
|
||||
return target, "<stdin>"
|
||||
else:
|
||||
try:
|
||||
target = os.path.join(work_folder, 'origin')
|
||||
re_symlink(options.input_file, target)
|
||||
return target
|
||||
safe_symlink(options.input_file, target)
|
||||
return target, os.fspath(options.input_file)
|
||||
except FileNotFoundError:
|
||||
raise InputFileError(f"File not found - {options.input_file}")
|
||||
|
||||
@@ -413,6 +428,20 @@ def report_output_file_size(options, input_file, output_file):
|
||||
f"The argument --{arg.replace('_', '-')} was issued, causing transcoding."
|
||||
)
|
||||
|
||||
if options.optimize == 0:
|
||||
reasons.append("Optimization was disabled.")
|
||||
else:
|
||||
image_optimizers = {
|
||||
'jbig2': jbig2enc.available(),
|
||||
'pngquant': pngquant.available(),
|
||||
}
|
||||
for name, available in image_optimizers.items():
|
||||
if not available:
|
||||
reasons.append(
|
||||
f"The optional dependency '{name}' was not found, so some image "
|
||||
f"optimizations could not be attempted."
|
||||
)
|
||||
|
||||
if reasons:
|
||||
explanation = "Possible reasons for this include:\n" + '\n'.join(reasons) + "\n"
|
||||
else:
|
||||
@@ -427,7 +456,7 @@ def report_output_file_size(options, input_file, output_file):
|
||||
def check_dependency_versions(options):
|
||||
check_external_program(
|
||||
program='tesseract',
|
||||
package={'darwin': 'tesseract', 'linux': 'tesseract-ocr'},
|
||||
package={'linux': 'tesseract-ocr'},
|
||||
version_checker=tesseract.version,
|
||||
need_version='4.0.0', # using backport for Travis CI
|
||||
)
|
||||
@@ -437,11 +466,12 @@ def check_dependency_versions(options):
|
||||
version_checker=ghostscript.version,
|
||||
need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports
|
||||
)
|
||||
if ghostscript.version() == '9.24':
|
||||
gs_version = ghostscript.version()
|
||||
if gs_version in ('9.24', '9.51'):
|
||||
raise MissingDependencyError(
|
||||
"Ghostscript 9.24 contains serious regressions and is not "
|
||||
"supported. Please upgrade to Ghostscript 9.25 or use an older "
|
||||
"version."
|
||||
f"Ghostscript {gs_version} contains serious regressions and is not "
|
||||
"supported. Please upgrade to a newer version, or downgrade to the "
|
||||
"previous version."
|
||||
)
|
||||
check_external_program(
|
||||
program='qpdf',
|
||||
|
||||
+86
-53
@@ -18,8 +18,10 @@
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from contextlib import suppress
|
||||
from enum import IntEnum
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterable
|
||||
|
||||
from tqdm import tqdm
|
||||
|
||||
@@ -29,11 +31,19 @@ from .cli import parser
|
||||
|
||||
|
||||
class TqdmConsole:
|
||||
"""Wrapper to log messages in a way that is compatible with tqdm progress bar"""
|
||||
"""Wrapper to log messages in a way that is compatible with tqdm progress bar
|
||||
|
||||
This routes log messages through tqdm so that it can print them above the
|
||||
progress bar, and then refresh the progress bar, rather than overwriting
|
||||
it which looks messy.
|
||||
|
||||
For some reason Python 3.6 prints extra empty messages from time to time,
|
||||
so we suppress those.
|
||||
"""
|
||||
|
||||
def __init__(self, file):
|
||||
self.file = file
|
||||
self.py36 = sys.version_info >= (3, 6)
|
||||
self.py36 = sys.version_info[0:2] == (3, 6)
|
||||
|
||||
def write(self, msg):
|
||||
# When no progress bar is active, tqdm.write() routes to print()
|
||||
@@ -44,7 +54,7 @@ class TqdmConsole:
|
||||
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
||||
|
||||
def flush(self):
|
||||
if hasattr(self.file, "flush"):
|
||||
with suppress(AttributeError):
|
||||
self.file.flush()
|
||||
|
||||
|
||||
@@ -57,7 +67,11 @@ class Verbosity(IntEnum):
|
||||
debug_all = 2 #: More detailed debugging from ocrmypdf and dependent modules
|
||||
|
||||
|
||||
def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger=False):
|
||||
def configure_logging(
|
||||
verbosity: Verbosity,
|
||||
progress_bar_friendly: bool = True,
|
||||
manage_root_logger: bool = False,
|
||||
):
|
||||
"""Set up logging.
|
||||
|
||||
Library users may wish to use this function if they want their log output to be
|
||||
@@ -78,11 +92,14 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger=
|
||||
overwrite the progress bar
|
||||
manage_root_logger (bool): Configure the process's root logger, to ensure
|
||||
all log output is sent through
|
||||
|
||||
Returns:
|
||||
The toplevel logger for ocrmypdf (or the root logger, if we are managing it).
|
||||
"""
|
||||
|
||||
prefix = '' if manage_root_logger else 'ocrmypdf'
|
||||
log = logging.getLogger(prefix)
|
||||
log.setLevel(logging.INFO)
|
||||
log.setLevel(logging.DEBUG)
|
||||
|
||||
if progress_bar_friendly:
|
||||
console = logging.StreamHandler(stream=TqdmConsole(sys.stderr))
|
||||
@@ -97,8 +114,6 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger=
|
||||
console.setLevel(logging.INFO)
|
||||
|
||||
formatter = logging.Formatter('%(levelname)7s - %(message)s')
|
||||
if verbosity >= 1:
|
||||
log.setLevel(logging.DEBUG)
|
||||
if verbosity >= 2:
|
||||
formatter = logging.Formatter('%(name)s - %(levelname)7s - %(message)s')
|
||||
|
||||
@@ -114,21 +129,39 @@ def configure_logging(verbosity, progress_bar_friendly=True, manage_root_logger=
|
||||
if manage_root_logger:
|
||||
logging.captureWarnings(True)
|
||||
|
||||
return log
|
||||
|
||||
def create_options(*, input_file, output_file, **kwargs):
|
||||
|
||||
def create_options(*, input_file: os.PathLike, output_file: os.PathLike, **kwargs):
|
||||
cmdline = []
|
||||
deferred = []
|
||||
|
||||
for arg, val in kwargs.items():
|
||||
if val is None:
|
||||
continue
|
||||
if arg == 'tesseract_env':
|
||||
|
||||
# These arguments with special handling for which we bypass
|
||||
# argparse
|
||||
if arg in {'tesseract_env', 'progress_bar'}:
|
||||
deferred.append((arg, val))
|
||||
continue
|
||||
|
||||
cmd_style_arg = arg.replace('_', '-')
|
||||
cmdline.append(f"--{cmd_style_arg}")
|
||||
|
||||
# Booleans are special: add only if True, omit for False
|
||||
if isinstance(val, bool):
|
||||
if val:
|
||||
cmdline.append(f"--{cmd_style_arg}")
|
||||
continue
|
||||
|
||||
if isinstance(val, Iterable) and not isinstance(val, str):
|
||||
for elem in val:
|
||||
cmdline.append(f"--{cmd_style_arg}")
|
||||
cmdline.append(elem)
|
||||
continue
|
||||
|
||||
# We have a parameter
|
||||
cmdline.append(f"--{cmd_style_arg}")
|
||||
if isinstance(val, (int, float)):
|
||||
cmdline.append(str(val))
|
||||
elif isinstance(val, str):
|
||||
@@ -148,58 +181,58 @@ def create_options(*, input_file, output_file, **kwargs):
|
||||
|
||||
# If we are running a Tesseract spoof, ensure it knows what the input file is
|
||||
if os.environ.get('PYTEST_CURRENT_TEST') and options.tesseract_env:
|
||||
options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = input_file
|
||||
options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file)
|
||||
|
||||
return options
|
||||
|
||||
|
||||
def ocr( # pylint: disable=unused-argument
|
||||
input_file,
|
||||
output_file,
|
||||
input_file: os.PathLike,
|
||||
output_file: os.PathLike,
|
||||
*,
|
||||
language=None,
|
||||
image_dpi=None,
|
||||
language: Iterable[str] = None,
|
||||
image_dpi: int = None,
|
||||
output_type=None,
|
||||
sidecar=None,
|
||||
jobs=None,
|
||||
use_threads=None,
|
||||
title=None,
|
||||
author=None,
|
||||
subject=None,
|
||||
keywords=None,
|
||||
rotate_pages=None,
|
||||
remove_background=None,
|
||||
deskew=None,
|
||||
clean=None,
|
||||
clean_final=None,
|
||||
unpaper_args=None,
|
||||
oversample=None,
|
||||
remove_vectors=None,
|
||||
threshold=None,
|
||||
force_ocr=None,
|
||||
skip_text=None,
|
||||
redo_ocr=None,
|
||||
skip_big=None,
|
||||
optimize=None,
|
||||
jpg_quality=None,
|
||||
png_quality=None,
|
||||
jbig2_lossy=None,
|
||||
jbig2_page_group_size=None,
|
||||
pages=None,
|
||||
max_image_mpixels=None,
|
||||
tesseract_config=None,
|
||||
tesseract_pagesegmode=None,
|
||||
tesseract_oem=None,
|
||||
sidecar: os.PathLike = None,
|
||||
jobs: int = None,
|
||||
use_threads: bool = None,
|
||||
title: str = None,
|
||||
author: str = None,
|
||||
subject: str = None,
|
||||
keywords: str = None,
|
||||
rotate_pages: bool = None,
|
||||
remove_background: bool = None,
|
||||
deskew: bool = None,
|
||||
clean: bool = None,
|
||||
clean_final: bool = None,
|
||||
unpaper_args: str = None,
|
||||
oversample: int = None,
|
||||
remove_vectors: bool = None,
|
||||
threshold: bool = None,
|
||||
force_ocr: bool = None,
|
||||
skip_text: bool = None,
|
||||
redo_ocr: bool = None,
|
||||
skip_big: float = None,
|
||||
optimize: int = None,
|
||||
jpg_quality: int = None,
|
||||
png_quality: int = None,
|
||||
jbig2_lossy: bool = None,
|
||||
jbig2_page_group_size: int = None,
|
||||
pages: str = None,
|
||||
max_image_mpixels: float = None,
|
||||
tesseract_config: Iterable[str] = None,
|
||||
tesseract_pagesegmode: int = None,
|
||||
tesseract_oem: int = None,
|
||||
pdf_renderer=None,
|
||||
tesseract_timeout=None,
|
||||
rotate_pages_threshold=None,
|
||||
tesseract_timeout: float = None,
|
||||
rotate_pages_threshold: float = None,
|
||||
pdfa_image_compression=None,
|
||||
user_words=None,
|
||||
user_patterns=None,
|
||||
fast_web_view=None,
|
||||
keep_temporary_files=None,
|
||||
progress_bar=None,
|
||||
tesseract_env=None,
|
||||
user_words: os.PathLike = None,
|
||||
user_patterns: os.PathLike = None,
|
||||
fast_web_view: float = None,
|
||||
keep_temporary_files: bool = None,
|
||||
progress_bar: bool = None,
|
||||
tesseract_env: Dict[str, str] = None,
|
||||
):
|
||||
"""Run OCRmyPDF on one PDF or image.
|
||||
|
||||
|
||||
@@ -20,6 +20,8 @@ import argparse
|
||||
from ._version import PROGRAM_NAME as _PROGRAM_NAME
|
||||
from ._version import __version__ as _VERSION
|
||||
|
||||
__all__ = ['parser']
|
||||
|
||||
|
||||
def numeric(basetype, min_=None, max_=None):
|
||||
"""Validator for numeric params"""
|
||||
|
||||
@@ -20,17 +20,83 @@
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
from collections.abc import Mapping
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, run
|
||||
from distutils.version import LooseVersion
|
||||
from functools import lru_cache
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
||||
from subprocess import run as subprocess_run
|
||||
|
||||
from ..exceptions import ExitCode, MissingDependencyError
|
||||
|
||||
log = logging.Logger(__name__)
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _get_program(args, env=None):
|
||||
program = args[0]
|
||||
test_path = env.get('_OCRMYPDF_TEST_PATH', '')
|
||||
if test_path:
|
||||
program = shutil.which(program, path=test_path)
|
||||
return program
|
||||
|
||||
|
||||
def run(args, *, env=None, **kwargs):
|
||||
"""Wrapper around subprocess.run()
|
||||
|
||||
The main purpose of this wrapper is to allow us to substitute the main program
|
||||
for a spoof in the test suite. The hidden variable _OCRMYPDF_TEST_PATH replaces
|
||||
the main PATH as a location to check for programs to run.
|
||||
|
||||
Secondly we have to account for behavioral differences in Windows in particular.
|
||||
Creating symbolic links in Windows requires administrator privileges and
|
||||
may not work if for some reason we're using a FAT file system or the temporary
|
||||
folder is on a different drive from the working folder. The test suite
|
||||
works around this by creating shim Python scripts that perform the same function
|
||||
as a symbolic link, but those shims require support on this side, to ensure
|
||||
we call them with Python.
|
||||
|
||||
"""
|
||||
if not env:
|
||||
env = os.environ
|
||||
|
||||
# Search in spoof path if necessary
|
||||
program = _get_program(args, env)
|
||||
|
||||
# If we are running a .py on Windows, ensure we call it with this Python
|
||||
# (to support test suite shims)
|
||||
if os.name == 'nt' and program.lower().endswith('.py'):
|
||||
args = [sys.executable, program] + args[1:]
|
||||
else:
|
||||
args = [program] + args[1:]
|
||||
|
||||
if os.name == 'nt':
|
||||
paths = os.pathsep.join(os.get_exec_path(env))
|
||||
if not shutil.which(args[0], path=paths):
|
||||
shimmed_path = shim_paths_with_program_files(env)
|
||||
new_args0 = shutil.which(args[0], path=shimmed_path)
|
||||
if new_args0:
|
||||
args[0] = new_args0
|
||||
|
||||
process_log = log.getChild(os.path.basename(program))
|
||||
process_log.debug("Running: %s", args)
|
||||
if sys.version_info < (3, 7) and os.name == 'nt':
|
||||
# Can't use close_fds=True on Windows with Python 3.6 or older
|
||||
# https://bugs.python.org/issue19575, etc.
|
||||
kwargs['close_fds'] = False
|
||||
proc = subprocess_run(args, env=env, **kwargs)
|
||||
if process_log.isEnabledFor(logging.DEBUG):
|
||||
try:
|
||||
stderr = proc.stderr.decode('utf-8', 'replace')
|
||||
except AttributeError:
|
||||
stderr = proc.stderr
|
||||
if stderr:
|
||||
process_log.debug("stderr = %s", stderr)
|
||||
return proc
|
||||
|
||||
|
||||
def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None):
|
||||
"Get the version of the specified program"
|
||||
"""Get the version of the specified program"""
|
||||
args_prog = [program, version_arg]
|
||||
try:
|
||||
proc = run(
|
||||
@@ -66,6 +132,32 @@ def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env
|
||||
return version
|
||||
|
||||
|
||||
def shim_paths_with_program_files(env=None):
|
||||
if not env:
|
||||
env = os.environ
|
||||
program_files = env.get('PROGRAMFILES', '')
|
||||
if not program_files:
|
||||
return env.get('PATH', '')
|
||||
paths = []
|
||||
try:
|
||||
for dirname in os.listdir(program_files):
|
||||
if dirname.lower() == 'tesseract-ocr':
|
||||
paths.append(os.path.join(program_files, dirname))
|
||||
elif dirname.lower() == 'gs':
|
||||
try:
|
||||
latest_gs = max(
|
||||
os.listdir(os.path.join(program_files, dirname)),
|
||||
key=lambda d: float(d[2:]),
|
||||
)
|
||||
except (FileNotFoundError, NotADirectoryError):
|
||||
continue
|
||||
paths.append(os.path.join(program_files, dirname, latest_gs, 'bin'))
|
||||
except EnvironmentError:
|
||||
pass
|
||||
paths.extend(path for path in os.get_exec_path(env) if path not in set(paths))
|
||||
return os.pathsep.join(paths)
|
||||
|
||||
|
||||
missing_program = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH.
|
||||
@@ -111,23 +203,33 @@ On RPM-based systems (Red Hat, Fedora), search for instructions on
|
||||
installing the RPM for {program}.
|
||||
'''
|
||||
|
||||
windows_install_advice = '''
|
||||
If not already installed, install the Chocolatey package manager. Then use
|
||||
a command prompt to install the missing package:
|
||||
choco install {package}
|
||||
'''
|
||||
|
||||
|
||||
def _get_platform():
|
||||
if sys.platform.startswith('freebsd'):
|
||||
return 'freebsd'
|
||||
elif sys.platform.startswith('linux'):
|
||||
return 'linux'
|
||||
elif sys.platform.startswith('win'):
|
||||
return 'windows'
|
||||
return sys.platform
|
||||
|
||||
|
||||
def _error_trailer(program, package, **kwargs):
|
||||
if isinstance(package, Mapping):
|
||||
package = package[_get_platform()]
|
||||
package = package.get(_get_platform(), program)
|
||||
|
||||
if _get_platform() == 'darwin':
|
||||
log.info(osx_install_advice.format(**locals()))
|
||||
elif _get_platform() == 'linux':
|
||||
log.info(linux_install_advice.format(**locals()))
|
||||
elif _get_platform() == 'windows':
|
||||
log.info(windows_install_advice.format(**locals()))
|
||||
|
||||
|
||||
def _error_missing_program(program, package, required_for, recommended):
|
||||
@@ -169,7 +271,15 @@ def check_external_program(
|
||||
raise MissingDependencyError()
|
||||
return
|
||||
|
||||
if found_version < need_version:
|
||||
def remove_leading_v(s):
|
||||
if s.startswith('v'):
|
||||
return s[1:]
|
||||
return s
|
||||
|
||||
found_version = remove_leading_v(found_version)
|
||||
need_version = remove_leading_v(need_version)
|
||||
|
||||
if LooseVersion(found_version) < LooseVersion(need_version):
|
||||
_error_old_version(program, package, need_version, found_version, required_for)
|
||||
if not recommended:
|
||||
raise MissingDependencyError()
|
||||
|
||||
@@ -15,25 +15,51 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
"""Interface to Ghostscript executable"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import warnings
|
||||
from contextlib import suppress
|
||||
from functools import lru_cache
|
||||
from io import BytesIO
|
||||
from os import fspath
|
||||
from shutil import copy
|
||||
from subprocess import PIPE, STDOUT, run
|
||||
from tempfile import NamedTemporaryFile
|
||||
from pathlib import Path
|
||||
from shutil import which
|
||||
from subprocess import PIPE, CalledProcessError
|
||||
|
||||
from PIL import Image
|
||||
|
||||
from ..exceptions import SubprocessOutputError
|
||||
from . import get_version
|
||||
from ..exceptions import MissingDependencyError, SubprocessOutputError
|
||||
from . import get_version, run
|
||||
|
||||
gslog = logging.getLogger()
|
||||
|
||||
GS = 'gs'
|
||||
if os.name == 'nt':
|
||||
GS = which('gswin64c')
|
||||
if not GS:
|
||||
GS = which('gswin32c')
|
||||
if not GS:
|
||||
raise MissingDependencyError(
|
||||
"""
|
||||
---------------------------------------------------------------------
|
||||
This error normally occurs when ocrmypdf can't Ghostscript. Please
|
||||
ensure Ghostscript is installed and its location is added to the
|
||||
system PATH environment variable.
|
||||
|
||||
For details see:
|
||||
https://ocrmypdf.readthedocs.io/en/latest/installation.html
|
||||
---------------------------------------------------------------------
|
||||
"""
|
||||
)
|
||||
GS = Path(GS).stem
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def version():
|
||||
return get_version('gs')
|
||||
return get_version(GS)
|
||||
|
||||
|
||||
def jpeg_passthrough_available():
|
||||
@@ -78,9 +104,12 @@ def extract_text(input_file, pageno=1):
|
||||
else:
|
||||
pages = []
|
||||
|
||||
# Note due to bug https://bugs.ghostscript.com/show_bug.cgi?id=701971
|
||||
# Ghostscript <= 9.50 will truncate output unless we write to stdout, so
|
||||
# don't write to a file.
|
||||
args_gs = (
|
||||
[
|
||||
'gs',
|
||||
GS,
|
||||
'-dQUIET',
|
||||
'-dSAFER',
|
||||
'-dBATCH',
|
||||
@@ -89,14 +118,15 @@ def extract_text(input_file, pageno=1):
|
||||
'-dTextFormat=0',
|
||||
]
|
||||
+ pages
|
||||
+ ['-o', '-', fspath(input_file)]
|
||||
+ ['-o', '-', fspath(input_file), "-sstdout=%stderr"]
|
||||
)
|
||||
|
||||
p = run(args_gs, stdout=PIPE, stderr=PIPE)
|
||||
if p.returncode != 0:
|
||||
try:
|
||||
p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True)
|
||||
except CalledProcessError as e:
|
||||
raise SubprocessOutputError(
|
||||
'Ghostscript text extraction failed\n%s\n%s\n%s'
|
||||
% (input_file, p.stdout.decode(), p.stderr.decode())
|
||||
'Ghostscript text extraction failed\n%s\n%s'
|
||||
% (input_file, e.stderr.decode(errors='replace'))
|
||||
)
|
||||
|
||||
return p.stdout
|
||||
@@ -138,54 +168,56 @@ def rasterize_pdf(
|
||||
if not log:
|
||||
log = gslog
|
||||
|
||||
with NamedTemporaryFile(delete=True) as tmp:
|
||||
args_gs = (
|
||||
[
|
||||
'gs',
|
||||
'-dQUIET',
|
||||
'-dSAFER',
|
||||
'-dBATCH',
|
||||
'-dNOPAUSE',
|
||||
f'-sDEVICE={raster_device}',
|
||||
f'-dFirstPage={pageno}',
|
||||
f'-dLastPage={pageno}',
|
||||
f'-r{res[0]:f}x{res[1]:f}',
|
||||
]
|
||||
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
||||
+ [
|
||||
'-o',
|
||||
tmp.name,
|
||||
'-dAutoRotatePages=/None', # Probably has no effect on raster
|
||||
'-f',
|
||||
fspath(input_file),
|
||||
]
|
||||
)
|
||||
args_gs = (
|
||||
[
|
||||
GS,
|
||||
'-dQUIET',
|
||||
'-dSAFER',
|
||||
'-dBATCH',
|
||||
'-dNOPAUSE',
|
||||
f'-sDEVICE={raster_device}',
|
||||
f'-dFirstPage={pageno}',
|
||||
f'-dLastPage={pageno}',
|
||||
f'-r{res[0]:f}x{res[1]:f}',
|
||||
]
|
||||
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
||||
+ [
|
||||
'-o',
|
||||
'-',
|
||||
'-sstdout=%stderr',
|
||||
'-dAutoRotatePages=/None', # Probably has no effect on raster
|
||||
'-f',
|
||||
fspath(input_file),
|
||||
]
|
||||
)
|
||||
|
||||
log.debug(args_gs)
|
||||
p = run(args_gs, stdout=PIPE, stderr=STDOUT, universal_newlines=True)
|
||||
if _gs_error_reported(p.stdout):
|
||||
log.error(p.stdout)
|
||||
elif p.stdout:
|
||||
log.debug(p.stdout)
|
||||
log.debug(args_gs)
|
||||
try:
|
||||
p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True)
|
||||
except CalledProcessError as e:
|
||||
log.error(e.stderr.decode(errors='replace'))
|
||||
raise SubprocessOutputError('Ghostscript rasterizing failed')
|
||||
else:
|
||||
stderr = p.stderr.decode(errors='replace')
|
||||
if _gs_error_reported(stderr):
|
||||
log.error(stderr)
|
||||
elif stderr:
|
||||
log.debug(stderr)
|
||||
|
||||
if p.returncode != 0:
|
||||
raise SubprocessOutputError('Ghostscript rasterizing failed')
|
||||
|
||||
tmp.seek(0)
|
||||
with Image.open(tmp) as im:
|
||||
if rotation is not None:
|
||||
log.debug("Rotating output by %i", rotation)
|
||||
# rotation is a clockwise angle and Image.ROTATE_* is
|
||||
# counterclockwise so this cancels out the rotation
|
||||
if rotation == 90:
|
||||
im = im.transpose(Image.ROTATE_90)
|
||||
elif rotation == 180:
|
||||
im = im.transpose(Image.ROTATE_180)
|
||||
elif rotation == 270:
|
||||
im = im.transpose(Image.ROTATE_270)
|
||||
if rotation % 180 == 90:
|
||||
page_dpi = page_dpi[1], page_dpi[0]
|
||||
im.save(fspath(output_file), dpi=page_dpi)
|
||||
with Image.open(BytesIO(p.stdout)) as im:
|
||||
if rotation is not None:
|
||||
log.debug("Rotating output by %i", rotation)
|
||||
# rotation is a clockwise angle and Image.ROTATE_* is
|
||||
# counterclockwise so this cancels out the rotation
|
||||
if rotation == 90:
|
||||
im = im.transpose(Image.ROTATE_90)
|
||||
elif rotation == 180:
|
||||
im = im.transpose(Image.ROTATE_180)
|
||||
elif rotation == 270:
|
||||
im = im.transpose(Image.ROTATE_270)
|
||||
if rotation % 180 == 90:
|
||||
page_dpi = page_dpi[1], page_dpi[0]
|
||||
im.save(fspath(output_file), dpi=page_dpi)
|
||||
|
||||
|
||||
def generate_pdfa(
|
||||
@@ -193,7 +225,7 @@ def generate_pdfa(
|
||||
output_file,
|
||||
compression,
|
||||
log,
|
||||
threads=1,
|
||||
threads=None, # deprecated parameter
|
||||
pdf_version='1.5',
|
||||
pdfa_part='2',
|
||||
):
|
||||
@@ -216,6 +248,10 @@ def generate_pdfa(
|
||||
"""
|
||||
if not log:
|
||||
log = gslog
|
||||
if threads is not None:
|
||||
warnings.warn(
|
||||
"use of deprecated parameter 'threads'", category=DeprecationWarning
|
||||
)
|
||||
|
||||
compression_args = []
|
||||
if compression == 'jpeg':
|
||||
@@ -249,37 +285,55 @@ def generate_pdfa(
|
||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699216
|
||||
compression_args.append('-dPassThroughJPEGImages=false')
|
||||
|
||||
with NamedTemporaryFile(delete=True) as gs_pdf:
|
||||
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
||||
# is set; see:
|
||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699392
|
||||
args_gs = (
|
||||
[
|
||||
"gs",
|
||||
"-dQUIET",
|
||||
"-dBATCH",
|
||||
"-dNOPAUSE",
|
||||
"-dSAFER",
|
||||
"-dCompatibilityLevel=" + str(pdf_version),
|
||||
"-sDEVICE=pdfwrite",
|
||||
"-dAutoRotatePages=/None",
|
||||
"-sColorConversionStrategy=" + strategy,
|
||||
]
|
||||
+ compression_args
|
||||
+ [
|
||||
"-dJPEGQ=95",
|
||||
"-dPDFA=" + pdfa_part,
|
||||
"-dPDFACompatibilityPolicy=1",
|
||||
"-sOutputFile=" + gs_pdf.name,
|
||||
]
|
||||
)
|
||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||
log.debug(args_gs)
|
||||
p = run(args_gs, stdout=PIPE, stderr=STDOUT, universal_newlines=True)
|
||||
|
||||
if _gs_error_reported(p.stdout):
|
||||
log.error(p.stdout)
|
||||
elif 'overprint mode not set' in p.stdout:
|
||||
# nb no need to specify ProcessColorModel when ColorConversionStrategy
|
||||
# is set; see:
|
||||
# https://bugs.ghostscript.com/show_bug.cgi?id=699392
|
||||
args_gs = (
|
||||
[
|
||||
GS,
|
||||
"-dQUIET",
|
||||
"-dBATCH",
|
||||
"-dNOPAUSE",
|
||||
"-dSAFER",
|
||||
"-dCompatibilityLevel=" + str(pdf_version),
|
||||
"-sDEVICE=pdfwrite",
|
||||
"-dAutoRotatePages=/None",
|
||||
"-sColorConversionStrategy=" + strategy,
|
||||
]
|
||||
+ compression_args
|
||||
+ [
|
||||
"-dJPEGQ=95",
|
||||
"-dPDFA=" + pdfa_part,
|
||||
"-dPDFACompatibilityPolicy=1",
|
||||
"-o",
|
||||
"-",
|
||||
"-sstdout=%stderr",
|
||||
]
|
||||
)
|
||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||
try:
|
||||
with Path(output_file).open('wb') as output:
|
||||
p = run(args_gs, stdout=output, stderr=PIPE, check=True)
|
||||
except CalledProcessError as e:
|
||||
# Ghostscript does not change return code when it fails to create
|
||||
# PDF/A - check PDF/A status elsewhere
|
||||
log.error(e.stderr.decode(errors='replace'))
|
||||
raise SubprocessOutputError('Ghostscript PDF/A rendering failed')
|
||||
else:
|
||||
stderr = p.stderr.decode('utf-8', errors='replace')
|
||||
if _gs_error_reported(stderr):
|
||||
last_part = None
|
||||
repcount = 0
|
||||
for part in stderr.split('****'):
|
||||
if part != last_part:
|
||||
if repcount > 1:
|
||||
log.error(f"(previous error message repeated {repcount} times)")
|
||||
repcount = 0
|
||||
log.error(part)
|
||||
else:
|
||||
repcount += 1
|
||||
last_part = part
|
||||
elif 'overprint mode not set' in stderr:
|
||||
# Unless someone is going to print PDF/A documents on a
|
||||
# magical sRGB printer I can't see the removal of overprinting
|
||||
# being a problem....
|
||||
@@ -287,12 +341,3 @@ def generate_pdfa(
|
||||
"Ghostscript had to remove PDF 'overprinting' from the "
|
||||
"input file to complete PDF/A conversion. "
|
||||
)
|
||||
else:
|
||||
log.debug(p.stdout)
|
||||
|
||||
if p.returncode == 0:
|
||||
# Ghostscript does not change return code when it fails to create
|
||||
# PDF/A - check PDF/A status elsewhere
|
||||
copy(gs_pdf.name, fspath(output_file))
|
||||
else:
|
||||
raise SubprocessOutputError('Ghostscript PDF/A rendering failed')
|
||||
|
||||
@@ -15,11 +15,13 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
"""Interface to jbig2 executable"""
|
||||
|
||||
from functools import lru_cache
|
||||
from subprocess import PIPE, run
|
||||
from subprocess import PIPE
|
||||
|
||||
from ..exceptions import MissingDependencyError
|
||||
from . import get_version
|
||||
from . import get_version, run
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
|
||||
@@ -15,6 +15,8 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
"""Interface to pngquant executable"""
|
||||
|
||||
from functools import lru_cache
|
||||
from subprocess import run
|
||||
from tempfile import NamedTemporaryFile
|
||||
|
||||
+35
-23
@@ -15,35 +15,47 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
from functools import lru_cache
|
||||
from os import fspath
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, run
|
||||
"""Interface to qpdf executable"""
|
||||
|
||||
from . import get_version
|
||||
from io import StringIO
|
||||
|
||||
import pikepdf
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
def version():
|
||||
return get_version('qpdf', regex=r'qpdf version (.+)')
|
||||
return pikepdf.__libqpdf_version__
|
||||
|
||||
|
||||
def check(input_file, log=None):
|
||||
args_qpdf = ['qpdf', '--check', fspath(input_file)]
|
||||
|
||||
if log is None:
|
||||
import logging as log
|
||||
|
||||
pdf = None
|
||||
try:
|
||||
run(args_qpdf, stderr=STDOUT, stdout=PIPE, universal_newlines=True, check=True)
|
||||
except CalledProcessError as e:
|
||||
if e.returncode == 2:
|
||||
log.error("%s: not a valid PDF, and could not repair it.", input_file)
|
||||
log.error("Details:")
|
||||
log.error(e.output)
|
||||
elif e.returncode == 3:
|
||||
log.info("qpdf --check returned warnings:")
|
||||
log.info(e.output)
|
||||
else:
|
||||
log.warning(e.output)
|
||||
pdf = pikepdf.open(input_file)
|
||||
except pikepdf.PdfError as e:
|
||||
if log:
|
||||
log.error(e)
|
||||
return False
|
||||
return True
|
||||
else:
|
||||
messages = pdf.check()
|
||||
for msg in messages:
|
||||
if 'error' in msg.lower():
|
||||
log.error(msg)
|
||||
else:
|
||||
log.warning(msg)
|
||||
|
||||
sio = StringIO()
|
||||
linearize = None
|
||||
try:
|
||||
pdf.check_linearization(sio)
|
||||
except RuntimeError:
|
||||
pass
|
||||
else:
|
||||
linearize = sio.getvalue()
|
||||
if linearize:
|
||||
log.warning(linearize)
|
||||
|
||||
if not messages and not linearize:
|
||||
return True
|
||||
return False
|
||||
finally:
|
||||
if pdf:
|
||||
pdf.close()
|
||||
|
||||
@@ -15,22 +15,23 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
"""Interface to Tesseract executable"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
from collections import namedtuple
|
||||
from contextlib import suppress
|
||||
from functools import lru_cache
|
||||
from os import fspath
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired, run
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
||||
|
||||
from ..exceptions import (
|
||||
MissingDependencyError,
|
||||
SubprocessOutputError,
|
||||
TesseractConfigError,
|
||||
)
|
||||
from ..helpers import page_number
|
||||
from . import get_version
|
||||
from ..helpers import page_number, safe_symlink
|
||||
from . import get_version, run
|
||||
|
||||
OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence'))
|
||||
|
||||
@@ -52,6 +53,12 @@ HOCR_TEMPLATE = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
"""
|
||||
|
||||
|
||||
class TesseractLoggerAdapter(logging.LoggerAdapter):
|
||||
def process(self, msg, kwargs):
|
||||
kwargs['extra'] = self.extra
|
||||
return '[tesseract] %s' % (msg), kwargs
|
||||
|
||||
|
||||
def version(tesseract_env=None):
|
||||
return get_version('tesseract', regex=r'tesseract\s(.+)', env=tesseract_env)
|
||||
|
||||
@@ -121,9 +128,10 @@ def languages(tesseract_env=None):
|
||||
except CalledProcessError as e:
|
||||
raise MissingDependencyError(lang_error(e.output)) from e
|
||||
|
||||
for line in output.splitlines():
|
||||
if line.startswith('Error'):
|
||||
raise MissingDependencyError(lang_error(output))
|
||||
header, *rest = output.splitlines()
|
||||
if not header.startswith('List of available languages'):
|
||||
raise MissingDependencyError(lang_error(output))
|
||||
return set(lang.strip() for lang in rest)
|
||||
|
||||
|
||||
@@ -131,7 +139,7 @@ def tess_base_args(langs, engine_mode):
|
||||
args = ['tesseract']
|
||||
if langs:
|
||||
args.extend(['-l', '+'.join(langs)])
|
||||
if engine_mode is not None and v4():
|
||||
if engine_mode is not None:
|
||||
args.extend(['--oem', str(engine_mode)])
|
||||
return args
|
||||
|
||||
@@ -179,19 +187,15 @@ def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env=
|
||||
return oc
|
||||
|
||||
|
||||
def tesseract_log_output(log, stdout, input_file):
|
||||
prefix = "[tesseract] "
|
||||
def tesseract_log_output(mainlog, stdout, input_file):
|
||||
log = TesseractLoggerAdapter(
|
||||
mainlog, extra=mainlog.extra if hasattr(mainlog, 'extra') else None
|
||||
)
|
||||
|
||||
try:
|
||||
text = stdout.decode()
|
||||
except UnicodeDecodeError:
|
||||
log.error(
|
||||
prefix
|
||||
+ "command line output was not utf-8. "
|
||||
+ "This usually means Tesseract's language packs do not match "
|
||||
"the installed version of Tesseract."
|
||||
)
|
||||
text = stdout.decode('utf-8', 'backslashreplace')
|
||||
text = stdout.decode('utf-8', 'ignore')
|
||||
|
||||
lines = text.splitlines()
|
||||
for line in lines:
|
||||
@@ -200,25 +204,25 @@ def tesseract_log_output(log, stdout, input_file):
|
||||
elif line.startswith("Warning in pixReadMem"):
|
||||
continue
|
||||
elif 'diacritics' in line:
|
||||
log.warning(prefix + "lots of diacritics - possibly poor OCR")
|
||||
log.warning("lots of diacritics - possibly poor OCR")
|
||||
elif line.startswith('OSD: Weak margin'):
|
||||
log.warning(prefix + "unsure about page orientation")
|
||||
log.warning("unsure about page orientation")
|
||||
elif 'Error in pixScanForForeground' in line:
|
||||
pass # Appears to be spurious/problem with nonwhite borders
|
||||
elif 'Error in boxClipToRectangle' in line:
|
||||
pass # Always appears with pixScanForForeground message
|
||||
elif 'parameter not found: ' in line.lower():
|
||||
log.error(prefix + line.strip())
|
||||
log.error(line.strip())
|
||||
problem = line.split('found: ')[1]
|
||||
raise TesseractConfigError(problem)
|
||||
elif 'error' in line.lower() or 'exception' in line.lower():
|
||||
log.error(prefix + line.strip())
|
||||
log.error(line.strip())
|
||||
elif 'warning' in line.lower():
|
||||
log.warning(prefix + line.strip())
|
||||
log.warning(line.strip())
|
||||
elif 'read_params_file' in line.lower():
|
||||
log.error(prefix + line.strip())
|
||||
log.error(line.strip())
|
||||
else:
|
||||
log.info(prefix + line.strip())
|
||||
log.info(line.strip())
|
||||
|
||||
|
||||
def page_timedout(log, input_file, timeout):
|
||||
@@ -256,8 +260,8 @@ def generate_hocr(
|
||||
log,
|
||||
):
|
||||
|
||||
output_hocr = next(o for o in output_files if o.endswith('.hocr'))
|
||||
output_sidecar = next(o for o in output_files if o.endswith('.txt'))
|
||||
output_hocr = next(o for o in output_files if fspath(o).endswith('.hocr'))
|
||||
output_sidecar = next(o for o in output_files if fspath(o).endswith('.txt'))
|
||||
prefix = os.path.splitext(output_hocr)[0]
|
||||
|
||||
args_tesseract = tess_base_args(language, engine_mode)
|
||||
@@ -275,7 +279,6 @@ def generate_hocr(
|
||||
# to the number of order parameters here
|
||||
args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig)
|
||||
try:
|
||||
log.debug(args_tesseract)
|
||||
p = run(
|
||||
args_tesseract,
|
||||
stdout=PIPE,
|
||||
@@ -314,7 +317,7 @@ def use_skip_page(text_only, skip_pdf, output_pdf, output_text):
|
||||
# Substitute a "skipped page"
|
||||
with suppress(FileNotFoundError):
|
||||
os.remove(output_pdf) # In case it was partially created
|
||||
os.symlink(skip_pdf, output_pdf)
|
||||
safe_symlink(skip_pdf, output_pdf)
|
||||
return
|
||||
|
||||
# Or normally, just write a 0 byte file to the output to indicate a skip
|
||||
@@ -374,7 +377,6 @@ def generate_pdf(
|
||||
|
||||
args_tesseract.extend([input_image, prefix, 'pdf', 'txt'] + tessconfig)
|
||||
try:
|
||||
log.debug(args_tesseract)
|
||||
p = run(
|
||||
args_tesseract,
|
||||
stdout=PIPE,
|
||||
|
||||
@@ -18,10 +18,10 @@
|
||||
# unpaper documentation:
|
||||
# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md
|
||||
|
||||
"""Interface to unpaper executable"""
|
||||
|
||||
import os
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
from functools import lru_cache
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
||||
from tempfile import TemporaryDirectory
|
||||
@@ -30,6 +30,7 @@ from PIL import Image
|
||||
|
||||
from ..exceptions import MissingDependencyError, SubprocessOutputError
|
||||
from . import get_version
|
||||
from . import run as external_run
|
||||
|
||||
|
||||
@lru_cache(maxsize=1)
|
||||
@@ -76,7 +77,7 @@ def run(input_file, output_file, dpi, log, mode_args):
|
||||
# their unpaper arguments (whether intentionally or otherwise)
|
||||
args_unpaper.extend([input_pnm, output_pnm])
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
proc = external_run(
|
||||
args_unpaper,
|
||||
check=True,
|
||||
close_fds=True,
|
||||
|
||||
+41
-27
@@ -18,6 +18,7 @@
|
||||
import logging
|
||||
import multiprocessing
|
||||
import os
|
||||
import shutil
|
||||
import warnings
|
||||
from collections.abc import Iterable
|
||||
from contextlib import suppress
|
||||
@@ -27,14 +28,14 @@ from pathlib import Path
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def re_symlink(input_file, soft_link_name, *args, **kwargs):
|
||||
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike, *args, **kwargs):
|
||||
"""
|
||||
Helper function: relinks soft symbolic link if necessary
|
||||
"""
|
||||
if len(args) == 1 and isinstance(args[0], logging.Logger):
|
||||
log.warning("Deprecated: re_symlink(,log)")
|
||||
log.warning("Deprecated: safe_symlink(,log)")
|
||||
if 'log' in kwargs:
|
||||
log.warning('Deprecated: re_symlink(...log=)')
|
||||
log.warning('Deprecated: safe_symlink(...log=)')
|
||||
|
||||
input_file = os.fspath(input_file)
|
||||
soft_link_name = os.fspath(soft_link_name)
|
||||
@@ -60,6 +61,11 @@ def re_symlink(input_file, soft_link_name, *args, **kwargs):
|
||||
if not os.path.exists(input_file):
|
||||
raise FileNotFoundError(f"trying to create a broken symlink to {input_file}")
|
||||
|
||||
if os.name == 'nt':
|
||||
# Don't actually use symlinks on Windows due to permission issues
|
||||
shutil.copyfile(input_file, soft_link_name)
|
||||
return
|
||||
|
||||
log.debug("os.symlink(%s, %s)", input_file, soft_link_name)
|
||||
|
||||
# Create symbolic link using absolute path
|
||||
@@ -70,12 +76,12 @@ def is_iterable_notstr(thing):
|
||||
return isinstance(thing, Iterable) and not isinstance(thing, str)
|
||||
|
||||
|
||||
def monotonic(L):
|
||||
def monotonic(L: Iterable):
|
||||
"""Does list increase monotonically?"""
|
||||
return all(b > a for a, b in zip(L, L[1:]))
|
||||
|
||||
|
||||
def page_number(input_file):
|
||||
def page_number(input_file: os.PathLike):
|
||||
"""Get one-based page number implied by filename (000002.pdf -> 2)"""
|
||||
return int(os.path.basename(os.fspath(input_file))[0:6])
|
||||
|
||||
@@ -91,35 +97,43 @@ def available_cpu_count():
|
||||
return 1
|
||||
|
||||
|
||||
def is_file_writable(test_file):
|
||||
def is_file_writable(test_file: os.PathLike):
|
||||
"""Intentionally racy test if target is writable.
|
||||
|
||||
We intend to write to the output file if and only if we succeed and
|
||||
can replace it atomically. Before doing the OCR work, make sure
|
||||
the location is writable.
|
||||
"""
|
||||
p = Path(test_file)
|
||||
|
||||
if p.is_symlink():
|
||||
p = p.resolve(strict=False)
|
||||
|
||||
# p.is_file() throws an exception in some cases
|
||||
if p.exists() and p.is_file():
|
||||
return os.access(
|
||||
os.fspath(p),
|
||||
os.W_OK,
|
||||
effective_ids=(os.access in os.supports_effective_ids),
|
||||
)
|
||||
else:
|
||||
try:
|
||||
fp = p.open('wb')
|
||||
except OSError:
|
||||
return False
|
||||
try:
|
||||
if not isinstance(test_file, Path):
|
||||
p = Path(test_file)
|
||||
else:
|
||||
fp.close()
|
||||
with suppress(OSError):
|
||||
p.unlink()
|
||||
return True
|
||||
p = test_file
|
||||
|
||||
if p.is_symlink():
|
||||
p = p.resolve(strict=False)
|
||||
|
||||
# p.is_file() throws an exception in some cases
|
||||
if p.exists() and p.is_file():
|
||||
return os.access(
|
||||
os.fspath(p),
|
||||
os.W_OK,
|
||||
effective_ids=(os.access in os.supports_effective_ids),
|
||||
)
|
||||
else:
|
||||
try:
|
||||
fp = p.open('wb')
|
||||
except OSError:
|
||||
return False
|
||||
else:
|
||||
fp.close()
|
||||
with suppress(OSError):
|
||||
p.unlink()
|
||||
return True
|
||||
except (EnvironmentError, RuntimeError) as e:
|
||||
log.debug(e)
|
||||
log.error(str(e))
|
||||
return False
|
||||
|
||||
|
||||
def deprecated(func):
|
||||
|
||||
@@ -88,7 +88,7 @@ class HocrTransform:
|
||||
if self.width is None or self.height is None:
|
||||
raise HocrTransformError("hocr file is missing page dimensions")
|
||||
|
||||
def __str__(self):
|
||||
def __str__(self): # pragma: no cover
|
||||
"""
|
||||
Return the textual content of the HTML body
|
||||
"""
|
||||
@@ -190,7 +190,7 @@ class HocrTransform:
|
||||
pt = self.pt_from_pixel(pxl_coords)
|
||||
|
||||
# draw the bbox border
|
||||
if showBoundingboxes:
|
||||
if showBoundingboxes: # pragma: no cover
|
||||
pdf.rect(
|
||||
pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, fill=1
|
||||
)
|
||||
@@ -231,7 +231,7 @@ class HocrTransform:
|
||||
pdf.save()
|
||||
|
||||
@classmethod
|
||||
def polyval(cls, poly, x):
|
||||
def polyval(cls, poly, x): # pragma: no cover
|
||||
return x * poly[0] + poly[1]
|
||||
|
||||
def _do_line(
|
||||
@@ -269,7 +269,7 @@ class HocrTransform:
|
||||
# of the line box
|
||||
baseline_y2 = self.height - (line_box.y2 + intercept)
|
||||
|
||||
if showBoundingboxes:
|
||||
if showBoundingboxes: # pragma: no cover
|
||||
# draw the baseline in magenta, dashed
|
||||
pdf.setDash()
|
||||
pdf.setStrokeColorRGB(0.95, 0.65, 0.95)
|
||||
@@ -318,7 +318,7 @@ class HocrTransform:
|
||||
font_width = pdf.stringWidth(elemtxt, fontname, fontsize)
|
||||
|
||||
# draw the bbox border
|
||||
if showBoundingboxes:
|
||||
if showBoundingboxes: # pragma: no cover
|
||||
pdf.rect(
|
||||
box.x1, self.height - line_box.y2, box_width, line_height, fill=0
|
||||
)
|
||||
|
||||
+57
-18
@@ -33,14 +33,46 @@ from io import BytesIO
|
||||
from os import fspath
|
||||
from tempfile import TemporaryFile
|
||||
|
||||
from .exceptions import MissingDependencyError
|
||||
from .exec import shim_paths_with_program_files
|
||||
from .lib._leptonica import ffi
|
||||
|
||||
# pylint: disable=protected-access
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
lept = ffi.dlopen(find_library('lept'))
|
||||
lept.setMsgSeverity(lept.L_SEVERITY_WARNING)
|
||||
if os.name == 'nt':
|
||||
libname = 'liblept-5'
|
||||
os.environ['PATH'] = shim_paths_with_program_files()
|
||||
else:
|
||||
libname = 'lept'
|
||||
_libpath = find_library(libname)
|
||||
if not _libpath:
|
||||
raise MissingDependencyError(
|
||||
"""
|
||||
---------------------------------------------------------------------
|
||||
This error normally occurs when ocrmypdf can't find the Leptonica
|
||||
library, which is usually installed with Tesseract OCR. It could be that
|
||||
Tesseract is not installed properly, we can't find the installation
|
||||
on your system PATH environment variable.
|
||||
|
||||
The library we are looking for is usually called:
|
||||
liblept-5.dll (Windows)
|
||||
liblept*.dylib (macOS)
|
||||
liblept*.so (Linux/BSD)
|
||||
|
||||
Please review our installation procedures to find a solution:
|
||||
https://ocrmypdf.readthedocs.io/en/latest/installation.html
|
||||
---------------------------------------------------------------------
|
||||
"""
|
||||
)
|
||||
try:
|
||||
lept = ffi.dlopen(_libpath)
|
||||
lept.setMsgSeverity(lept.L_SEVERITY_WARNING)
|
||||
except ffi.error as e:
|
||||
raise MissingDependencyError(
|
||||
f"Leptonica library found at {_libpath}, but we could not access it"
|
||||
) from e
|
||||
|
||||
|
||||
class _LeptonicaErrorTrap:
|
||||
@@ -292,9 +324,11 @@ class Pix(LeptonicaObject):
|
||||
Leptonica can load TIFF, PNM (PBM, PGM, PPM), PNG, and JPEG. If
|
||||
loading fails then the object will wrap a C null pointer.
|
||||
"""
|
||||
filename = fspath(path)
|
||||
with _LeptonicaErrorTrap():
|
||||
return cls(lept.pixRead(os.fsencode(filename)))
|
||||
with open(path, 'rb') as py_file:
|
||||
data = py_file.read()
|
||||
buffer = ffi.from_buffer(data)
|
||||
with _LeptonicaErrorTrap():
|
||||
return cls(lept.pixReadMem(buffer, len(buffer)))
|
||||
|
||||
def write_implied_format(self, path, jpeg_quality=0, jpeg_progressive=0):
|
||||
"""Write pix to the filename, with the extension indicating format.
|
||||
@@ -302,11 +336,19 @@ class Pix(LeptonicaObject):
|
||||
jpeg_quality -- quality (iff JPEG; 1 - 100, 0 for default)
|
||||
jpeg_progressive -- (iff JPEG; 0 for baseline seq., 1 for progressive)
|
||||
"""
|
||||
filename = fspath(path)
|
||||
with _LeptonicaErrorTrap():
|
||||
lept.pixWriteImpliedFormat(
|
||||
os.fsencode(filename), self._cdata, jpeg_quality, jpeg_progressive
|
||||
)
|
||||
lept_format = lept.getImpliedFileFormat(os.fsencode(path))
|
||||
with open(path, 'wb') as py_file:
|
||||
data = ffi.new('l_uint8 **pdata')
|
||||
size = ffi.new('size_t *psize')
|
||||
with _LeptonicaErrorTrap():
|
||||
if lept_format == lept.L_JPEG_ENCODE:
|
||||
lept.pixWriteMemJpeg(
|
||||
data, size, self._cdata, jpeg_quality, jpeg_progressive
|
||||
)
|
||||
else:
|
||||
lept.pixWriteMem(data, size, self._cdata, lept_format)
|
||||
buffer = ffi.buffer(data[0], size[0])
|
||||
py_file.write(buffer)
|
||||
|
||||
@classmethod
|
||||
def frompil(self, pillow_image):
|
||||
@@ -502,17 +544,14 @@ class Pix(LeptonicaObject):
|
||||
display=0,
|
||||
pdfdir=ffi.NULL,
|
||||
):
|
||||
if get_leptonica_version() < 'leptonica-1.76':
|
||||
# Leptonica 1.76 changed the API for pixFindPageForeground; we don't
|
||||
# support the old version
|
||||
raise LeptonicaError("Not available in this version of Leptonica")
|
||||
with _LeptonicaErrorTrap():
|
||||
cropbox = Box(
|
||||
lept.pixFindPageForeground(
|
||||
self._cdata,
|
||||
threshold,
|
||||
mindist,
|
||||
erasedist,
|
||||
pagenum,
|
||||
showmorph,
|
||||
display,
|
||||
pdfdir,
|
||||
self._cdata, threshold, mindist, erasedist, showmorph, ffi.NULL
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -16,6 +16,8 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from cffi import FFI
|
||||
|
||||
ffibuilder = FFI()
|
||||
@@ -74,6 +76,17 @@ struct Pixa
|
||||
};
|
||||
typedef struct Pixa PIXA;
|
||||
|
||||
/*! Array of compressed pix */
|
||||
struct PixaComp
|
||||
{
|
||||
l_int32 n; /*!< number of PixComp in ptr array */
|
||||
l_int32 nalloc; /*!< number of PixComp ptrs allocated */
|
||||
l_int32 offset; /*!< indexing offset into ptr array */
|
||||
struct PixComp **pixc; /*!< the array of ptrs to PixComp */
|
||||
struct Boxa *boxa; /*!< array of boxes */
|
||||
};
|
||||
typedef struct PixaComp PIXAC;
|
||||
|
||||
struct Box
|
||||
{
|
||||
l_int32 x;
|
||||
@@ -210,9 +223,15 @@ ffibuilder.cdef(
|
||||
"""
|
||||
PIX * pixRead ( const char *filename );
|
||||
PIX * pixReadMem ( const l_uint8 *data, size_t size );
|
||||
PIX * pixReadStream ( FILE *fp, l_int32 hint );
|
||||
PIX * pixScale ( PIX *pixs, l_float32 scalex, l_float32 scaley );
|
||||
l_int32 pixFindSkew ( PIX *pixs, l_float32 *pangle, l_float32 *pconf );
|
||||
l_int32 pixWriteImpliedFormat ( const char *filename, PIX *pix, l_int32 quality, l_int32 progressive );
|
||||
l_int32 getImpliedFileFormat ( const char *filename );
|
||||
l_ok pixWriteStream ( FILE *fp, PIX *pix, l_int32 format );
|
||||
l_ok pixWriteStreamJpeg ( FILE *fp, PIX *pixs, l_int32 quality, l_int32 progressive );
|
||||
l_ok pixWriteMem ( l_uint8 **pdata, size_t *psize, PIX *pix, l_int32 format );
|
||||
l_ok pixWriteMemJpeg ( l_uint8 **pdata, size_t *psize, PIX *pix, l_int32 quality, l_int32 progressive );
|
||||
l_int32
|
||||
pixWriteMemPng(l_uint8 **pdata,
|
||||
size_t *psize,
|
||||
@@ -294,14 +313,12 @@ pixCleanBackgroundToWhite(PIX *pixs,
|
||||
l_int32 whiteval);
|
||||
|
||||
BOX *
|
||||
pixFindPageForeground(PIX *pixs,
|
||||
l_int32 threshold,
|
||||
l_int32 mindist,
|
||||
l_int32 erasedist,
|
||||
l_int32 pagenum,
|
||||
l_int32 showmorph,
|
||||
l_int32 display,
|
||||
const char *pdfdir);
|
||||
pixFindPageForeground ( PIX *pixs,
|
||||
l_int32 threshold,
|
||||
l_int32 mindist,
|
||||
l_int32 erasedist,
|
||||
l_int32 showmorph,
|
||||
PIXAC *pixac );
|
||||
|
||||
PIX *
|
||||
pixClipRectangle(PIX *pixs,
|
||||
@@ -414,7 +431,10 @@ pixExtractBarcodes(PIX *pixs,
|
||||
l_int32 debugflag);
|
||||
|
||||
BOXA *
|
||||
pixLocateBarcodes ( PIX *pixs, l_int32 thresh, PIX **ppixb, PIX **ppixm );
|
||||
pixLocateBarcodes ( PIX *pixs,
|
||||
l_int32 thresh,
|
||||
PIX **ppixb,
|
||||
PIX **ppixm );
|
||||
|
||||
SARRAY *
|
||||
pixReadBarcodes(PIXA *pixa,
|
||||
@@ -423,6 +443,12 @@ pixReadBarcodes(PIXA *pixa,
|
||||
SARRAY **psaw,
|
||||
l_int32 debugflag);
|
||||
|
||||
PIX *
|
||||
pixGenHalftoneMask(PIX *pixs,
|
||||
PIX **ppixtext,
|
||||
l_int32 *phtfound,
|
||||
PIXA *pixadb);
|
||||
|
||||
l_int32
|
||||
l_generateCIDataForPdf(const char *fname,
|
||||
PIX *pix,
|
||||
@@ -491,3 +517,8 @@ ffibuilder.set_source("ocrmypdf.lib._leptonica", None)
|
||||
|
||||
if __name__ == '__main__':
|
||||
ffibuilder.compile(verbose=True)
|
||||
if Path('ocrmypdf/lib/_leptonica.py').exists() and Path('src/ocrmypdf').exists():
|
||||
output = Path('ocrmypdf/lib/_leptonica.py')
|
||||
output.rename('src/ocrmypdf/lib/_leptonica.py')
|
||||
Path('ocrmypdf/lib').rmdir()
|
||||
Path('ocrmypdf').rmdir()
|
||||
|
||||
@@ -31,7 +31,7 @@ from . import leptonica
|
||||
from ._jobcontext import PDFContext
|
||||
from .exceptions import OutputFileAccessError
|
||||
from .exec import jbig2enc, pngquant
|
||||
from .helpers import re_symlink
|
||||
from .helpers import safe_symlink
|
||||
|
||||
DEFAULT_JPEG_QUALITY = 75
|
||||
DEFAULT_PNG_QUALITY = 70
|
||||
@@ -111,6 +111,11 @@ def extract_image_generic(*, pike, root, log, image, xref, options):
|
||||
if pim.bits_per_component == 1:
|
||||
return None
|
||||
|
||||
try:
|
||||
pim.indexed # pikepdf 1.6.3 can't handle [/Indexed [/Array...]]
|
||||
except NotImplementedError:
|
||||
return None
|
||||
|
||||
if filtdp[0] == Name.DCTDecode and options.optimize >= 2:
|
||||
# This is a simple heuristic derived from some training data, that has
|
||||
# about a 70% chance of guessing whether the JPEG is high quality,
|
||||
@@ -150,6 +155,17 @@ def extract_image_generic(*, pike, root, log, image, xref, options):
|
||||
# generating a PNG from compressed data
|
||||
pim.as_pil_image().save(png_name(root, xref))
|
||||
return xref, '.png'
|
||||
elif (
|
||||
not pim.indexed
|
||||
and pim.colorspace == Name.ICCBased
|
||||
and pim.bits_per_component == 1
|
||||
and not options.jbig2_lossy
|
||||
):
|
||||
# We can losslessly optimize 1-bit images to CCITT or JBIG2 without
|
||||
# paying any attention to the ICC profile, provided we're not doing
|
||||
# lossy JBIG2
|
||||
pim.as_pil_image().save(png_name(root, xref))
|
||||
return xref, '.png'
|
||||
|
||||
return None
|
||||
|
||||
@@ -487,7 +503,7 @@ def optimize(input_file, output_file, context, save_settings):
|
||||
log = context.log
|
||||
options = context.options
|
||||
if options.optimize == 0:
|
||||
re_symlink(input_file, output_file)
|
||||
safe_symlink(input_file, output_file)
|
||||
return
|
||||
|
||||
if options.jpeg_quality == 0:
|
||||
@@ -533,7 +549,7 @@ def optimize(input_file, output_file, context, save_settings):
|
||||
pike.remove_unreferenced_resources()
|
||||
pike.save(output_file, **save_settings)
|
||||
else:
|
||||
re_symlink(target_file, output_file)
|
||||
safe_symlink(target_file, output_file)
|
||||
|
||||
|
||||
def main(infile, outfile, level, jobs=1):
|
||||
@@ -544,11 +560,11 @@ def main(infile, outfile, level, jobs=1):
|
||||
"""Emulate ocrmypdf's options"""
|
||||
|
||||
def __init__(
|
||||
self, input_file, jobs, optimize, jpeg_quality, png_quality, jb2lossy
|
||||
self, input_file, jobs, optimize_, jpeg_quality, png_quality, jb2lossy
|
||||
):
|
||||
self.input_file = input_file
|
||||
self.jobs = jobs
|
||||
self.optimize = optimize
|
||||
self.optimize = optimize_
|
||||
self.jpeg_quality = jpeg_quality
|
||||
self.png_quality = png_quality
|
||||
self.jbig2_page_group_size = 0
|
||||
@@ -559,7 +575,7 @@ def main(infile, outfile, level, jobs=1):
|
||||
options = OptimizeOptions(
|
||||
input_file=infile,
|
||||
jobs=jobs,
|
||||
optimize=int(level),
|
||||
optimize_=int(level),
|
||||
jpeg_quality=0, # Use default
|
||||
png_quality=0,
|
||||
jb2lossy=False,
|
||||
|
||||
+2
-16
@@ -32,8 +32,6 @@ Ghostscript's handling of pdfmark.
|
||||
"""
|
||||
|
||||
import base64
|
||||
import os
|
||||
from binascii import hexlify
|
||||
from pathlib import Path
|
||||
from string import Template
|
||||
|
||||
@@ -54,18 +52,7 @@ pdfa_def_template = u"""%!
|
||||
def
|
||||
|
||||
[/_objdef {icc_PDFA} /type /stream /OBJ pdfmark
|
||||
[{icc_PDFA}
|
||||
<<
|
||||
/N currentpagedevice /ProcessColorModel known {
|
||||
currentpagedevice /ProcessColorModel get dup /DeviceGray eq
|
||||
{pop 1} {
|
||||
/DeviceRGB eq
|
||||
{3}{4} ifelse
|
||||
} ifelse
|
||||
} {
|
||||
(ERROR, unable to determine ProcessColorModel) == flush
|
||||
} ifelse
|
||||
>> /PUT pdfmark
|
||||
[{icc_PDFA} << /N 3 >> /PUT pdfmark
|
||||
[{icc_PDFA} ICCProfile /PUT pdfmark
|
||||
|
||||
% Define the output intent dictionary :
|
||||
@@ -95,8 +82,6 @@ def generate_pdfa_ps(target_filename, icc='sRGB'):
|
||||
|
||||
:param target_filename: filename to save
|
||||
:param icc: ICC identifier such as 'sRGB'
|
||||
|
||||
:returns: a string containing the entire pdfmark
|
||||
"""
|
||||
if icc == 'sRGB':
|
||||
icc_profile = SRGB_ICC_PROFILE
|
||||
@@ -114,6 +99,7 @@ def generate_pdfa_ps(target_filename, icc='sRGB'):
|
||||
# We should have encoded everything to pure ASCII by this point, and
|
||||
# to be safe, only allow ASCII in PostScript
|
||||
Path(target_filename).write_text(ps, encoding='ascii')
|
||||
return target_filename
|
||||
|
||||
|
||||
def file_claims_pdfa(filename):
|
||||
|
||||
@@ -96,6 +96,7 @@ def extract_text_xml(infile, pdf, pageno=None, log=gslog):
|
||||
page_count_difference = len(pdf.pages) - len(page_xml)
|
||||
if page_count_difference != 0:
|
||||
log.error("The number of pages in the input file is inconsistent.")
|
||||
log.error(f"Expected {len(pdf.pages)}, txtwrite says {len(page_xml)}")
|
||||
if page_count_difference > 0:
|
||||
page_xml.extend([None] * page_count_difference)
|
||||
return page_xml
|
||||
|
||||
@@ -18,11 +18,11 @@
|
||||
|
||||
import logging
|
||||
import re
|
||||
from collections import namedtuple
|
||||
from collections import defaultdict, namedtuple
|
||||
from decimal import Decimal
|
||||
from enum import Enum
|
||||
from math import hypot, isclose
|
||||
from os import fspath
|
||||
from os import PathLike, fspath
|
||||
from pathlib import Path
|
||||
from warnings import warn
|
||||
|
||||
@@ -30,17 +30,17 @@ import pikepdf
|
||||
from pikepdf import PdfMatrix
|
||||
from tqdm import tqdm
|
||||
|
||||
from ocrmypdf.exceptions import EncryptedPdfError, MissingDependencyError
|
||||
|
||||
from . import ghosttext
|
||||
from .layout import get_page_analysis, get_text_boxes
|
||||
from ocrmypdf.exceptions import EncryptedPdfError
|
||||
from ocrmypdf.exec import ghostscript
|
||||
from ocrmypdf.pdfinfo import ghosttext
|
||||
from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes
|
||||
|
||||
logger = logging.getLogger()
|
||||
|
||||
Colorspace = Enum('Colorspace', 'gray rgb cmyk lab icc index sep devn pattern jpeg2000')
|
||||
|
||||
Encoding = Enum(
|
||||
'Encoding', 'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate ' + 'runlength'
|
||||
'Encoding', 'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate runlength'
|
||||
)
|
||||
|
||||
FRIENDLY_COLORSPACE = {
|
||||
@@ -98,7 +98,7 @@ XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_dep
|
||||
InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth'])
|
||||
|
||||
ContentsInfo = namedtuple(
|
||||
'ContentsInfo', ['xobject_settings', 'inline_images', 'found_vector']
|
||||
'ContentsInfo', ['xobject_settings', 'inline_images', 'found_vector', 'name_index']
|
||||
)
|
||||
|
||||
TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt'])
|
||||
@@ -151,6 +151,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
||||
ctm = PdfMatrix(initial_shorthand)
|
||||
xobject_settings = []
|
||||
inline_images = []
|
||||
name_index = defaultdict(lambda: [])
|
||||
found_vector = False
|
||||
vector_ops = set('S s f F f* B B* b b*'.split())
|
||||
image_ops = set('BI ID EI q Q Do cm'.split())
|
||||
@@ -185,6 +186,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
||||
name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack)
|
||||
)
|
||||
xobject_settings.append(settings)
|
||||
name_index[image_name].append(settings)
|
||||
elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this
|
||||
iimage = operands[0]
|
||||
inline = InlineSettings(
|
||||
@@ -198,6 +200,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
||||
xobject_settings=xobject_settings,
|
||||
inline_images=inline_images,
|
||||
found_vector=found_vector,
|
||||
name_index=name_index,
|
||||
)
|
||||
|
||||
|
||||
@@ -303,14 +306,22 @@ class ImageInfo:
|
||||
if self._enc == Encoding.jpeg2000:
|
||||
self._color = Colorspace.jpeg2000
|
||||
|
||||
self._comp = FRIENDLY_COMP.get(self._color, '?')
|
||||
if self._color == Colorspace.icc:
|
||||
# Check the ICC profile to determine actual colorspace
|
||||
pim_icc = pim.icc
|
||||
if pim_icc.profile.xcolor_space == 'GRAY':
|
||||
self._comp = 1
|
||||
elif pim_icc.profile.xcolor_space == 'CMYK':
|
||||
self._comp = 4
|
||||
else:
|
||||
self._comp = 3
|
||||
else:
|
||||
self._comp = FRIENDLY_COMP.get(self._color, '?')
|
||||
|
||||
# Bit of a hack... infer grayscale if component count is uncertain
|
||||
# but encoding must be monochrome. This happens if a monochrome image
|
||||
# has an ICC profile attached. Better solution would be to examine
|
||||
# the ICC profile.
|
||||
if self._comp == '?' and self._enc in (Encoding.ccitt, Encoding.jbig2):
|
||||
self._comp = FRIENDLY_COMP[Colorspace.gray]
|
||||
# Bit of a hack... infer grayscale if component count is uncertain
|
||||
# but encoding only supports monochrome.
|
||||
if self._comp == '?' and self._enc in (Encoding.ccitt, Encoding.jbig2):
|
||||
self._comp = FRIENDLY_COMP[Colorspace.gray]
|
||||
|
||||
@property
|
||||
def name(self):
|
||||
@@ -411,13 +422,9 @@ def _find_regular_images(container, contentsinfo):
|
||||
"""
|
||||
|
||||
for pdfimage, xobj in _image_xobjects(container):
|
||||
|
||||
# For each image that is drawn on this, check if we drawing the
|
||||
# current image - yes this is O(n^2), but n == 1 almost always
|
||||
for draw in contentsinfo.xobject_settings:
|
||||
if draw.name != xobj:
|
||||
continue
|
||||
|
||||
if xobj not in contentsinfo.name_index:
|
||||
continue
|
||||
for draw in contentsinfo.name_index[xobj]:
|
||||
if draw.stack_depth == 0 and _is_unit_square(draw.shorthand):
|
||||
# At least one PDF in the wild (and test suite) draws an image
|
||||
# when the graphics stack depth is 0, meaning that the image
|
||||
@@ -551,7 +558,7 @@ def simplify_textboxes(miner, textbox_getter):
|
||||
yield TextboxInfo(box.bbox, visible, corrupt)
|
||||
|
||||
|
||||
def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext):
|
||||
def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str):
|
||||
pageinfo = {}
|
||||
pageinfo['pageno'] = pageno
|
||||
pageinfo['images'] = []
|
||||
@@ -611,25 +618,28 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile, xmltext):
|
||||
|
||||
def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None, progbar=False):
|
||||
pdf = pikepdf.open(infile) # Do not close in this function
|
||||
if pdf.is_encrypted:
|
||||
pdf.close()
|
||||
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
||||
if detailed_analysis:
|
||||
pages_xml = None
|
||||
else:
|
||||
pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log)
|
||||
try:
|
||||
if pdf.is_encrypted:
|
||||
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
||||
if detailed_analysis:
|
||||
pages_xml = None
|
||||
else:
|
||||
pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log)
|
||||
|
||||
pages = []
|
||||
for n, _ in tqdm(
|
||||
enumerate(pdf.pages),
|
||||
total=len(pdf.pages),
|
||||
desc="Scan",
|
||||
unit='page',
|
||||
disable=not progbar,
|
||||
):
|
||||
page_xml = pages_xml[n] if pages_xml else None
|
||||
page = PageInfo(pdf, n, infile, page_xml, detailed_analysis)
|
||||
pages.append(page)
|
||||
pages = []
|
||||
for n, _ in tqdm(
|
||||
enumerate(pdf.pages),
|
||||
total=len(pdf.pages),
|
||||
desc="Scan",
|
||||
unit='page',
|
||||
disable=not progbar,
|
||||
):
|
||||
page_xml = pages_xml[n] if pages_xml else None
|
||||
page = PageInfo(pdf, n, infile, page_xml, detailed_analysis)
|
||||
pages.append(page)
|
||||
except Exception:
|
||||
pdf.close()
|
||||
raise
|
||||
|
||||
return pages, pdf
|
||||
|
||||
@@ -750,6 +760,8 @@ class PdfInfo:
|
||||
|
||||
def __init__(self, infile, detailed_page_analysis=False, log=logger, progbar=False):
|
||||
self._infile = infile
|
||||
if ghostscript.version() in ('9.52',):
|
||||
detailed_page_analysis = True # txtwrite doesn't work in these versions
|
||||
self._pages, pdf = _pdf_get_all_pageinfo(
|
||||
infile, detailed_page_analysis, log=log, progbar=progbar
|
||||
)
|
||||
@@ -805,10 +817,14 @@ def main():
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument('infile')
|
||||
args = parser.parse_args()
|
||||
info = _pdf_get_all_pageinfo(args.infile)
|
||||
pagesinfo, pdfinfo = _pdf_get_all_pageinfo(args.infile)
|
||||
from pprint import pprint
|
||||
|
||||
pprint(info)
|
||||
pprint(pdfinfo)
|
||||
for page in pagesinfo:
|
||||
pprint(page)
|
||||
for im in page.images:
|
||||
pprint(im)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
|
||||
@@ -20,6 +20,7 @@ from math import copysign
|
||||
from pathlib import Path
|
||||
from unittest.mock import patch
|
||||
|
||||
import pdfminer
|
||||
import pdfminer.encodingdb
|
||||
import pdfminer.pdfdevice
|
||||
import pdfminer.pdfinterp
|
||||
@@ -36,51 +37,54 @@ from ..exceptions import EncryptedPdfError
|
||||
STRIP_NAME = re.compile(r'[0-9]+')
|
||||
|
||||
#
|
||||
# Unconditional pdfminer patches
|
||||
# pdfminer 20181108 patches
|
||||
#
|
||||
|
||||
if pdfminer.__version__ == '20181108':
|
||||
|
||||
def name2unicode(name):
|
||||
"""Fix pdfminer's name2unicode function
|
||||
def name2unicode(name):
|
||||
"""Fix pdfminer's name2unicode function
|
||||
|
||||
Font cids that are mapped to names of the form /g123 seem to be, by convention
|
||||
characters with no corresponding Unicode entry. These can be subsetted fonts
|
||||
or symbolic fonts. There seems to be no way to map /g123 fonts to Unicode,
|
||||
barring a ToUnicode data structure.
|
||||
"""
|
||||
if name in glyphname2unicode:
|
||||
return glyphname2unicode[name]
|
||||
if name.startswith('g') or name.startswith('a'):
|
||||
raise KeyError(name)
|
||||
if name.startswith('uni'):
|
||||
try:
|
||||
return chr(int(name[3:], 16))
|
||||
except ValueError: # Not hexadecimal
|
||||
Font cids that are mapped to names of the form /g123 seem to be, by convention
|
||||
characters with no corresponding Unicode entry. These can be subsetted fonts
|
||||
or symbolic fonts. There seems to be no way to map /g123 fonts to Unicode,
|
||||
barring a ToUnicode data structure.
|
||||
"""
|
||||
if name in glyphname2unicode:
|
||||
return glyphname2unicode[name]
|
||||
if name.startswith('g') or name.startswith('a'):
|
||||
raise KeyError(name)
|
||||
m = STRIP_NAME.search(name)
|
||||
if not m:
|
||||
raise KeyError(name)
|
||||
return chr(int(m.group(0)))
|
||||
if name.startswith('uni'):
|
||||
try:
|
||||
return chr(int(name[3:], 16))
|
||||
except ValueError: # Not hexadecimal
|
||||
raise KeyError(name)
|
||||
m = STRIP_NAME.search(name)
|
||||
if not m:
|
||||
raise KeyError(name)
|
||||
return chr(int(m.group(0)))
|
||||
|
||||
pdfminer.encodingdb.name2unicode = name2unicode
|
||||
|
||||
pdfminer.encodingdb.name2unicode = name2unicode
|
||||
original_PDFFont_init = PDFFont.__init__
|
||||
|
||||
original_PDFFont_init = PDFFont.__init__
|
||||
def PDFFont__init__(self, descriptor, widths, default_width=None):
|
||||
original_PDFFont_init(self, descriptor, widths, default_width)
|
||||
# PDF spec says descent should be negative
|
||||
# A font with a positive descent implies it floats entirely above the
|
||||
# baseline, i.e. it's not really a baseline anymore. I have fonts that
|
||||
# claim a positive descent, but treating descent as positive always seems
|
||||
# to misposition text.
|
||||
if self.descent > 0:
|
||||
self.descent = -self.descent
|
||||
|
||||
PDFFont.__init__ = PDFFont__init__
|
||||
|
||||
def PDFFont__init__(self, descriptor, widths, default_width=None):
|
||||
original_PDFFont_init(self, descriptor, widths, default_width)
|
||||
# PDF spec says descent should be negative
|
||||
# A font with a positive descent implies it floats entirely above the
|
||||
# baseline, i.e. it's not really a baseline anymore. I have fonts that
|
||||
# claim a positive descent, but treating descent as positive always seems
|
||||
# to misposition text.
|
||||
if self.descent > 0:
|
||||
self.descent = -self.descent
|
||||
#
|
||||
# end of pdfminer 20181108 patches
|
||||
#
|
||||
|
||||
|
||||
PDFFont.__init__ = PDFFont__init__
|
||||
|
||||
original_PDFSimpleFont_init = PDFSimpleFont.__init__
|
||||
|
||||
|
||||
@@ -97,6 +101,7 @@ def PDFSimpleFont__init__(self, descriptor, widths, spec):
|
||||
|
||||
|
||||
PDFSimpleFont.__init__ = PDFSimpleFont__init__
|
||||
|
||||
#
|
||||
# pdfminer patches when creator is PScript5.dll
|
||||
#
|
||||
@@ -207,6 +212,7 @@ class TextPositionTracker(PDFLayoutAnalyzer):
|
||||
super().__init__(rsrcmgr, pageno, laparams)
|
||||
self.textstate = None
|
||||
self.result = None
|
||||
self.cur_item = None # not defined in pdfminer code as it should be
|
||||
|
||||
def begin_page(self, page, ctm):
|
||||
super().begin_page(page, ctm)
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# This file is part of OCRmyPDF.
|
||||
#
|
||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||
# it under the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation, either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
import re
|
||||
from typing import Iterable
|
||||
|
||||
"""Utilities to measure OCR quality"""
|
||||
|
||||
|
||||
class OcrQualityDictionary:
|
||||
"""Manages a dictionary for simple OCR quality checks."""
|
||||
|
||||
def __init__(self, *, wordlist: Iterable[str] = []):
|
||||
"""Construct a dictionary from a list of words.
|
||||
|
||||
Words for which capitalization is important should be capitalized in the
|
||||
dictionary. Words that contain spaces or other punctuation will never match.
|
||||
"""
|
||||
self.dictionary = set()
|
||||
self.dictionary.update(w for w in wordlist)
|
||||
|
||||
def measure_words_matched(self, ocr_text: str) -> float:
|
||||
"""Check how many unique words in the OCR text match a dictionary.
|
||||
|
||||
Words with mixed capitalized are only considered a match if the test word
|
||||
matches that capitalization.
|
||||
|
||||
Returns:
|
||||
number of words that match / number
|
||||
"""
|
||||
text = re.sub(r"[0-9_]+", ' ', ocr_text)
|
||||
text = re.sub(r'\W+', ' ', text)
|
||||
text_words_list = re.split(r'\s+', text)
|
||||
text_words = {w for w in text_words_list if len(w) >= 3}
|
||||
|
||||
matches = 0
|
||||
for w in text_words:
|
||||
if w in self.dictionary or (
|
||||
w != w.lower() and w.lower() in self.dictionary
|
||||
):
|
||||
matches += 1
|
||||
if matches > 0:
|
||||
hit_ratio = matches / len(text_words)
|
||||
else:
|
||||
hit_ratio = 0.0
|
||||
return hit_ratio
|
||||
BIN
Binary file not shown.
+1
@@ -0,0 +1 @@
|
||||
Tesseract Open Source OCR Engine v4.1.0 with Leptonica
|
||||
+30
@@ -0,0 +1,30 @@
|
||||
Tarnose
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Bokale oa
|
||||
|
||||
|
||||
|
||||
Lehuntze
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Mugerre
|
||||
|
||||
|
||||
|
||||
|
||||
Milafranga Komunikabideak
|
||||
|
||||
BAIONA i zeettnansise —
|
||||
|
||||
1 Trenbideak -- ~~~
|
||||
|
||||
t\ Basusarri — spmsans20141004 se: . a ~
|
||||
|
||||
BIN
Binary file not shown.
+1
@@ -0,0 +1 @@
|
||||
Tesseract Open Source OCR Engine v4.1.0 with Leptonica
|
||||
+2
@@ -0,0 +1,2 @@
|
||||
Covfefe is a perfectly cromulent word.
|
||||
|
||||
BIN
Binary file not shown.
+1
@@ -0,0 +1 @@
|
||||
Tesseract Open Source OCR Engine v4.1.0 with Leptonica
|
||||
+27
@@ -0,0 +1,27 @@
|
||||
Linzensoep a la Waterman
|
||||
|
||||
|
||||
|
||||
4 ons linzen
|
||||
|
||||
3 liter water
|
||||
|
||||
3 uien
|
||||
|
||||
bloem, boter
|
||||
|
||||
2 kopjes melk
|
||||
|
||||
laurier, kruidnagel, kerrie, zout
|
||||
|
||||
De linzgen wassen en in -l liter kokend wa-
|
||||
ter 1 dag laten weken, 2 liter water bij
|
||||
de linzen voegen, zonder het water waarin
|
||||
ze geweekt zijn af te gieten, De helft van
|
||||
de uien bakken met laurier en Kruicdnagel.
|
||||
Alle uien, kerrie en gout bij de linzen
|
||||
voegen, Alles aan de kook brengen,. Van de
|
||||
bloem met boter en melk een papje maken en
|
||||
verder afmaken met de soep, Als de linzen
|
||||
gaar Zijn is de soep klaar.
|
||||
|
||||
Vendored
+3
@@ -66,3 +66,6 @@
|
||||
{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
|
||||
{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
|
||||
{"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.4", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/graph_ocred.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
|
||||
{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]}
|
||||
{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
|
||||
{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]}
|
||||
|
||||
+114
-43
@@ -15,24 +15,19 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
import ast
|
||||
import os
|
||||
import platform
|
||||
import sys
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, run
|
||||
from ocrmypdf import api, cli
|
||||
|
||||
import pytest
|
||||
|
||||
from ocrmypdf import api, cli
|
||||
|
||||
pytest_plugins = ['helpers_namespace']
|
||||
|
||||
try:
|
||||
from pytest_cov.embed import cleanup_on_sigterm
|
||||
except ImportError:
|
||||
pass
|
||||
else:
|
||||
cleanup_on_sigterm()
|
||||
|
||||
# pylint: disable=E1101
|
||||
# pytest.helpers is dynamic so it confuses pylint
|
||||
@@ -81,14 +76,69 @@ PROJECT_ROOT = os.path.dirname(TESTS_ROOT)
|
||||
OCRMYPDF = [sys.executable, '-m', 'ocrmypdf']
|
||||
|
||||
|
||||
WINDOWS_SHIM_TEMPLATE = """
|
||||
# This is a shim for Windows that has the same effect as a symlink to the target .py
|
||||
# file
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
args = [sys.executable, {spoofer}, *sys.argv[1:]]
|
||||
p = subprocess.run(args, check=False, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
|
||||
sys.stdout.buffer.write(p.stdout)
|
||||
sys.stderr.buffer.write(p.stderr)
|
||||
sys.exit(p.returncode)
|
||||
"""
|
||||
|
||||
assert ast.parse(WINDOWS_SHIM_TEMPLATE.format(spoofer=repr(r"C:\\Temp\\file.py")))
|
||||
|
||||
|
||||
@pytest.helpers.register
|
||||
def spoof(tmp_path_factory, **kwargs):
|
||||
"""Modify PATH to override subprocess executables
|
||||
|
||||
spoof(program1='replacement', ...)
|
||||
spoof(tmp_path_factory, program1='replacement', ...)
|
||||
|
||||
Creates temporary directory with symlinks to targets.
|
||||
For the test suite we need a way override executables, so that we can
|
||||
substitute desired results such as errors or just speed up OCR.
|
||||
|
||||
On POSIXish platforms we create a temporary folder with overrides that
|
||||
are symlinks to the executables we want to run. We do not actually override
|
||||
PATH. We also set an environment variable _OCRMYPDF_TEST_PATH, which
|
||||
OCRmyPDF's subprocess wrapper will check before they use regular PATH. The
|
||||
output is a folder full of executables we are overriding. We can override
|
||||
multiple executables. The end result is a folder we can use in a PATH-style
|
||||
lookup to override some executables:
|
||||
|
||||
/tmp/abcxyz/tesseract -> ocrmypdf/tests/resources/spoof/tesseract_crash.py
|
||||
/tmp/abcxyz/gs -> ocrmypdf/tests/resources/spoof/gs_backflip.py
|
||||
|
||||
Windows needs extra help from us because usually, only the Administrator
|
||||
can create symlinks. Instead we create small Python scripts that call
|
||||
the programs we want, implementing the effect of a symlink. This is cleaner
|
||||
than creating Windows executables or trying to use non-Python scripts.
|
||||
The temporary folder generated for Windows could like:
|
||||
|
||||
%TEMP%\abcxyz\tesseract.py:
|
||||
(script that runs ocrmypdf/tests/resources/spoof/tesseract_crash.py)
|
||||
%TEMP%\abcxyz\gswin32c.py:
|
||||
(script that runs ocrmypdf/tests/resources/spoof/gs_backflip.py)
|
||||
%TEMP%\abcxyz\gswin64c.py:
|
||||
(script that runs ocrmypdf/tests/resources/spoof/gs_backflip.py)
|
||||
|
||||
We also address one quirk here, that Ghostscript may be known as gswin32c
|
||||
or gswin64c, depending on what the user installed (regardless of Windows
|
||||
itself). On POSIX, Ghostscript is just 'gs'. We handle the special case here
|
||||
too.
|
||||
|
||||
All of this is intimately dependent on the machinery in ocrmypdf.exec.run().
|
||||
In particular, for Windows, that code has to know that if there is a .py
|
||||
file, it needs to run it with Python, since Windows does not like being
|
||||
asked to execute files.
|
||||
|
||||
We don't overload PATH directly because we have some tests where we call
|
||||
ocrmypdf as a subprocess (to exercise the command line interface) and some
|
||||
tests where we call it as an API.
|
||||
"""
|
||||
env = os.environ.copy()
|
||||
slug = '-'.join(v.replace('.py', '') for v in sorted(kwargs.values()))
|
||||
@@ -98,45 +148,33 @@ def spoof(tmp_path_factory, **kwargs):
|
||||
|
||||
for replace_program, with_spoof in kwargs.items():
|
||||
spoofer = Path(SPOOF_PATH) / with_spoof
|
||||
spoofer.chmod(0o755)
|
||||
(tmpdir / replace_program).symlink_to(spoofer)
|
||||
|
||||
env['_OCRMYPDF_SAVE_PATH'] = env['PATH']
|
||||
env['PATH'] = str(tmpdir) + ":" + env['PATH']
|
||||
if os.name != 'nt':
|
||||
spoofer.chmod(0o755)
|
||||
(tmpdir / replace_program).symlink_to(spoofer)
|
||||
else:
|
||||
py_file = WINDOWS_SHIM_TEMPLATE.format(
|
||||
spoofer=repr(os.fspath(spoofer.absolute()))
|
||||
)
|
||||
if replace_program == 'gs':
|
||||
programs = ['gswin64c', 'gswin32c']
|
||||
else:
|
||||
programs = [replace_program]
|
||||
for prog in programs:
|
||||
(tmpdir / f'{prog}.py').write_text(py_file, encoding='utf-8')
|
||||
|
||||
env['_OCRMYPDF_TEST_PATH'] = str(tmpdir) + os.pathsep + env['PATH']
|
||||
if os.name == 'nt':
|
||||
if '.py' not in env['PATHEXT'].lower():
|
||||
raise EnvironmentError("PATHEXT is not configured to support .py")
|
||||
return env
|
||||
|
||||
|
||||
@pytest.helpers.register
|
||||
@contextmanager
|
||||
def os_environ(new_env):
|
||||
old_env = os.environ.copy()
|
||||
if new_env is None:
|
||||
new_env = {}
|
||||
|
||||
for k, v in new_env.items():
|
||||
if k != 'PYTEST_CURRENT_TEST':
|
||||
os.environ[k] = v
|
||||
yield
|
||||
new_keys = set(os.environ.copy()) - set(old_env)
|
||||
for k in new_keys:
|
||||
if k != 'PYTEST_CURRENT_TEST':
|
||||
del os.environ[k]
|
||||
for k in old_env:
|
||||
if k != 'PYTEST_CURRENT_TEST':
|
||||
os.environ[k] = old_env[k]
|
||||
|
||||
for k, v in os.environ.copy().items():
|
||||
if k != 'PYTEST_CURRENT_TEST':
|
||||
assert v == old_env[k]
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
@pytest.fixture
|
||||
def spoof_tesseract_noop(tmp_path_factory):
|
||||
return spoof(tmp_path_factory, tesseract='tesseract_noop.py')
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
@pytest.fixture
|
||||
def spoof_tesseract_cache(tmp_path_factory):
|
||||
if running_in_docker():
|
||||
return os.environ.copy()
|
||||
@@ -183,7 +221,7 @@ def check_ocrmypdf(input_file, output_file, *args, env=None):
|
||||
api.check_options(options)
|
||||
if env:
|
||||
options.tesseract_env = env
|
||||
options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = input_file
|
||||
options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file)
|
||||
result = api.run_pipeline(options, api=True)
|
||||
|
||||
assert result == 0
|
||||
@@ -193,18 +231,51 @@ def check_ocrmypdf(input_file, output_file, *args, env=None):
|
||||
return output_file
|
||||
|
||||
|
||||
@pytest.helpers.register
|
||||
def run_ocrmypdf_api(input_file, output_file, *args, env=None):
|
||||
"""Run ocrmypdf via API and let caller deal with results
|
||||
|
||||
Does not currently have a way to manipulate the PATH except for Tesseract.
|
||||
"""
|
||||
|
||||
options = cli.parser.parse_args(
|
||||
[str(input_file), str(output_file)]
|
||||
+ [str(arg) for arg in args if arg is not None]
|
||||
)
|
||||
api.check_options(options)
|
||||
if env:
|
||||
options.tesseract_env = env.copy()
|
||||
options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file)
|
||||
first_path = env.get('_OCRMYPDF_TEST_PATH', '').split(os.pathsep)[0]
|
||||
if 'spoof' in first_path:
|
||||
assert 'gs' not in first_path, "use run_ocrmypdf() for gs"
|
||||
assert 'tesseract' in first_path
|
||||
if options.tesseract_env:
|
||||
assert all(isinstance(v, (str, bytes)) for v in options.tesseract_env.values())
|
||||
|
||||
return api.run_pipeline(options, api=False)
|
||||
|
||||
|
||||
@pytest.helpers.register
|
||||
def run_ocrmypdf(input_file, output_file, *args, env=None, universal_newlines=True):
|
||||
"Run ocrmypdf and let caller deal with results"
|
||||
|
||||
if env is None:
|
||||
env = os.environ
|
||||
env = os.environ.copy()
|
||||
|
||||
p_args = (
|
||||
OCRMYPDF
|
||||
+ [str(arg) for arg in args if arg is not None]
|
||||
+ [str(input_file), str(output_file)]
|
||||
)
|
||||
|
||||
# Tell subprocess where to find coverage.py configuration
|
||||
# This has no effect except when coverage is running
|
||||
# Details: https://coverage.readthedocs.io/en/coverage-5.0/subprocess.html
|
||||
coverage_rc = Path(__file__).parent.parent / '.coveragerc'
|
||||
assert coverage_rc.exists()
|
||||
env['COVERAGE_PROCESS_START'] = os.fspath(coverage_rc)
|
||||
|
||||
p = run(
|
||||
p_args, stdout=PIPE, stderr=PIPE, universal_newlines=universal_newlines, env=env
|
||||
)
|
||||
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
|
After Width: | Height: | Size: 145 KiB |
Binary file not shown.
@@ -0,0 +1,40 @@
|
||||
# © 2019 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# Permission is hereby granted, free of charge, to any person obtaining a
|
||||
# copy of this software and associated documentation files (the
|
||||
# "Software"), to deal in the Software without restriction, including
|
||||
# without limitation the rights to use, copy, modify, merge, publish,
|
||||
# distribute, sublicense, and/or sell copies of the Software, and to
|
||||
# permit persons to whom the Software is furnished to do so, subject to
|
||||
# the following conditions:
|
||||
#
|
||||
# The above copyright notice and this permission notice shall be included
|
||||
# in all copies or substantial portions of the Software.
|
||||
#
|
||||
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS
|
||||
# OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
|
||||
# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT.
|
||||
# IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY
|
||||
# CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT,
|
||||
# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
"""Find Ghostscript executable"""
|
||||
|
||||
|
||||
import os
|
||||
import shutil
|
||||
|
||||
|
||||
def real_ghostscript(argv):
|
||||
if os.name != 'nt':
|
||||
gs = shutil.which('gs')
|
||||
gs_args = [gs] + argv[1:]
|
||||
os.execv(gs_args[0], gs_args)
|
||||
else:
|
||||
gs = shutil.which('gswin64c')
|
||||
if not gs:
|
||||
gs = shutil.which('gswin32c')
|
||||
os.execv(gs, argv[1:])
|
||||
|
||||
return # Not reachable
|
||||
@@ -25,23 +25,17 @@ import os
|
||||
import sys
|
||||
from subprocess import check_call
|
||||
|
||||
from gs import real_ghostscript
|
||||
|
||||
"""Replicate one type of Ghostscript feature elision warning during
|
||||
PDF/A creation."""
|
||||
|
||||
|
||||
def real_ghostscript(argv):
|
||||
gs_args = ['gs'] + argv[1:]
|
||||
os.execvp("gs", gs_args)
|
||||
return # Not reachable
|
||||
|
||||
|
||||
elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1
|
||||
not permitted in PDF/A-2, overprint mode not set"""
|
||||
|
||||
|
||||
def main():
|
||||
os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH']
|
||||
if '--version' in sys.argv:
|
||||
print('9.20')
|
||||
print('SPOOFED: ' + os.path.basename(__file__))
|
||||
|
||||
@@ -23,19 +23,14 @@
|
||||
import os
|
||||
import sys
|
||||
|
||||
from gs import real_ghostscript
|
||||
|
||||
|
||||
"""Replicate Ghostscript PDF/A conversion failure by suppressing some
|
||||
arguments"""
|
||||
|
||||
|
||||
def real_ghostscript(argv):
|
||||
gs_args = ['gs'] + argv[1:]
|
||||
os.execvp("gs", gs_args)
|
||||
return # Not reachable
|
||||
|
||||
|
||||
def main():
|
||||
os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH']
|
||||
if '--version' in sys.argv:
|
||||
print('9.20')
|
||||
print('SPOOFED: ' + os.path.basename(__file__))
|
||||
|
||||
@@ -24,29 +24,24 @@
|
||||
import os
|
||||
import sys
|
||||
|
||||
from gs import real_ghostscript
|
||||
|
||||
"""Replicate Ghostscript raster failure while allowing rendering"""
|
||||
|
||||
|
||||
def real_ghostscript(argv):
|
||||
gs_args = ['gs'] + argv[1:]
|
||||
os.execvp("gs", gs_args)
|
||||
return # Not reachable
|
||||
|
||||
|
||||
def main():
|
||||
os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH']
|
||||
if '--version' in sys.argv:
|
||||
print('9.20')
|
||||
print('SPOOFED: ' + os.path.basename(__file__))
|
||||
sys.exit(0)
|
||||
|
||||
# For any rendering calls (device == pdfwrite) call real ghostscript
|
||||
if '-sDEVICE=pdfwrite' in sys.argv:
|
||||
# For non-image rastering calls, use real ghostscript
|
||||
if '-sDEVICE=pdfwrite' in sys.argv or '-sDEVICE=txtwrite' in sys.argv:
|
||||
real_ghostscript(sys.argv)
|
||||
return
|
||||
|
||||
# Fail
|
||||
print("ERROR: Ghost story archive not found")
|
||||
print("ERROR: Ghost story archive not found", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
|
||||
@@ -25,15 +25,10 @@
|
||||
import os
|
||||
import sys
|
||||
|
||||
|
||||
def real_ghostscript(argv):
|
||||
gs_args = ['gs'] + argv[1:]
|
||||
os.execvp("gs", gs_args)
|
||||
return # Not reachable
|
||||
from gs import real_ghostscript
|
||||
|
||||
|
||||
def main():
|
||||
os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH']
|
||||
if '--version' in sys.argv:
|
||||
print('9.20')
|
||||
print('SPOOFED: ' + os.path.basename(__file__))
|
||||
@@ -45,7 +40,7 @@ def main():
|
||||
return
|
||||
|
||||
# Fail
|
||||
print("ERROR: Casper is not a friendly ghost")
|
||||
print("ERROR: Casper is not a friendly ghost", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
|
||||
@@ -22,7 +22,6 @@
|
||||
|
||||
import sys
|
||||
|
||||
|
||||
"""Tesseract bad utf8 spoof
|
||||
|
||||
In 'hocr' mode or 'pdf' mode, return error code 1 and some non-Unicode
|
||||
|
||||
@@ -59,9 +59,6 @@ import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
if '_OCRMYPDF_SAVE_PATH' in os.environ:
|
||||
os.environ['PATH'] = os.environ['_OCRMYPDF_SAVE_PATH']
|
||||
|
||||
__version__ = subprocess.check_output(
|
||||
['tesseract', '--version'], stderr=subprocess.STDOUT
|
||||
).decode()
|
||||
|
||||
@@ -32,9 +32,10 @@ In orientation check mode, report the orientation is upright.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import img2pdf
|
||||
import PyPDF2 as pypdf
|
||||
import pikepdf
|
||||
from PIL import Image
|
||||
|
||||
VERSION_STRING = '''tesseract 4.0.0
|
||||
@@ -99,12 +100,10 @@ def main():
|
||||
pagesize = im.size[0] / dpi[0], im.size[1] / dpi[1]
|
||||
ptsize = pagesize[0] * 72, pagesize[1] * 72
|
||||
|
||||
pdf_out = pypdf.PdfFileWriter()
|
||||
pdf_out.addBlankPage(ptsize[0], ptsize[1])
|
||||
with open(output + '.pdf', 'wb') as f:
|
||||
pdf_out.write(f)
|
||||
with open(output + '.txt', 'w') as f:
|
||||
f.write('')
|
||||
pdf_out = pikepdf.new()
|
||||
pdf_out.add_blank_page(page_size=ptsize)
|
||||
pdf_out.save(Path(output).with_suffix('.pdf'), static_id=True)
|
||||
Path(output).with_suffix('.txt').write_text('')
|
||||
else:
|
||||
inputf = sys.argv[-4]
|
||||
output = sys.argv[-3]
|
||||
|
||||
@@ -0,0 +1,42 @@
|
||||
# © 2019 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# This file is part of OCRmyPDF.
|
||||
#
|
||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||
# it under the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation, either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
import logging
|
||||
|
||||
import pytest
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
check_ocrmypdf = pytest.helpers.check_ocrmypdf
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def acroform(resources):
|
||||
return resources / 'acroform.pdf'
|
||||
|
||||
|
||||
def test_acroform_and_redo(acroform, caplog, no_outpdf):
|
||||
with pytest.raises(ocrmypdf.exceptions.InputFileError):
|
||||
check_ocrmypdf(acroform, no_outpdf, '--redo-ocr')
|
||||
assert '--redo-ocr is not currently possible' in caplog.text
|
||||
|
||||
|
||||
def test_acroform_message(acroform, caplog, spoof_tesseract_noop, outpdf):
|
||||
caplog.set_level(logging.INFO)
|
||||
check_ocrmypdf(acroform, outpdf, env=spoof_tesseract_noop)
|
||||
assert 'fillable form' in caplog.text
|
||||
assert '--force-ocr' in caplog.text
|
||||
@@ -0,0 +1,68 @@
|
||||
# © 2019 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# This file is part of OCRmyPDF.
|
||||
#
|
||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||
# it under the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation, either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
import logging
|
||||
from io import StringIO
|
||||
|
||||
import pytest
|
||||
from tqdm import tqdm
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
|
||||
def test_raw_console():
|
||||
bio = StringIO()
|
||||
tqconsole = ocrmypdf.api.TqdmConsole(file=bio)
|
||||
tqconsole.write("Test")
|
||||
tqconsole.flush()
|
||||
assert "Test" in bio.getvalue()
|
||||
|
||||
|
||||
def test_tqdm_console():
|
||||
log = logging.getLogger()
|
||||
log.setLevel(logging.INFO)
|
||||
|
||||
formatter = logging.Formatter('%(message)s')
|
||||
|
||||
bio = StringIO()
|
||||
console = logging.StreamHandler(ocrmypdf.api.TqdmConsole(file=bio))
|
||||
console.setFormatter(formatter)
|
||||
|
||||
log.addHandler(console)
|
||||
|
||||
def before_pbar(message):
|
||||
# Ensure that log messages appear before the progress bar, even when
|
||||
# printed after the progress bar updates.
|
||||
v = bio.getvalue()
|
||||
pbar_start_marker = '|#'
|
||||
return v.index(message) < v.index(pbar_start_marker)
|
||||
|
||||
with tqdm(total=2, file=bio, disable=False) as pbar:
|
||||
pbar.update()
|
||||
msg = "1/2 above progress bar"
|
||||
log.info(msg)
|
||||
assert before_pbar(msg)
|
||||
|
||||
log.info("done")
|
||||
assert not before_pbar("done")
|
||||
|
||||
|
||||
def test_language_list():
|
||||
with pytest.raises(
|
||||
(ocrmypdf.exceptions.InputFileError, ocrmypdf.exceptions.MissingDependencyError)
|
||||
):
|
||||
ocrmypdf.ocr('doesnotexist.pdf', '_.pdf', language=['eng', 'deu'])
|
||||
@@ -15,10 +15,15 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
from subprocess import run, PIPE
|
||||
from subprocess import PIPE, run
|
||||
|
||||
import pytest
|
||||
|
||||
pytestmark = pytest.mark.skipif(
|
||||
pytest.helpers.running_in_docker(), # pylint: disable=no-member
|
||||
reason="docker can't complete",
|
||||
)
|
||||
|
||||
|
||||
def test_fish():
|
||||
try:
|
||||
|
||||
+77
-14
@@ -22,22 +22,56 @@ import pikepdf
|
||||
import pytest
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf.exceptions import ExitCode
|
||||
from ocrmypdf.exec.ghostscript import rasterize_pdf
|
||||
|
||||
check_ocrmypdf = pytest.helpers.check_ocrmypdf
|
||||
run_ocrmypdf = pytest.helpers.run_ocrmypdf
|
||||
run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api
|
||||
spoof = pytest.helpers.spoof
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def linn(resources):
|
||||
path = resources / 'linn.pdf'
|
||||
def spoof_no_tess_gs_render_fail(tmp_path_factory):
|
||||
return spoof(
|
||||
tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_render_failure.py'
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def spoof_no_tess_gs_raster_fail(tmp_path_factory):
|
||||
return spoof(
|
||||
tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_raster_failure.py'
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def spoof_no_tess_no_pdfa(tmp_path_factory):
|
||||
return spoof(
|
||||
tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_pdfa_failure.py'
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def spoof_no_tess_pdfa_warning(tmp_path_factory):
|
||||
return spoof(
|
||||
tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_feature_elision.py'
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def francais(resources):
|
||||
path = resources / 'francais.pdf'
|
||||
return path, pikepdf.open(path)
|
||||
|
||||
|
||||
def test_rasterize_size(linn, outdir, caplog):
|
||||
path, pdf = linn
|
||||
def test_rasterize_size(francais, outdir, caplog):
|
||||
path, pdf = francais
|
||||
page_size_pts = (pdf.pages[0].MediaBox[2], pdf.pages[0].MediaBox[3])
|
||||
assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0
|
||||
page_size = (page_size_pts[0] / Decimal(72), page_size_pts[1] / Decimal(72))
|
||||
target_size = Decimal('200.0'), Decimal('150.0')
|
||||
target_dpi = 42.0, 4242.0
|
||||
target_size = Decimal('50.0'), Decimal('30.0')
|
||||
forced_dpi = 42.0, 4242.0
|
||||
|
||||
log = logging.getLogger()
|
||||
rasterize_pdf(
|
||||
@@ -47,21 +81,21 @@ def test_rasterize_size(linn, outdir, caplog):
|
||||
target_size[1] / page_size[1],
|
||||
raster_device='pngmono',
|
||||
log=log,
|
||||
page_dpi=target_dpi,
|
||||
page_dpi=forced_dpi,
|
||||
)
|
||||
|
||||
with Image.open(outdir / 'out.png') as im:
|
||||
assert im.size == target_size
|
||||
assert im.info['dpi'] == target_dpi
|
||||
assert im.info['dpi'] == forced_dpi
|
||||
|
||||
|
||||
def test_rasterize_rotated(linn, outdir, caplog):
|
||||
path, pdf = linn
|
||||
def test_rasterize_rotated(francais, outdir, caplog):
|
||||
path, pdf = francais
|
||||
page_size_pts = (pdf.pages[0].MediaBox[2], pdf.pages[0].MediaBox[3])
|
||||
assert pdf.pages[0].MediaBox[0] == pdf.pages[0].MediaBox[1] == 0
|
||||
page_size = (page_size_pts[0] / Decimal(72), page_size_pts[1] / Decimal(72))
|
||||
target_size = Decimal('200.0'), Decimal('150.0')
|
||||
target_dpi = 42.0, 4242.0
|
||||
target_size = Decimal('50.0'), Decimal('30.0')
|
||||
forced_dpi = 42.0, 4242.0
|
||||
|
||||
log = logging.getLogger()
|
||||
caplog.set_level(logging.DEBUG)
|
||||
@@ -72,10 +106,39 @@ def test_rasterize_rotated(linn, outdir, caplog):
|
||||
target_size[1] / page_size[1],
|
||||
raster_device='pngmono',
|
||||
log=log,
|
||||
page_dpi=target_dpi,
|
||||
page_dpi=forced_dpi,
|
||||
rotation=90,
|
||||
)
|
||||
|
||||
with Image.open(outdir / 'out.png') as im:
|
||||
assert im.size == (target_size[1], target_size[0])
|
||||
assert im.info['dpi'] == (target_dpi[1], target_dpi[0])
|
||||
assert im.info['dpi'] == (forced_dpi[1], forced_dpi[0])
|
||||
|
||||
|
||||
def test_gs_render_failure(spoof_no_tess_gs_render_fail, resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(
|
||||
resources / 'blank.pdf', outpdf, env=spoof_no_tess_gs_render_fail
|
||||
)
|
||||
assert 'Casper is not a friendly ghost' in err
|
||||
assert p.returncode == ExitCode.child_process_error
|
||||
|
||||
|
||||
def test_gs_raster_failure(spoof_no_tess_gs_raster_fail, resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(
|
||||
resources / 'francais.pdf', outpdf, env=spoof_no_tess_gs_raster_fail
|
||||
)
|
||||
assert 'Ghost story archive not found' in err
|
||||
assert p.returncode == ExitCode.child_process_error
|
||||
|
||||
|
||||
def test_ghostscript_pdfa_failure(spoof_no_tess_no_pdfa, resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(
|
||||
resources / 'francais.pdf', outpdf, env=spoof_no_tess_no_pdfa
|
||||
)
|
||||
assert (
|
||||
p.returncode == ExitCode.pdfa_conversion_failed
|
||||
), "Unexpected return when PDF/A fails"
|
||||
|
||||
|
||||
def test_ghostscript_feature_elision(spoof_no_tess_pdfa_warning, resources, outpdf):
|
||||
check_ocrmypdf(resources / 'francais.pdf', outpdf, env=spoof_no_tess_pdfa_warning)
|
||||
|
||||
+20
-18
@@ -16,37 +16,39 @@
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import pikepdf
|
||||
import pytest
|
||||
|
||||
import ocrmypdf
|
||||
import pikepdf
|
||||
|
||||
os_environ = pytest.helpers.os_environ
|
||||
|
||||
|
||||
def test_no_glyphless_graft(resources, outdir):
|
||||
pdf = pikepdf.open(resources / 'francais.pdf')
|
||||
pdf_aspect = pikepdf.open(resources / 'aspect.pdf')
|
||||
pdf_cmyk = pikepdf.open(resources / 'cmyk.pdf')
|
||||
pdf.pages.extend(pdf_aspect.pages)
|
||||
pdf.pages.extend(pdf_cmyk.pages)
|
||||
pdf.save(outdir / 'test.pdf')
|
||||
with pikepdf.open(resources / 'francais.pdf') as pdf, pikepdf.open(
|
||||
resources / 'aspect.pdf'
|
||||
) as pdf_aspect, pikepdf.open(resources / 'cmyk.pdf') as pdf_cmyk:
|
||||
pdf.pages.extend(pdf_aspect.pages)
|
||||
pdf.pages.extend(pdf_cmyk.pages)
|
||||
pdf.save(outdir / 'test.pdf')
|
||||
|
||||
env = os.environ.copy()
|
||||
env['_OCRMYPDF_MAX_REPLACE_PAGES'] = '2'
|
||||
with os_environ(env):
|
||||
with patch('ocrmypdf._graft.MAX_REPLACE_PAGES', 2):
|
||||
ocrmypdf.ocr(
|
||||
outdir / 'test.pdf', outdir / 'out.pdf', deskew=True, tesseract_timeout=0
|
||||
outdir / 'test.pdf',
|
||||
outdir / 'out.pdf',
|
||||
deskew=True,
|
||||
tesseract_timeout=0,
|
||||
force_ocr=True,
|
||||
)
|
||||
# This test needs asserts
|
||||
|
||||
|
||||
def test_links(resources, outpdf):
|
||||
ocrmypdf.ocr(
|
||||
resources / 'link.pdf', outpdf, redo_ocr=True, oversample=200, output_type='pdf'
|
||||
)
|
||||
pdf = pikepdf.open(outpdf)
|
||||
p1 = pdf.pages[0]
|
||||
p2 = pdf.pages[1]
|
||||
assert p1.Annots[0].A.D[0].objgen == p2.objgen
|
||||
assert p2.Annots[0].A.D[0].objgen == p1.objgen
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
p1 = pdf.pages[0]
|
||||
p2 = pdf.pages[1]
|
||||
assert p1.Annots[0].A.D[0].objgen == p2.objgen
|
||||
assert p2.Annots[0].A.D[0].objgen == p1.objgen
|
||||
|
||||
@@ -0,0 +1,97 @@
|
||||
# © 2019 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# This file is part of OCRmyPDF.
|
||||
#
|
||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||
# it under the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation, either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
import logging
|
||||
import multiprocessing
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
import ocrmypdf.helpers as helpers
|
||||
|
||||
|
||||
class TestSafeSymlink:
|
||||
def test_safe_symlink_link_self(self, tmp_path, caplog):
|
||||
helpers.safe_symlink(tmp_path / 'self', tmp_path / 'self')
|
||||
assert caplog.record_tuples[0][1] == logging.WARNING
|
||||
|
||||
def test_safe_symlink_overwrite(self, tmp_path):
|
||||
(tmp_path / 'regular_file').touch()
|
||||
with pytest.raises(FileExistsError):
|
||||
helpers.safe_symlink(tmp_path / 'input', tmp_path / 'regular_file')
|
||||
|
||||
def test_safe_symlink_relink(self, tmp_path):
|
||||
(tmp_path / 'regular_file_a').touch()
|
||||
(tmp_path / 'regular_file_b').write_bytes(b'ABC')
|
||||
(tmp_path / 'link').symlink_to(tmp_path / 'regular_file_a')
|
||||
helpers.safe_symlink(tmp_path / 'regular_file_b', tmp_path / 'link')
|
||||
assert (tmp_path / 'link').samefile(tmp_path / 'regular_file_b') or (
|
||||
tmp_path / 'link'
|
||||
).read_bytes() == b'ABC'
|
||||
|
||||
|
||||
def test_no_cpu_count(monkeypatch):
|
||||
def cpu_count_raises():
|
||||
raise NotImplementedError()
|
||||
|
||||
monkeypatch.setattr(multiprocessing, 'cpu_count', cpu_count_raises)
|
||||
with pytest.warns(expected_warning=UserWarning):
|
||||
assert helpers.available_cpu_count() == 1
|
||||
|
||||
|
||||
def test_deprecated():
|
||||
@helpers.deprecated
|
||||
def old_function():
|
||||
return 42
|
||||
|
||||
with pytest.deprecated_call():
|
||||
assert old_function() == 42
|
||||
|
||||
|
||||
class TestFileIsWritable:
|
||||
@pytest.fixture
|
||||
def non_existent(self, tmp_path):
|
||||
return tmp_path / 'nofile'
|
||||
|
||||
@pytest.fixture
|
||||
def basic_file(self, tmp_path):
|
||||
basic = tmp_path / 'basic'
|
||||
basic.touch()
|
||||
return basic
|
||||
|
||||
def test_plain(self, non_existent):
|
||||
assert helpers.is_file_writable(non_existent)
|
||||
|
||||
def test_symlink_loop(self, tmp_path):
|
||||
loop = tmp_path / 'loop'
|
||||
loop.symlink_to(loop)
|
||||
assert not helpers.is_file_writable(loop)
|
||||
|
||||
def test_chmod(self, basic_file):
|
||||
assert helpers.is_file_writable(basic_file)
|
||||
basic_file.chmod(0o400)
|
||||
assert not helpers.is_file_writable(basic_file)
|
||||
basic_file.chmod(0o000)
|
||||
assert not helpers.is_file_writable(basic_file)
|
||||
|
||||
def test_permission_error(self, basic_file):
|
||||
pathmock = MagicMock(spec_set=basic_file)
|
||||
pathmock.is_symlink.return_value = False
|
||||
pathmock.exists.return_value = True
|
||||
pathmock.is_file.side_effect = PermissionError
|
||||
assert not helpers.is_file_writable(pathmock)
|
||||
@@ -0,0 +1,92 @@
|
||||
# © 2019 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# This file is part of OCRmyPDF.
|
||||
#
|
||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||
# it under the terms of the GNU General Public License as published by
|
||||
# the Free Software Foundation, either version 3 of the License, or
|
||||
# (at your option) any later version.
|
||||
#
|
||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||
# GNU General Public License for more details.
|
||||
#
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
import img2pdf
|
||||
import pikepdf
|
||||
import pytest
|
||||
from PIL import Image
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
check_ocrmypdf = pytest.helpers.check_ocrmypdf
|
||||
run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def baiona(resources):
|
||||
return Image.open(resources / 'baiona_gray.png')
|
||||
|
||||
|
||||
def test_image_to_pdf(spoof_tesseract_noop, resources, outpdf):
|
||||
check_ocrmypdf(
|
||||
resources / 'crom.png', outpdf, '--image-dpi', '200', env=spoof_tesseract_noop
|
||||
)
|
||||
|
||||
|
||||
def test_no_dpi_info(caplog, baiona, outdir, no_outpdf):
|
||||
im = baiona
|
||||
assert 'dpi' not in im.info
|
||||
input_image = outdir / 'baiona_no_dpi.png'
|
||||
im.save(input_image)
|
||||
|
||||
rc = run_ocrmypdf_api(input_image, no_outpdf)
|
||||
assert rc == ocrmypdf.ExitCode.input_file
|
||||
assert "--image-dpi" in caplog.text
|
||||
|
||||
|
||||
def test_dpi_not_credible(caplog, baiona, outdir, no_outpdf):
|
||||
im = baiona
|
||||
assert 'dpi' not in im.info
|
||||
input_image = outdir / 'baiona_no_dpi.png'
|
||||
im.save(input_image, dpi=(30, 30))
|
||||
|
||||
rc = run_ocrmypdf_api(input_image, no_outpdf)
|
||||
assert rc == ocrmypdf.ExitCode.input_file
|
||||
assert "not credible" in caplog.text
|
||||
|
||||
|
||||
def test_cmyk_no_icc(caplog, resources, no_outpdf):
|
||||
rc = run_ocrmypdf_api(resources / 'baiona_cmyk.jpg', no_outpdf)
|
||||
assert rc == ocrmypdf.ExitCode.input_file
|
||||
assert "no ICC profile" in caplog.text
|
||||
|
||||
|
||||
def test_img2pdf_fails(resources, no_outpdf):
|
||||
with patch(
|
||||
'ocrmypdf._pipeline.img2pdf.convert', side_effect=img2pdf.ImageOpenError()
|
||||
):
|
||||
rc = run_ocrmypdf_api(
|
||||
resources / 'baiona_gray.png', no_outpdf, '--image-dpi', '200'
|
||||
)
|
||||
assert rc == ocrmypdf.ExitCode.input_file
|
||||
|
||||
|
||||
def test_jpeg_in_jpeg_out(resources, outpdf, spoof_tesseract_noop):
|
||||
check_ocrmypdf(
|
||||
resources / 'congress.jpg',
|
||||
outpdf,
|
||||
'--image-dpi',
|
||||
'100',
|
||||
'--output-type',
|
||||
'pdf', # specifically check pdf because Ghostscript may convert to JPEG
|
||||
'--remove-background',
|
||||
env=spoof_tesseract_noop,
|
||||
)
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert next(pdf.pages[0].images.values()).Filter == pikepdf.Name.DCTDecode
|
||||
+8
-14
@@ -16,8 +16,8 @@
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
|
||||
from os import fspath
|
||||
import os
|
||||
from os import fspath
|
||||
from pickle import dumps, loads
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -63,6 +63,10 @@ def test_pix_otsu(crom_pix):
|
||||
assert im1bpp.mode == '1'
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
lept.get_leptonica_version() < 'leptonica-1.76',
|
||||
reason="needs new leptonica for API change",
|
||||
)
|
||||
def test_crop(resources):
|
||||
pix = lept.Pix.open(resources / 'linn.png')
|
||||
foreground = pix.crop_to_foreground()
|
||||
@@ -90,16 +94,6 @@ def test_leptonica_compile(tmp_path):
|
||||
ffibuilder.compile(tmpdir=fspath(tmp_path), target=fspath(tmp_path / 'lepttest.*'))
|
||||
|
||||
|
||||
def test_with_stderr(capsys):
|
||||
# pytest redirects stderr too; we must disable this for the test to be valid
|
||||
with capsys.disabled():
|
||||
with pytest.raises(FileNotFoundError):
|
||||
lept.Pix.open("does_not_exist1")
|
||||
|
||||
|
||||
def test_without_stderr(capsys):
|
||||
# pytest redirects stderr too; we must disable this for the test to be valid
|
||||
with capsys.disabled():
|
||||
with patch('sys.stderr', new=None):
|
||||
with pytest.raises(FileNotFoundError):
|
||||
lept.Pix.open("does_not_exist2")
|
||||
def test_file_not_found():
|
||||
with pytest.raises(FileNotFoundError):
|
||||
lept.Pix.open("does_not_exist1")
|
||||
|
||||
+92
-294
@@ -15,22 +15,21 @@
|
||||
# You should have received a copy of the GNU General Public License
|
||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
from math import isclose
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, run
|
||||
from unittest.mock import patch
|
||||
|
||||
import pikepdf
|
||||
import PIL
|
||||
import pytest
|
||||
from PIL import Image
|
||||
|
||||
import ocrmypdf
|
||||
import pikepdf
|
||||
from ocrmypdf.exceptions import ExitCode, MissingDependencyError
|
||||
from ocrmypdf.exec import ghostscript, qpdf, tesseract
|
||||
from ocrmypdf.leptonica import Pix
|
||||
from ocrmypdf.pdfa import file_claims_pdfa
|
||||
from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
||||
|
||||
@@ -39,144 +38,27 @@ from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
||||
|
||||
check_ocrmypdf = pytest.helpers.check_ocrmypdf
|
||||
run_ocrmypdf = pytest.helpers.run_ocrmypdf
|
||||
run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api
|
||||
spoof = pytest.helpers.spoof
|
||||
|
||||
|
||||
RENDERERS = ['hocr', 'sandwich']
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
@pytest.fixture
|
||||
def spoof_tesseract_crash(tmp_path_factory):
|
||||
return spoof(tmp_path_factory, tesseract='tesseract_crash.py')
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
@pytest.fixture
|
||||
def spoof_tesseract_big_image_error(tmp_path_factory):
|
||||
return spoof(tmp_path_factory, tesseract='tesseract_big_image_error.py')
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
def spoof_no_tess_no_pdfa(tmp_path_factory):
|
||||
return spoof(
|
||||
tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_pdfa_failure.py'
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
def spoof_no_tess_pdfa_warning(tmp_path_factory):
|
||||
return spoof(
|
||||
tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_feature_elision.py'
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
def spoof_no_tess_gs_render_fail(tmp_path_factory):
|
||||
return spoof(
|
||||
tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_render_failure.py'
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(scope='session')
|
||||
def spoof_no_tess_gs_raster_fail(tmp_path_factory):
|
||||
return spoof(
|
||||
tmp_path_factory, tesseract='tesseract_noop.py', gs='gs_raster_failure.py'
|
||||
)
|
||||
|
||||
|
||||
def test_quick(spoof_tesseract_cache, resources, outpdf):
|
||||
check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_tesseract_cache)
|
||||
|
||||
|
||||
def test_deskew(spoof_tesseract_noop, resources, outdir):
|
||||
# Run with deskew
|
||||
deskewed_pdf = check_ocrmypdf(
|
||||
resources / 'skew.pdf', outdir / 'skew.pdf', '-d', env=spoof_tesseract_noop
|
||||
)
|
||||
|
||||
# Now render as an image again and use Leptonica to find the skew angle
|
||||
# to confirm that it was deskewed
|
||||
log = logging.getLogger()
|
||||
|
||||
deskewed_png = outdir / 'deskewed.png'
|
||||
|
||||
ghostscript.rasterize_pdf(
|
||||
deskewed_pdf,
|
||||
deskewed_png,
|
||||
xres=150,
|
||||
yres=150,
|
||||
raster_device='pngmono',
|
||||
log=log,
|
||||
pageno=1,
|
||||
)
|
||||
|
||||
pix = Pix.open(deskewed_png)
|
||||
skew_angle, _skew_confidence = pix.find_skew()
|
||||
|
||||
print(skew_angle)
|
||||
assert -0.5 < skew_angle < 0.5, "Deskewing failed"
|
||||
|
||||
|
||||
def test_remove_background(spoof_tesseract_noop, resources, outdir):
|
||||
# Ensure the input image does not contain pure white/black
|
||||
with Image.open(resources / 'congress.jpg') as im:
|
||||
assert im.getextrema() != ((0, 255), (0, 255), (0, 255))
|
||||
|
||||
output_pdf = check_ocrmypdf(
|
||||
resources / 'congress.jpg',
|
||||
outdir / 'test_remove_bg.pdf',
|
||||
'--remove-background',
|
||||
'--image-dpi',
|
||||
'150',
|
||||
env=spoof_tesseract_noop,
|
||||
)
|
||||
|
||||
log = logging.getLogger()
|
||||
|
||||
output_png = outdir / 'remove_bg.png'
|
||||
|
||||
ghostscript.rasterize_pdf(
|
||||
output_pdf,
|
||||
output_png,
|
||||
xres=100,
|
||||
yres=100,
|
||||
raster_device='png16m',
|
||||
log=log,
|
||||
pageno=1,
|
||||
)
|
||||
|
||||
# The output image should contain pure white and black
|
||||
with Image.open(output_png) as im:
|
||||
assert im.getextrema() == ((0, 255), (0, 255), (0, 255))
|
||||
|
||||
|
||||
# This will run 5 * 2 * 2 = 20 test cases
|
||||
@pytest.mark.parametrize(
|
||||
"pdf", ['palette.pdf', 'cmyk.pdf', 'ccitt.pdf', 'jbig2.pdf', 'lichtenstein.pdf']
|
||||
)
|
||||
@pytest.mark.parametrize("renderer", ['sandwich', 'hocr'])
|
||||
@pytest.mark.parametrize("output_type", ['pdf', 'pdfa'])
|
||||
def test_exotic_image(
|
||||
spoof_tesseract_cache, pdf, renderer, output_type, resources, outdir
|
||||
):
|
||||
outfile = outdir / f'test_{pdf}_{renderer}.pdf'
|
||||
check_ocrmypdf(
|
||||
resources / pdf,
|
||||
outfile,
|
||||
'-dc' if pytest.helpers.have_unpaper() else '-d',
|
||||
'-v',
|
||||
'1',
|
||||
'--output-type',
|
||||
output_type,
|
||||
'--sidecar',
|
||||
'--skip-text',
|
||||
'--pdf-renderer',
|
||||
renderer,
|
||||
env=spoof_tesseract_cache,
|
||||
)
|
||||
|
||||
assert outfile.with_suffix('.pdf.txt').exists()
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', RENDERERS)
|
||||
def test_oversample(spoof_tesseract_cache, renderer, resources, outpdf):
|
||||
oversampled_pdf = check_ocrmypdf(
|
||||
@@ -197,8 +79,8 @@ def test_oversample(spoof_tesseract_cache, renderer, resources, outpdf):
|
||||
|
||||
|
||||
def test_repeat_ocr(resources, no_outpdf):
|
||||
p, _, _ = run_ocrmypdf(resources / 'graph_ocred.pdf', no_outpdf)
|
||||
assert p.returncode != 0
|
||||
result = run_ocrmypdf_api(resources / 'graph_ocred.pdf', no_outpdf)
|
||||
assert result == ExitCode.already_done_ocr
|
||||
|
||||
|
||||
def test_force_ocr(spoof_tesseract_cache, resources, outpdf):
|
||||
@@ -217,7 +99,7 @@ def test_skip_ocr(spoof_tesseract_cache, resources, outpdf):
|
||||
assert pdfinfo[0].has_text
|
||||
|
||||
|
||||
def test_redo_ocr(spoof_tesseract_cache, resources, outpdf):
|
||||
def test_redo_ocr(resources, outpdf):
|
||||
in_ = resources / 'graph_ocred.pdf'
|
||||
before = PdfInfo(in_, detailed_page_analysis=True)
|
||||
out = outpdf
|
||||
@@ -300,34 +182,34 @@ def test_maximum_options(
|
||||
)
|
||||
|
||||
|
||||
def test_tesseract_missing_tessdata(resources, no_outpdf):
|
||||
def test_tesseract_missing_tessdata(resources, no_outpdf, tmpdir):
|
||||
env = os.environ.copy()
|
||||
env['TESSDATA_PREFIX'] = '/tmp'
|
||||
env['TESSDATA_PREFIX'] = os.fspath(tmpdir)
|
||||
|
||||
p, _, err = run_ocrmypdf(
|
||||
resources / 'graph_ocred.pdf', no_outpdf, '-v', '1', '--skip-text', env=env
|
||||
returncode = run_ocrmypdf_api(
|
||||
resources / 'graph.pdf', no_outpdf, '-v', '1', '--skip-text', env=env
|
||||
)
|
||||
assert p.returncode == ExitCode.missing_dependency, err
|
||||
assert returncode == ExitCode.missing_dependency
|
||||
|
||||
|
||||
def test_invalid_input_pdf(resources, no_outpdf):
|
||||
p, out, err = run_ocrmypdf(resources / 'invalid.pdf', no_outpdf)
|
||||
assert p.returncode == ExitCode.input_file, err
|
||||
result = run_ocrmypdf_api(resources / 'invalid.pdf', no_outpdf)
|
||||
assert result == ExitCode.input_file
|
||||
|
||||
|
||||
def test_blank_input_pdf(resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(resources / 'blank.pdf', outpdf)
|
||||
assert p.returncode == ExitCode.ok
|
||||
result = run_ocrmypdf_api(resources / 'blank.pdf', outpdf)
|
||||
assert result == ExitCode.ok
|
||||
|
||||
|
||||
def test_force_ocr_on_pdf_with_no_images(spoof_tesseract_crash, resources, no_outpdf):
|
||||
# As a correctness test, make sure that --force-ocr on a PDF with no
|
||||
# content still triggers tesseract. If tesseract crashes, then it was
|
||||
# called.
|
||||
p, _, err = run_ocrmypdf(
|
||||
p, _, _ = run_ocrmypdf(
|
||||
resources / 'blank.pdf', no_outpdf, '--force-ocr', env=spoof_tesseract_crash
|
||||
)
|
||||
assert p.returncode == ExitCode.child_process_error, err
|
||||
assert p.returncode == ExitCode.child_process_error
|
||||
assert not os.path.exists(no_outpdf)
|
||||
|
||||
|
||||
@@ -340,27 +222,29 @@ def test_german(spoof_tesseract_cache, resources, outdir):
|
||||
# properly. It is fine that we are testing -l deu on a French file because
|
||||
# we are exercising the functionality not going for accuracy.
|
||||
sidecar = outdir / 'francais.txt'
|
||||
p, out, err = run_ocrmypdf(
|
||||
resources / 'francais.pdf',
|
||||
outdir / 'francais.pdf',
|
||||
'-l',
|
||||
'deu', # more commonly installed
|
||||
'--sidecar',
|
||||
sidecar,
|
||||
env=spoof_tesseract_cache,
|
||||
)
|
||||
if 'deu' not in tesseract.languages():
|
||||
pytest.xfail(reason="tesseract-deu language pack not installed")
|
||||
assert p.returncode == ExitCode.ok, "Requires tesseract deu language pack"
|
||||
try:
|
||||
check_ocrmypdf(
|
||||
resources / 'francais.pdf',
|
||||
outdir / 'francais.pdf',
|
||||
'-l',
|
||||
'deu', # more commonly installed
|
||||
'--sidecar',
|
||||
sidecar,
|
||||
env=spoof_tesseract_cache,
|
||||
)
|
||||
except MissingDependencyError:
|
||||
if 'deu' not in tesseract.languages():
|
||||
pytest.xfail(reason="tesseract-deu language pack not installed")
|
||||
raise
|
||||
|
||||
|
||||
def test_klingon(resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(resources / 'francais.pdf', outpdf, '-l', 'klz')
|
||||
p, _, _ = run_ocrmypdf(resources / 'francais.pdf', outpdf, '-l', 'klz')
|
||||
assert p.returncode == ExitCode.missing_dependency
|
||||
|
||||
|
||||
def test_missing_docinfo(spoof_tesseract_noop, resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(
|
||||
result = run_ocrmypdf_api(
|
||||
resources / 'missing_docinfo.pdf',
|
||||
outpdf,
|
||||
'-l',
|
||||
@@ -368,7 +252,7 @@ def test_missing_docinfo(spoof_tesseract_noop, resources, outpdf):
|
||||
'--skip-text',
|
||||
env=spoof_tesseract_noop,
|
||||
)
|
||||
assert p.returncode == ExitCode.ok, err
|
||||
assert result == ExitCode.ok
|
||||
|
||||
|
||||
def test_uppercase_extension(spoof_tesseract_noop, resources, outdir):
|
||||
@@ -379,24 +263,35 @@ def test_uppercase_extension(spoof_tesseract_noop, resources, outdir):
|
||||
)
|
||||
|
||||
|
||||
def test_input_file_not_found(no_outpdf):
|
||||
def test_input_file_not_found(caplog, no_outpdf):
|
||||
input_file = "does not exist.pdf"
|
||||
p, out, err = run_ocrmypdf(input_file, no_outpdf)
|
||||
assert p.returncode == ExitCode.input_file
|
||||
assert input_file in out or input_file in err
|
||||
result = run_ocrmypdf_api(input_file, no_outpdf)
|
||||
assert result == ExitCode.input_file
|
||||
assert input_file in caplog.text
|
||||
|
||||
|
||||
def test_input_file_not_a_pdf(no_outpdf):
|
||||
@pytest.mark.skipif(os.name == 'nt', reason="chmod")
|
||||
def test_input_file_not_readable(caplog, resources, outdir, no_outpdf):
|
||||
input_file = outdir / 'trivial.pdf'
|
||||
shutil.copy(resources / 'trivial.pdf', input_file)
|
||||
input_file.chmod(0o000)
|
||||
result = run_ocrmypdf_api(input_file, no_outpdf)
|
||||
assert result == ExitCode.input_file
|
||||
assert str(input_file) in caplog.text
|
||||
|
||||
|
||||
def test_input_file_not_a_pdf(caplog, no_outpdf):
|
||||
input_file = __file__ # Try to OCR this file
|
||||
p, out, err = run_ocrmypdf(input_file, no_outpdf)
|
||||
assert p.returncode == ExitCode.input_file
|
||||
assert input_file in out or input_file in err
|
||||
result = run_ocrmypdf_api(input_file, no_outpdf)
|
||||
assert result == ExitCode.input_file
|
||||
if os.name != 'nt': # name will be mangled with \\'s on nt
|
||||
assert input_file in caplog.text
|
||||
|
||||
|
||||
def test_encrypted(resources, no_outpdf):
|
||||
p, out, err = run_ocrmypdf(resources / 'skew-encrypted.pdf', no_outpdf)
|
||||
assert p.returncode == ExitCode.encrypted_pdf
|
||||
assert out.find('encrypted')
|
||||
def test_encrypted(resources, caplog, no_outpdf):
|
||||
result = run_ocrmypdf_api(resources / 'skew-encrypted.pdf', no_outpdf)
|
||||
assert result == ExitCode.encrypted_pdf
|
||||
assert 'encryption must be removed' in caplog.text
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', RENDERERS)
|
||||
@@ -415,8 +310,8 @@ def test_pagesegmode(renderer, spoof_tesseract_cache, resources, outpdf):
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', RENDERERS)
|
||||
def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf):
|
||||
p, out, err = run_ocrmypdf(
|
||||
def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf, caplog):
|
||||
p, _, err = run_ocrmypdf(
|
||||
resources / 'ccitt.pdf',
|
||||
no_outpdf,
|
||||
'-v',
|
||||
@@ -427,7 +322,7 @@ def test_tesseract_crash(renderer, spoof_tesseract_crash, resources, no_outpdf):
|
||||
)
|
||||
assert p.returncode == ExitCode.child_process_error
|
||||
assert not os.path.exists(no_outpdf)
|
||||
assert "ERROR" in err
|
||||
assert "SubprocessOutputError" in err
|
||||
|
||||
|
||||
def test_tesseract_crash_autorotate(spoof_tesseract_crash, resources, no_outpdf):
|
||||
@@ -465,70 +360,6 @@ def test_algo4(resources, spoof_tesseract_noop, outpdf):
|
||||
assert p.returncode == ExitCode.encrypted_pdf
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', RENDERERS)
|
||||
def test_non_square_resolution(renderer, spoof_tesseract_cache, resources, outpdf):
|
||||
# Confirm input image is non-square resolution
|
||||
in_pageinfo = PdfInfo(resources / 'aspect.pdf')
|
||||
assert in_pageinfo[0].xres != in_pageinfo[0].yres
|
||||
|
||||
check_ocrmypdf(
|
||||
resources / 'aspect.pdf',
|
||||
outpdf,
|
||||
'--pdf-renderer',
|
||||
renderer,
|
||||
env=spoof_tesseract_cache,
|
||||
)
|
||||
|
||||
out_pageinfo = PdfInfo(outpdf)
|
||||
|
||||
# Confirm resolution was kept the same
|
||||
assert in_pageinfo[0].xres == out_pageinfo[0].xres
|
||||
assert in_pageinfo[0].yres == out_pageinfo[0].yres
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', RENDERERS)
|
||||
def test_convert_to_square_resolution(
|
||||
renderer, spoof_tesseract_cache, resources, outpdf
|
||||
):
|
||||
# Confirm input image is non-square resolution
|
||||
in_pageinfo = PdfInfo(resources / 'aspect.pdf')
|
||||
assert in_pageinfo[0].xres != in_pageinfo[0].yres
|
||||
|
||||
# --force-ocr requires means forced conversion to square resolution
|
||||
check_ocrmypdf(
|
||||
resources / 'aspect.pdf',
|
||||
outpdf,
|
||||
'--force-ocr',
|
||||
'--pdf-renderer',
|
||||
renderer,
|
||||
env=spoof_tesseract_cache,
|
||||
)
|
||||
|
||||
out_pageinfo = PdfInfo(outpdf)
|
||||
|
||||
in_p0, out_p0 = in_pageinfo[0], out_pageinfo[0]
|
||||
|
||||
# Resolution show now be equal
|
||||
assert out_p0.xres == out_p0.yres
|
||||
|
||||
# Page size should match input page size
|
||||
assert isclose(in_p0.width_inches, out_p0.width_inches)
|
||||
assert isclose(in_p0.height_inches, out_p0.height_inches)
|
||||
|
||||
# Because we rasterized the page to produce a new image, it should occupy
|
||||
# the entire page
|
||||
out_im_w = out_p0.images[0].width / out_p0.images[0].xres
|
||||
out_im_h = out_p0.images[0].height / out_p0.images[0].yres
|
||||
assert isclose(out_p0.width_inches, out_im_w)
|
||||
assert isclose(out_p0.height_inches, out_im_h)
|
||||
|
||||
|
||||
def test_image_to_pdf(spoof_tesseract_noop, resources, outpdf):
|
||||
check_ocrmypdf(
|
||||
resources / 'crom.png', outpdf, '--image-dpi', '200', env=spoof_tesseract_noop
|
||||
)
|
||||
|
||||
|
||||
def test_jbig2_passthrough(spoof_tesseract_cache, resources, outpdf):
|
||||
out = check_ocrmypdf(
|
||||
resources / 'jbig2.pdf',
|
||||
@@ -556,19 +387,6 @@ def test_linearized_pdf_and_indirect_object(spoof_tesseract_noop, resources, out
|
||||
check_ocrmypdf(resources / 'epson.pdf', outpdf, env=spoof_tesseract_noop)
|
||||
|
||||
|
||||
def test_ghostscript_pdfa_failure(spoof_no_tess_no_pdfa, resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(
|
||||
resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_no_pdfa
|
||||
)
|
||||
assert (
|
||||
p.returncode == ExitCode.pdfa_conversion_failed
|
||||
), "Unexpected return when PDF/A fails"
|
||||
|
||||
|
||||
def test_ghostscript_feature_elision(spoof_no_tess_pdfa_warning, resources, outpdf):
|
||||
check_ocrmypdf(resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_pdfa_warning)
|
||||
|
||||
|
||||
def test_very_high_dpi(spoof_tesseract_cache, resources, outpdf):
|
||||
"Checks for a Decimal quantize error with high DPI, etc"
|
||||
check_ocrmypdf(resources / '2400dpi.pdf', outpdf, env=spoof_tesseract_cache)
|
||||
@@ -586,7 +404,7 @@ def test_overlay(spoof_tesseract_noop, resources, outpdf):
|
||||
|
||||
|
||||
def test_destination_not_writable(spoof_tesseract_noop, resources, outdir):
|
||||
if os.getuid() == 0 or os.geteuid() == 0:
|
||||
if os.name != 'nt' and (os.getuid() == 0 or os.geteuid() == 0):
|
||||
pytest.xfail(reason="root can write to anything")
|
||||
protected_file = outdir / 'protected.pdf'
|
||||
protected_file.touch()
|
||||
@@ -609,27 +427,16 @@ language_model_penalty_non_freq_dict_word 0
|
||||
)
|
||||
|
||||
check_ocrmypdf(
|
||||
resources / 'ccitt.pdf', outdir / 'out.pdf', '--tesseract-config', cfg_file
|
||||
resources / '3small.pdf',
|
||||
outdir / 'out.pdf',
|
||||
'--tesseract-config',
|
||||
cfg_file,
|
||||
'--pages',
|
||||
'1',
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.slow # This test sometimes times out in CI
|
||||
@pytest.mark.parametrize('renderer', RENDERERS)
|
||||
def test_tesseract_config_notfound(renderer, resources, outdir):
|
||||
cfg_file = outdir / 'nofile.cfg'
|
||||
|
||||
p, out, err = run_ocrmypdf(
|
||||
resources / 'ccitt.pdf',
|
||||
outdir / 'out.pdf',
|
||||
'--pdf-renderer',
|
||||
renderer,
|
||||
'--tesseract-config',
|
||||
cfg_file,
|
||||
)
|
||||
assert "Can't open" in err, "No error message about missing config file"
|
||||
assert p.returncode == ExitCode.ok, err
|
||||
|
||||
|
||||
@pytest.mark.parametrize('renderer', RENDERERS)
|
||||
def test_tesseract_config_invalid(renderer, resources, outdir):
|
||||
cfg_file = outdir / 'test.cfg'
|
||||
@@ -648,7 +455,10 @@ THIS FILE IS INVALID
|
||||
'--tesseract-config',
|
||||
cfg_file,
|
||||
)
|
||||
assert "parameter not found" in err.lower(), "No error message"
|
||||
assert (
|
||||
"parameter not found" in err.lower()
|
||||
or "error occurred while parsing" in err.lower()
|
||||
), "No error message"
|
||||
assert p.returncode == ExitCode.invalid_config
|
||||
|
||||
|
||||
@@ -684,7 +494,7 @@ def test_pagesize_consistency(renderer, resources, outpdf):
|
||||
|
||||
first_page_dimensions = pytest.helpers.first_page_dimensions
|
||||
|
||||
infile = resources / 'linn.pdf'
|
||||
infile = resources / '3small.pdf'
|
||||
|
||||
before_dims = first_page_dimensions(infile)
|
||||
|
||||
@@ -697,12 +507,14 @@ def test_pagesize_consistency(renderer, resources, outpdf):
|
||||
'--deskew',
|
||||
'--remove-background',
|
||||
'--clean-final' if pytest.helpers.have_unpaper() else None,
|
||||
'--pages',
|
||||
'1',
|
||||
)
|
||||
|
||||
after_dims = first_page_dimensions(outpdf)
|
||||
|
||||
assert isclose(before_dims[0], after_dims[0])
|
||||
assert isclose(before_dims[1], after_dims[1])
|
||||
assert isclose(before_dims[0], after_dims[0], rel_tol=1e-4)
|
||||
assert isclose(before_dims[1], after_dims[1], rel_tol=1e-4)
|
||||
|
||||
|
||||
def test_skip_big_with_no_images(spoof_tesseract_noop, resources, outpdf):
|
||||
@@ -716,22 +528,6 @@ def test_skip_big_with_no_images(spoof_tesseract_noop, resources, outpdf):
|
||||
)
|
||||
|
||||
|
||||
def test_gs_render_failure(spoof_no_tess_gs_render_fail, resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(
|
||||
resources / 'blank.pdf', outpdf, env=spoof_no_tess_gs_render_fail
|
||||
)
|
||||
print(err)
|
||||
assert p.returncode == ExitCode.child_process_error
|
||||
|
||||
|
||||
def test_gs_raster_failure(spoof_no_tess_gs_raster_fail, resources, outpdf):
|
||||
p, out, err = run_ocrmypdf(
|
||||
resources / 'ccitt.pdf', outpdf, env=spoof_no_tess_gs_raster_fail
|
||||
)
|
||||
print(err)
|
||||
assert p.returncode == ExitCode.child_process_error
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
'8.0.0' <= qpdf.version() <= '8.0.1',
|
||||
reason="qpdf regression on pages with no contents",
|
||||
@@ -860,7 +656,7 @@ def test_compression_changed(
|
||||
def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf):
|
||||
sidecar = outpdf.with_suffix('.txt')
|
||||
check_ocrmypdf(
|
||||
resources / 'multipage.pdf',
|
||||
resources / '3small.pdf',
|
||||
outpdf,
|
||||
'--skip-text',
|
||||
'--sidecar',
|
||||
@@ -868,10 +664,10 @@ def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf):
|
||||
env=spoof_tesseract_cache,
|
||||
)
|
||||
|
||||
pdfinfo = PdfInfo(resources / 'multipage.pdf')
|
||||
pdfinfo = PdfInfo(resources / '3small.pdf')
|
||||
num_pages = len(pdfinfo)
|
||||
|
||||
with open(sidecar, 'r') as f:
|
||||
with open(sidecar, 'r', encoding='utf-8') as f:
|
||||
ocr_text = f.read()
|
||||
|
||||
# There should a formfeed between each pair of pages, so the count of
|
||||
@@ -887,7 +683,7 @@ def test_sidecar_nonempty(spoof_tesseract_cache, resources, outpdf):
|
||||
resources / 'ccitt.pdf', outpdf, '--sidecar', sidecar, env=spoof_tesseract_cache
|
||||
)
|
||||
|
||||
with open(sidecar, 'r') as f:
|
||||
with open(sidecar, 'r', encoding='utf-8') as f:
|
||||
ocr_text = f.read()
|
||||
assert 'the' in ocr_text
|
||||
|
||||
@@ -924,17 +720,18 @@ def test_decompression_bomb(resources, outpdf):
|
||||
|
||||
|
||||
def test_text_curves(spoof_tesseract_noop, resources, outpdf):
|
||||
check_ocrmypdf(resources / 'vector.pdf', outpdf, env=spoof_tesseract_noop)
|
||||
with patch('ocrmypdf._pipeline.VECTOR_PAGE_DPI', 100):
|
||||
check_ocrmypdf(resources / 'vector.pdf', outpdf, env=spoof_tesseract_noop)
|
||||
|
||||
info = PdfInfo(outpdf)
|
||||
assert len(info.pages[0].images) == 0, "added images to the vector PDF"
|
||||
info = PdfInfo(outpdf)
|
||||
assert len(info.pages[0].images) == 0, "added images to the vector PDF"
|
||||
|
||||
check_ocrmypdf(
|
||||
resources / 'vector.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop
|
||||
)
|
||||
check_ocrmypdf(
|
||||
resources / 'vector.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop
|
||||
)
|
||||
|
||||
info = PdfInfo(outpdf)
|
||||
assert len(info.pages[0].images) != 0, "force did not rasterize"
|
||||
info = PdfInfo(outpdf)
|
||||
assert len(info.pages[0].images) != 0, "force did not rasterize"
|
||||
|
||||
|
||||
def test_output_is_dir(spoof_tesseract_noop, resources, outdir):
|
||||
@@ -945,6 +742,7 @@ def test_output_is_dir(spoof_tesseract_noop, resources, outdir):
|
||||
assert 'is not a writable file' in err
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.name == 'nt', reason="symlink needs admin permissions")
|
||||
def test_output_is_symlink(spoof_tesseract_noop, resources, outdir):
|
||||
sym = Path(outdir / 'this_is_a_symlink')
|
||||
sym.symlink_to(outdir / 'out.pdf')
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user