Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6a302fdb88 | ||
|
|
1d9cc239ee | ||
|
|
d240fc1ea6 | ||
|
|
e7d21dd826 | ||
|
|
e774b4650b | ||
|
|
8f8e6dcdd4 | ||
|
|
5252b88f0f | ||
|
|
ea69883386 | ||
|
|
eb343b1e37 | ||
|
|
9f02de55be | ||
|
|
7394a4cf49 | ||
|
|
ed9fb110b1 | ||
|
|
4650074428 | ||
|
|
70aa644c10 | ||
|
|
2ccb3edc58 | ||
|
|
1f40a70554 | ||
|
|
e14ffbf03f | ||
|
|
25a1dde57c | ||
|
|
1d10eac764 | ||
|
|
3f868118cd |
@@ -2,11 +2,4 @@ Please include the command line and a test file with your issue report.
|
|||||||
|
|
||||||
If possible, please use a test file that we can include in future test cases (no personal information, no copyrighted material).
|
If possible, please use a test file that we can include in future test cases (no personal information, no copyrighted material).
|
||||||
|
|
||||||
If you wish to encrypt a test file so that only the maintainer of OCRmyPDF can view it, you may use:
|
If you wish to encrypt a test file for the OCRmyPDF maintainer only, see the [Wiki](https://github.com/jbarlow83/OCRmyPDF/wiki).
|
||||||
|
|
||||||
```bash
|
|
||||||
|
|
||||||
gpg --recv-keys 4434eb74c4a35f7f --keyserver pgp.mit.edu
|
|
||||||
gpg --output test.pdf.gpg --encrypt --receipient 4434eb74c4a35f7f test.pdf
|
|
||||||
|
|
||||||
```
|
|
||||||
|
|||||||
@@ -2,12 +2,14 @@
|
|||||||
*.pyc
|
*.pyc
|
||||||
*.sublime-*
|
*.sublime-*
|
||||||
venv*/
|
venv*/
|
||||||
|
.venv/
|
||||||
pyvenv.cfg
|
pyvenv.cfg
|
||||||
tasks.py
|
tasks.py
|
||||||
.bash_history
|
.bash_history
|
||||||
.ruffus_history.sqlite
|
.ruffus_history.sqlite
|
||||||
.idea/
|
.idea/
|
||||||
.pytest_cache/
|
.pytest_cache/
|
||||||
|
.pylintrc
|
||||||
|
|
||||||
# Package building
|
# Package building
|
||||||
*.egg-info/
|
*.egg-info/
|
||||||
@@ -42,3 +44,4 @@ pdfbox-app*.jar
|
|||||||
.vscode/
|
.vscode/
|
||||||
IDEAS
|
IDEAS
|
||||||
_Dockerfile.local
|
_Dockerfile.local
|
||||||
|
/scratch.py
|
||||||
|
|||||||
+31
-36
@@ -1,32 +1,51 @@
|
|||||||
dist: trusty
|
dist: trusty
|
||||||
language: python
|
language: python
|
||||||
cache:
|
cache:
|
||||||
ccache: true
|
|
||||||
pip: true
|
pip: true
|
||||||
directories:
|
directories:
|
||||||
- $HOME/Library/Caches/Homebrew
|
- $HOME/Library/Caches/Homebrew
|
||||||
|
|
||||||
env:
|
addons:
|
||||||
global:
|
apt:
|
||||||
- secure: "hsf6MT+n2x3OiDM2fQyJZdV0/PWYmv81LdVqC6cfnHBE/8N3DloJRqQ7WfO14TxhiK9PEC7MpyCj0lSabUHEO7gSH6Vks6I1asoSkt8S9/bSMlhT4hei+pwVpeGEiU5xHVATNjY+D919VC3IFvc3XmjT74h/2SLhaZ+jhEmDggM=" # HOMEBREW_OCRMYPDF_TOKEN
|
update: true
|
||||||
|
sources:
|
||||||
|
- sourceline: 'ppa:alex-p/tesseract-ocr'
|
||||||
|
- sourceline: 'ppa:heyarje/libav-11'
|
||||||
|
- sourceline: 'ppa:vshn/ghostscript'
|
||||||
|
packages:
|
||||||
|
- ghostscript
|
||||||
|
- libavcodec56
|
||||||
|
- libavformat56
|
||||||
|
- libavutil54
|
||||||
|
- libffi-dev
|
||||||
|
- poppler-utils
|
||||||
|
- qpdf
|
||||||
|
- tesseract-ocr
|
||||||
|
- tesseract-ocr-deu
|
||||||
|
- tesseract-ocr-eng
|
||||||
|
- tesseract-ocr-fra
|
||||||
|
|
||||||
matrix:
|
matrix:
|
||||||
include:
|
include:
|
||||||
- os: linux
|
- os: linux
|
||||||
sudo: required
|
sudo: required
|
||||||
language: python
|
language: python
|
||||||
python: 3.5
|
python: "3.5"
|
||||||
env: EXTRAS=
|
env: EXTRAS=
|
||||||
- os: linux
|
- os: linux
|
||||||
sudo: required
|
sudo: required
|
||||||
language: python
|
language: python
|
||||||
python: 3.6
|
python: "3.6"
|
||||||
env: EXTRAS=
|
env: EXTRAS=
|
||||||
- os: linux
|
- os: linux
|
||||||
sudo: required
|
sudo: required
|
||||||
language: python
|
language: python
|
||||||
python: 3.6
|
python: "3.6"
|
||||||
env: EXTRAS=[fitz]
|
env: EXTRAS=[fitz]
|
||||||
|
- os: linux
|
||||||
|
sudo: required
|
||||||
|
language: python
|
||||||
|
python: "3.7-dev"
|
||||||
- os: osx
|
- os: osx
|
||||||
osx_image: xcode8
|
osx_image: xcode8
|
||||||
language: generic
|
language: generic
|
||||||
@@ -41,7 +60,10 @@ before_cache:
|
|||||||
|
|
||||||
before_install: |
|
before_install: |
|
||||||
if [[ "$TRAVIS_OS_NAME" == "linux" ]]; then
|
if [[ "$TRAVIS_OS_NAME" == "linux" ]]; then
|
||||||
bash .travis/linux_before_install.sh
|
pip install --upgrade pip
|
||||||
|
mkdir -p packages
|
||||||
|
wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O packages/unpaper_6.1-1.deb
|
||||||
|
sudo dpkg -i packages/unpaper_6.1-1.deb
|
||||||
elif [[ "$TRAVIS_OS_NAME" == "osx" ]]; then
|
elif [[ "$TRAVIS_OS_NAME" == "osx" ]]; then
|
||||||
brew update && brew bundle --file=.travis/Brewfile
|
brew update && brew bundle --file=.travis/Brewfile
|
||||||
pip3 install --upgrade pip
|
pip3 install --upgrade pip
|
||||||
@@ -49,6 +71,7 @@ before_install: |
|
|||||||
fi
|
fi
|
||||||
|
|
||||||
install:
|
install:
|
||||||
|
- pip3 install pycparser # py3.7 workaround for https://github.com/eliben/pycparser/issues/251
|
||||||
- pip3 install ".$EXTRAS"
|
- pip3 install ".$EXTRAS"
|
||||||
- pip3 install -r test_requirements.txt
|
- pip3 install -r test_requirements.txt
|
||||||
|
|
||||||
@@ -73,31 +96,3 @@ deploy:
|
|||||||
tags: true
|
tags: true
|
||||||
condition: $TRAVIS_PYTHON_VERSION == "3.6" && $TRAVIS_OS_NAME == "linux" && $EXTRAS == ""
|
condition: $TRAVIS_PYTHON_VERSION == "3.6" && $TRAVIS_OS_NAME == "linux" && $EXTRAS == ""
|
||||||
skip_upload_docs: true
|
skip_upload_docs: true
|
||||||
|
|
||||||
# test pypi
|
|
||||||
- provider: pypi
|
|
||||||
server: https://testpypi.pypi.org/legacy/
|
|
||||||
user: ocrmypdf-travis
|
|
||||||
password:
|
|
||||||
secure: "DTFOmmNL6olA0+yXvp4u9jXZlZeqrJsJ0526jzqf4a3gZ6jnGTq5UI6WzRsslSyoMMfXKtHQebqHM6ogSgCZinyZ3ufHJo8fn9brxbEc2gsiWkbj5o3bGwdWMT1vNNE7XW0VCpw87rZ1EEwjl4FJHFudMlPR1yfU5+uq0k0PACo="
|
|
||||||
distributions: "sdist"
|
|
||||||
on:
|
|
||||||
branch: develop
|
|
||||||
tags: false
|
|
||||||
condition: $TRAVIS_OS_NAME == "osx"
|
|
||||||
skip_upload_docs: true
|
|
||||||
|
|
||||||
# null deploy for osx
|
|
||||||
# we really just want to run after_deploy *after* pypi upload is done, but
|
|
||||||
# after_deploy on runs if a given box deployed
|
|
||||||
- provider: script
|
|
||||||
script: /usr/bin/true
|
|
||||||
on:
|
|
||||||
branch: master
|
|
||||||
tags: true
|
|
||||||
condition: $TRAVIS_OS_NAME == "osx"
|
|
||||||
|
|
||||||
after_deploy: |
|
|
||||||
if [[ "$TRAVIS_OS_NAME" == "osx" ]]; then
|
|
||||||
bash .travis/osx_brew.sh
|
|
||||||
fi
|
|
||||||
|
|||||||
@@ -1,93 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
# © 2017-18 James R. Barlow: github.com/jbarlow83
|
|
||||||
|
|
||||||
from string import Template
|
|
||||||
from subprocess import run, PIPE
|
|
||||||
import re
|
|
||||||
|
|
||||||
recipe_template = Template("""
|
|
||||||
class Ocrmypdf < Formula
|
|
||||||
include Language::Python::Virtualenv
|
|
||||||
|
|
||||||
desc "Adds an OCR text layer to scanned PDF files"
|
|
||||||
homepage "https://github.com/jbarlow83/OCRmyPDF"
|
|
||||||
${ocrmypdf_url}
|
|
||||||
${ocrmypdf_sha256}
|
|
||||||
|
|
||||||
depends_on "pkg-config" => :build
|
|
||||||
depends_on "mupdf-tools" => :build # statically links libmupdf.a
|
|
||||||
depends_on "freetype"
|
|
||||||
depends_on "ghostscript"
|
|
||||||
depends_on "jpeg"
|
|
||||||
depends_on "libpng"
|
|
||||||
depends_on "python"
|
|
||||||
depends_on "qpdf"
|
|
||||||
depends_on "tesseract"
|
|
||||||
depends_on "unpaper"
|
|
||||||
|
|
||||||
${resources}
|
|
||||||
def install
|
|
||||||
venv = virtualenv_create(libexec, "python3")
|
|
||||||
|
|
||||||
resource("Pillow").stage do
|
|
||||||
inreplace "setup.py" do |s|
|
|
||||||
sdkprefix = MacOS::CLT.installed? ? "" : MacOS.sdk_path
|
|
||||||
s.gsub! "openjpeg.h", "probably_not_a_header_called_this_eh.h"
|
|
||||||
s.gsub! "ZLIB_ROOT = None", "ZLIB_ROOT = ('#{sdkprefix}/usr/lib', '#{sdkprefix}/usr/include')"
|
|
||||||
s.gsub! "JPEG_ROOT = None", "JPEG_ROOT = ('#{Formula["jpeg"].opt_prefix}/lib', '#{Formula["jpeg"].opt_prefix}/include')"
|
|
||||||
s.gsub! "FREETYPE_ROOT = None", "FREETYPE_ROOT = ('#{Formula["freetype"].opt_prefix}/lib', '#{Formula["freetype"].opt_prefix}/include')"
|
|
||||||
end
|
|
||||||
|
|
||||||
# avoid triggering "helpful" distutils code that doesn't recognize Xcode 7 .tbd stubs
|
|
||||||
ENV.append "CFLAGS", "-I#{MacOS.sdk_path}/System/Library/Frameworks/Tk.framework/Versions/8.5/Headers" unless MacOS::CLT.installed?
|
|
||||||
venv.pip_install Pathname.pwd
|
|
||||||
end
|
|
||||||
|
|
||||||
res = resources.map(&:name).to_set - ["Pillow"]
|
|
||||||
|
|
||||||
res.each do |r|
|
|
||||||
venv.pip_install resource(r)
|
|
||||||
end
|
|
||||||
|
|
||||||
venv.pip_install_and_link buildpath
|
|
||||||
end
|
|
||||||
|
|
||||||
test do
|
|
||||||
# Since we use Python 3, we require a UTF-8 locale
|
|
||||||
ENV["LC_ALL"] = "en_US.UTF-8"
|
|
||||||
|
|
||||||
system "#{bin}/ocrmypdf", "-f", "-q", "--deskew",
|
|
||||||
test_fixtures("test.pdf"), "ocr.pdf"
|
|
||||||
assert_predicate testpath/"ocr.pdf", :exist?
|
|
||||||
end
|
|
||||||
end
|
|
||||||
""")
|
|
||||||
|
|
||||||
def main():
|
|
||||||
p = run(['poet', '--single', 'ocrmypdf'],
|
|
||||||
encoding='utf-8', stdout=PIPE, check=True)
|
|
||||||
|
|
||||||
ocrmypdf_lines = p.stdout.splitlines()
|
|
||||||
ocrmypdf_url = ocrmypdf_lines[1].strip()
|
|
||||||
ocrmypdf_sha256 = ocrmypdf_lines[2].strip()
|
|
||||||
|
|
||||||
ocrmypdf_version = re.search(
|
|
||||||
r'ocrmypdf-(.+)\.tar.*', ocrmypdf_url).group(1)
|
|
||||||
print(f"Autobrewing {ocrmypdf_version}")
|
|
||||||
|
|
||||||
p = run(['poet', '--resources', 'ocrmypdf'],
|
|
||||||
encoding='utf-8', stdout=PIPE, check=True)
|
|
||||||
|
|
||||||
poet_resources = p.stdout
|
|
||||||
|
|
||||||
# Remove the duplicate "ocrmypdf" resource block
|
|
||||||
all_resources = poet_resources.split('resource')
|
|
||||||
kept_resources = [block for block in all_resources if 'ocrmypdf' not in block]
|
|
||||||
resources = 'resource'.join(kept_resources)
|
|
||||||
|
|
||||||
with open('ocrmypdf.rb', 'w') as out:
|
|
||||||
out.write(recipe_template.substitute(**locals()))
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
main()
|
|
||||||
@@ -1,42 +0,0 @@
|
|||||||
#!/bin/bash
|
|
||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
|
||||||
set -euo pipefail
|
|
||||||
set -x
|
|
||||||
|
|
||||||
sudo add-apt-repository ppa:vshn/ghostscript -y
|
|
||||||
sudo add-apt-repository ppa:heyarje/libav-11 -y
|
|
||||||
sudo apt-get update -qq
|
|
||||||
sudo apt-get install -y \
|
|
||||||
ghostscript \
|
|
||||||
poppler-utils \
|
|
||||||
libavformat56 \
|
|
||||||
libavcodec56 \
|
|
||||||
libavutil54 \
|
|
||||||
libffi-dev
|
|
||||||
|
|
||||||
sudo add-apt-repository ppa:alex-p/tesseract-ocr -y
|
|
||||||
|
|
||||||
sudo apt-get update
|
|
||||||
sudo apt-get autoremove -y
|
|
||||||
sudo apt-get install -y --no-install-recommends \
|
|
||||||
tesseract-ocr \
|
|
||||||
tesseract-ocr-eng \
|
|
||||||
tesseract-ocr-fra \
|
|
||||||
tesseract-ocr-deu
|
|
||||||
|
|
||||||
pip install --upgrade pip
|
|
||||||
mkdir -p packages
|
|
||||||
wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O packages/unpaper_6.1-1.deb
|
|
||||||
sudo dpkg -i packages/unpaper_6.1-1.deb
|
|
||||||
|
|
||||||
if [ ! -f /usr/local/bin/qpdf ]; then
|
|
||||||
export QPDF_RELEASE='https://github.com/qpdf/qpdf/releases/download/release-qpdf-8.0.2/qpdf-8.0.2.tar.gz'
|
|
||||||
mkdir qpdf
|
|
||||||
wget -q $QPDF_RELEASE -O - | tar xz -C qpdf --strip-components=1
|
|
||||||
cd qpdf/
|
|
||||||
export PATH="/usr/local/opt/ccache/libexec:$PATH"
|
|
||||||
./configure --prefix=/usr
|
|
||||||
make -j 2
|
|
||||||
sudo make install
|
|
||||||
cd ..
|
|
||||||
fi
|
|
||||||
@@ -1,23 +0,0 @@
|
|||||||
#!/bin/bash
|
|
||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
|
||||||
set -uo pipefail
|
|
||||||
set -x
|
|
||||||
|
|
||||||
pip3 install homebrew-pypi-poet
|
|
||||||
python3 .travis/autobrew.py
|
|
||||||
cat ocrmypdf.rb
|
|
||||||
|
|
||||||
# brew audit crashes Travis
|
|
||||||
#brew audit ocrmypdf.rb
|
|
||||||
|
|
||||||
# Important: disable debug output so token is hidden
|
|
||||||
set +x
|
|
||||||
git clone https://$HOMEBREW_OCRMYPDF_TOKEN@github.com/jbarlow83/homebrew-ocrmypdf.git
|
|
||||||
set -x
|
|
||||||
|
|
||||||
pushd homebrew-ocrmypdf
|
|
||||||
cp ../ocrmypdf.rb Formula/ocrmypdf.rb
|
|
||||||
git add Formula/ocrmypdf.rb
|
|
||||||
git commit -m "homebrew-ocrmypdf: automatic release $TRAVIS_BUILD_NUMBER $TRAVIS_TAG"
|
|
||||||
git push origin master
|
|
||||||
popd
|
|
||||||
+1
-3
@@ -126,9 +126,7 @@ If you detect an issue, please:
|
|||||||
Requirements
|
Requirements
|
||||||
------------
|
------------
|
||||||
|
|
||||||
Runs on CPython 3.6, and requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings.
|
Runs on CPython 3.5, 3.6 and 3.7. Requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings.
|
||||||
|
|
||||||
Python 3.5 is also supported.
|
|
||||||
|
|
||||||
Press & Media
|
Press & Media
|
||||||
-------------
|
-------------
|
||||||
|
|||||||
@@ -168,7 +168,7 @@ Install or upgrade the required Homebrew packages, if any are missing:
|
|||||||
brew install libxml2 libffi leptonica
|
brew install libxml2 libffi leptonica
|
||||||
brew install unpaper # optional
|
brew install unpaper # optional
|
||||||
|
|
||||||
Python 3.5 and 3.6 are supported.
|
Python 3.5, 3.6 and 3.7 are supported.
|
||||||
|
|
||||||
Install the required Tesseract OCR engine with the language packs you plan to use:
|
Install the required Tesseract OCR engine with the language packs you plan to use:
|
||||||
|
|
||||||
|
|||||||
@@ -9,6 +9,20 @@ The OCRmyPDF package itself does not contain a public API, although it is fairly
|
|||||||
find: [^`]\#([0-9]{1,3})[^0-9]
|
find: [^`]\#([0-9]{1,3})[^0-9]
|
||||||
replace: `#$1 <https://github.com/jbarlow83/OCRmyPDF/issues/$1>`_
|
replace: `#$1 <https://github.com/jbarlow83/OCRmyPDF/issues/$1>`_
|
||||||
|
|
||||||
|
|
||||||
|
v6.2.2
|
||||||
|
------
|
||||||
|
|
||||||
|
- Backport compatibility fixes for Python 3.7 and ruffus 2.7.0 from v7.0.0
|
||||||
|
- Backport fix to ignore masks when deciding what colors are on a page
|
||||||
|
- Backport some minor improvements from v7.0.0: better argument validation and warnings about the Tesseract 4.0.0 ``--user-words`` regression
|
||||||
|
|
||||||
|
v6.2.1
|
||||||
|
------
|
||||||
|
|
||||||
|
- Fix recent versions of Tesseract (after 4.0.0-beta1) not being detected as supporting the ``sandwich`` renderer (`#271 <https://github.com/ppjbarlow83/OCRmyPDF/issues/271>`_).
|
||||||
|
|
||||||
|
|
||||||
v6.2.0
|
v6.2.0
|
||||||
------
|
------
|
||||||
|
|
||||||
|
|||||||
+3
-2
@@ -1,10 +1,11 @@
|
|||||||
# requirements.txt can be used to replicate the developer's build environment
|
# requirements.txt can be used to replicate the developer's build environment
|
||||||
# setup.py lists a separate set of requirements that are looser to simplify
|
# setup.py lists a separate set of requirements that are looser to simplify
|
||||||
# installation
|
# installation
|
||||||
ruffus == 2.6.3
|
ruffus == 2.7.0
|
||||||
Pillow == 5.1.0
|
Pillow == 5.2.0
|
||||||
reportlab == 3.4.0
|
reportlab == 3.4.0
|
||||||
PyPDF2 == 1.26.0
|
PyPDF2 == 1.26.0
|
||||||
img2pdf == 0.2.4
|
img2pdf == 0.2.4
|
||||||
cffi == 1.11.5
|
cffi == 1.11.5
|
||||||
PyMuPDF == 1.12.5
|
PyMuPDF == 1.12.5
|
||||||
|
defusedxml == 0.5.0
|
||||||
|
|||||||
@@ -215,6 +215,7 @@ setup(
|
|||||||
classifiers=[
|
classifiers=[
|
||||||
"Programming Language :: Python :: 3.5",
|
"Programming Language :: Python :: 3.5",
|
||||||
"Programming Language :: Python :: 3.6",
|
"Programming Language :: Python :: 3.6",
|
||||||
|
"Programming Language :: Python :: 3.7",
|
||||||
"Development Status :: 5 - Production/Stable",
|
"Development Status :: 5 - Production/Stable",
|
||||||
"Environment :: Console",
|
"Environment :: Console",
|
||||||
"Intended Audience :: End Users/Desktop",
|
"Intended Audience :: End Users/Desktop",
|
||||||
@@ -248,7 +249,7 @@ setup(
|
|||||||
# block 5.1.0, broken wheels
|
# block 5.1.0, broken wheels
|
||||||
'PyPDF2 >= 1.26', # pure Python, so track HEAD closely
|
'PyPDF2 >= 1.26', # pure Python, so track HEAD closely
|
||||||
'reportlab >= 3.3.0', # oldest released version with sane image handling
|
'reportlab >= 3.3.0', # oldest released version with sane image handling
|
||||||
'ruffus == 2.6.3', # pinned - ocrmypdf implements a 2.6.3 workaround
|
'ruffus >= 2.7.0',
|
||||||
],
|
],
|
||||||
extras_require={
|
extras_require={
|
||||||
'fitz': ['PyMuPDF >= 1.12.5'] # for table of contents bug
|
'fitz': ['PyMuPDF >= 1.12.5'] # for table of contents bug
|
||||||
|
|||||||
+43
-26
@@ -85,6 +85,20 @@ if tesseract.version() < MINIMUM_TESS_VERSION:
|
|||||||
# -------------
|
# -------------
|
||||||
# Parser
|
# Parser
|
||||||
|
|
||||||
|
def numeric(basetype, min_=None, max_=None):
|
||||||
|
"Validator for numeric params"
|
||||||
|
min_ = basetype(min_) if min_ is not None else None
|
||||||
|
max_ = basetype(max_) if max_ is not None else None
|
||||||
|
def _numeric(string):
|
||||||
|
value = basetype(string)
|
||||||
|
if (min_ is not None and value < min_
|
||||||
|
or max_ is not None and value > max_):
|
||||||
|
msg = "%r not in valid range %r" % (string, (min_, max_))
|
||||||
|
raise argparse.ArgumentTypeError(msg)
|
||||||
|
return value
|
||||||
|
return _numeric
|
||||||
|
|
||||||
|
|
||||||
parser = argparse.ArgumentParser(
|
parser = argparse.ArgumentParser(
|
||||||
prog=PROGRAM_NAME,
|
prog=PROGRAM_NAME,
|
||||||
fromfile_prefix_chars='@',
|
fromfile_prefix_chars='@',
|
||||||
@@ -233,7 +247,7 @@ preprocessing.add_argument(
|
|||||||
help="Clean page as above, and incorporate the cleaned image in the final "
|
help="Clean page as above, and incorporate the cleaned image in the final "
|
||||||
"PDF. Might remove desired content.")
|
"PDF. Might remove desired content.")
|
||||||
preprocessing.add_argument(
|
preprocessing.add_argument(
|
||||||
'--oversample', metavar='DPI', type=int, default=0,
|
'--oversample', metavar='DPI', type=numeric(int, 0, 5000), default=0,
|
||||||
help="Oversample images to at least the specified DPI, to improve OCR "
|
help="Oversample images to at least the specified DPI, to improve OCR "
|
||||||
"results slightly")
|
"results slightly")
|
||||||
|
|
||||||
@@ -255,7 +269,7 @@ ocrsettings.add_argument(
|
|||||||
# "pages")
|
# "pages")
|
||||||
|
|
||||||
ocrsettings.add_argument(
|
ocrsettings.add_argument(
|
||||||
'--skip-big', type=float, metavar='MPixels',
|
'--skip-big', type=numeric(float, 0, 5000), metavar='MPixels',
|
||||||
help="Skip OCR on pages larger than the specified amount of megapixels, "
|
help="Skip OCR on pages larger than the specified amount of megapixels, "
|
||||||
"but include skipped pages in final output")
|
"but include skipped pages in final output")
|
||||||
|
|
||||||
@@ -263,7 +277,7 @@ advanced = parser.add_argument_group(
|
|||||||
"Advanced",
|
"Advanced",
|
||||||
"Advanced options to control Tesseract's OCR behavior")
|
"Advanced options to control Tesseract's OCR behavior")
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--max-image-mpixels', action='store', type=float, metavar='MPixels',
|
'--max-image-mpixels', action='store', type=numeric(float, 0), metavar='MPixels',
|
||||||
help="Set maximum number of pixels to unpack before treating an image as a "
|
help="Set maximum number of pixels to unpack before treating an image as a "
|
||||||
"decompression bomb",
|
"decompression bomb",
|
||||||
default=128.0)
|
default=128.0)
|
||||||
@@ -296,11 +310,11 @@ advanced.add_argument(
|
|||||||
" of Ghostscript; deprecated"
|
" of Ghostscript; deprecated"
|
||||||
)
|
)
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--tesseract-timeout', default=180.0, type=float, metavar='SECONDS',
|
'--tesseract-timeout', default=180.0, type=numeric(float, 0), metavar='SECONDS',
|
||||||
help='Give up on OCR after the timeout, but copy the preprocessed page '
|
help='Give up on OCR after the timeout, but copy the preprocessed page '
|
||||||
'into the final output')
|
'into the final output')
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
'--rotate-pages-threshold', default=14.0, type=float, metavar='CONFIDENCE',
|
'--rotate-pages-threshold', default=14.0, type=numeric(float, max_=1000), metavar='CONFIDENCE',
|
||||||
help="Only rotate pages when confidence is above this value (arbitrary "
|
help="Only rotate pages when confidence is above this value (arbitrary "
|
||||||
"units reported by tesseract)")
|
"units reported by tesseract)")
|
||||||
advanced.add_argument(
|
advanced.add_argument(
|
||||||
@@ -491,6 +505,10 @@ def check_options_advanced(options, log):
|
|||||||
"--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'"
|
"--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if tesseract.v4() and (options.user_words or options.user_patterns):
|
||||||
|
log.warning(
|
||||||
|
'Tesseract 4.x ignores --user-words, so this has no effect')
|
||||||
|
|
||||||
|
|
||||||
def check_options_metadata(options, log):
|
def check_options_metadata(options, log):
|
||||||
import unicodedata
|
import unicodedata
|
||||||
@@ -638,33 +656,31 @@ def do_ruffus_exception(ruffus_five_tuple, options, log):
|
|||||||
return ExitCode.other_error
|
return ExitCode.other_error
|
||||||
|
|
||||||
|
|
||||||
def traverse_ruffus_exception(e_args, options, log):
|
def traverse_ruffus_exception(exceptions, options, log):
|
||||||
"""Walk through a RethrownJobError and find the first exception.
|
"""Traverse a RethrownJobError and output the exceptions
|
||||||
|
|
||||||
Ruffus flattens exception to 5 element tuples. Because of a bug
|
Ruffus presents exceptions as 5 element tuples. The RethrownJobException
|
||||||
in <= 2.6.3 it may present either the single:
|
has a list of exceptions like
|
||||||
(task, job, exc, value, stack)
|
e.job_exceptions = [(5-tuple), (5-tuple), ...]
|
||||||
or something like:
|
|
||||||
[[(task, job, exc, value, stack)]]
|
|
||||||
|
|
||||||
Generally cross-process exception marshalling doesn't work well
|
ruffus < 2.7.0 had a bug with exception marshalling that would give
|
||||||
and ruffus doesn't support because BaseException has its own
|
different output whether the main or child process raised the exception.
|
||||||
implementation of __reduce__ that attempts to reconstruct the
|
We no longer support this.
|
||||||
exception based on e.__init__(e.args).
|
|
||||||
|
|
||||||
Attempting to log the exception directly marshalls it to the logger
|
Attempting to log the exception itself will re-marshall it to the logger
|
||||||
which is probably in another process, so it's better to log only
|
which is normally running in another process. It's better to avoid re-
|
||||||
data from the exception at this point.
|
marshalling.
|
||||||
|
|
||||||
The exit code will be based on this, even if multiple exceptions occurred
|
The exit code will be based on this, even if multiple exceptions occurred
|
||||||
at the same time."""
|
at the same time."""
|
||||||
|
|
||||||
if isinstance(e_args, Sequence) and isinstance(e_args[0], str) and \
|
exit_codes = []
|
||||||
len(e_args) == 5:
|
for exc in exceptions:
|
||||||
return do_ruffus_exception(e_args, options, log)
|
exit_code = do_ruffus_exception(exc, options, log)
|
||||||
elif is_iterable_notstr(e_args):
|
exit_codes.append(exit_code)
|
||||||
for exc in e_args:
|
|
||||||
return traverse_ruffus_exception(exc, options, log)
|
return exit_codes[0] # Multiple codes are rare so take the first one
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def check_closed_streams(options):
|
def check_closed_streams(options):
|
||||||
@@ -886,7 +902,8 @@ def run_pipeline():
|
|||||||
except ruffus_exceptions.RethrownJobError as e:
|
except ruffus_exceptions.RethrownJobError as e:
|
||||||
if options.verbose:
|
if options.verbose:
|
||||||
_log.debug(str(e)) # stringify exception so logger doesn't have to
|
_log.debug(str(e)) # stringify exception so logger doesn't have to
|
||||||
exitcode = traverse_ruffus_exception(e.args, options, _log)
|
exceptions = e.job_exceptions
|
||||||
|
exitcode = traverse_ruffus_exception(exceptions, options, _log)
|
||||||
if exitcode is None:
|
if exitcode is None:
|
||||||
_log.error("Unexpected ruffus exception: " + str(e))
|
_log.error("Unexpected ruffus exception: " + str(e))
|
||||||
_log.error(repr(e))
|
_log.error(repr(e))
|
||||||
|
|||||||
@@ -40,6 +40,11 @@ import codecs
|
|||||||
|
|
||||||
def verify_python3_env():
|
def verify_python3_env():
|
||||||
"""Ensures that the environment is good for unicode on Python 3."""
|
"""Ensures that the environment is good for unicode on Python 3."""
|
||||||
|
|
||||||
|
# PEP 538 changes in Python 3.7 should make this wrangling unnecessary
|
||||||
|
if sys.version_info[0:3] >= (3, 7, 0):
|
||||||
|
return
|
||||||
|
|
||||||
try:
|
try:
|
||||||
import locale
|
import locale
|
||||||
fs_enc = codecs.lookup(locale.getpreferredencoding()).name
|
fs_enc = codecs.lookup(locale.getpreferredencoding()).name
|
||||||
|
|||||||
@@ -37,6 +37,10 @@ def get_version(program, *,
|
|||||||
args_prog, close_fds=True, universal_newlines=True,
|
args_prog, close_fds=True, universal_newlines=True,
|
||||||
stdout=PIPE, stderr=STDOUT, check=True)
|
stdout=PIPE, stderr=STDOUT, check=True)
|
||||||
output = proc.stdout
|
output = proc.stdout
|
||||||
|
except FileNotFoundError as e:
|
||||||
|
raise MissingDependencyError(
|
||||||
|
"Could not find program '{}' on the PATH".format(
|
||||||
|
program)) from e
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
if e.returncode < 0:
|
if e.returncode < 0:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
|
|||||||
@@ -73,7 +73,8 @@ def has_textonly_pdf():
|
|||||||
"""
|
"""
|
||||||
args_tess = [
|
args_tess = [
|
||||||
'tesseract',
|
'tesseract',
|
||||||
'--print-parameters'
|
'--print-parameters',
|
||||||
|
'pdf'
|
||||||
]
|
]
|
||||||
params = ''
|
params = ''
|
||||||
try:
|
try:
|
||||||
@@ -159,7 +160,7 @@ def get_orientation(input_file, language: list, engine_mode, timeout: float,
|
|||||||
assert 'Rotate' not in osd
|
assert 'Rotate' not in osd
|
||||||
angle = -angle % 360
|
angle = -angle % 360
|
||||||
else:
|
else:
|
||||||
# Tesseract == 3.04.01, hopefully also Tesseract > 3.04.01
|
# Tesseract >= 3.04.01
|
||||||
# reports "Orientation in degrees" as a clockwise angle
|
# reports "Orientation in degrees" as a clockwise angle
|
||||||
assert 'Rotate' in osd
|
assert 'Rotate' in osd
|
||||||
|
|
||||||
|
|||||||
+17
-11
@@ -496,17 +496,23 @@ def rasterize_with_ghostscript(
|
|||||||
options = context.get_options()
|
options = context.get_options()
|
||||||
pageinfo = get_pageinfo(input_file, context)
|
pageinfo = get_pageinfo(input_file, context)
|
||||||
|
|
||||||
device = 'png16m' # 24-bit
|
colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m']
|
||||||
if pageinfo.images:
|
device_idx = 0
|
||||||
if all(image.comp == 1 for image in pageinfo.images):
|
def at_least(cs):
|
||||||
if all(image.bpc == 1 for image in pageinfo.images):
|
return max(device_idx, colorspaces.index(cs))
|
||||||
device = 'pngmono'
|
|
||||||
elif all(image.bpc > 1 and image.color == Colorspace.index
|
for image in pageinfo.images:
|
||||||
for image in pageinfo.images):
|
if image.type_ != 'image':
|
||||||
device = 'png256'
|
continue # ignore masks
|
||||||
elif all(image.bpc > 1 and image.color == Colorspace.gray
|
if image.bpc > 1:
|
||||||
for image in pageinfo.images):
|
if image.color == Colorspace.index:
|
||||||
device = 'pnggray'
|
device_idx = at_least('png256')
|
||||||
|
elif image.color == Colorspace.gray:
|
||||||
|
device_idx = at_least('pnggray')
|
||||||
|
else:
|
||||||
|
device_idx = at_least('png16m')
|
||||||
|
|
||||||
|
device = colorspaces[device_idx]
|
||||||
|
|
||||||
log.debug("Rasterize {0} with {1}".format(
|
log.debug("Rasterize {0} with {1}".format(
|
||||||
os.path.basename(input_file), device))
|
os.path.basename(input_file), device))
|
||||||
|
|||||||
@@ -2,4 +2,4 @@ pytest >= 3.2
|
|||||||
pytest-helpers-namespace
|
pytest-helpers-namespace
|
||||||
pytest-xdist
|
pytest-xdist
|
||||||
pytest-cov
|
pytest-cov
|
||||||
pytest-timeout
|
pytest-timeout == 1.2.1
|
||||||
|
|||||||
@@ -106,6 +106,8 @@ def main():
|
|||||||
source = os.environ['_OCRMYPDF_TEST_INFILE'] # required
|
source = os.environ['_OCRMYPDF_TEST_INFILE'] # required
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
cache_disabled = os.environ.get('_OCRMYPDF_CACHE_DISABLED', False)
|
||||||
|
|
||||||
if args.imagename == 'stdin':
|
if args.imagename == 'stdin':
|
||||||
real_tesseract()
|
real_tesseract()
|
||||||
|
|
||||||
@@ -128,7 +130,7 @@ def main():
|
|||||||
print("Tesseract cache folder {} - ".format(cache_folder), end='',
|
print("Tesseract cache folder {} - ".format(cache_folder), end='',
|
||||||
file=sys.stderr)
|
file=sys.stderr)
|
||||||
|
|
||||||
if (cache_folder / 'stderr.bin').exists():
|
if (cache_folder / 'stderr.bin').exists() and not cache_disabled:
|
||||||
# Cache hit
|
# Cache hit
|
||||||
print("HIT", file=sys.stderr)
|
print("HIT", file=sys.stderr)
|
||||||
|
|
||||||
|
|||||||
+3
-1
@@ -300,7 +300,8 @@ def test_autorotate_threshold(
|
|||||||
@pytest.mark.parametrize('renderer',RENDERERS)
|
@pytest.mark.parametrize('renderer',RENDERERS)
|
||||||
def test_ocr_timeout(renderer, resources, outpdf):
|
def test_ocr_timeout(renderer, resources, outpdf):
|
||||||
out = check_ocrmypdf(resources / 'skew.pdf', outpdf,
|
out = check_ocrmypdf(resources / 'skew.pdf', outpdf,
|
||||||
'--tesseract-timeout', '1.0')
|
'--tesseract-timeout', '0.01',
|
||||||
|
'--pdf-renderer', renderer)
|
||||||
pdfinfo = PdfInfo(out)
|
pdfinfo = PdfInfo(out)
|
||||||
assert not pdfinfo[0].has_text
|
assert not pdfinfo[0].has_text
|
||||||
|
|
||||||
@@ -966,6 +967,7 @@ def test_pdfa_n(spoof_tesseract_cache, pdfa_level, resources, outpdf):
|
|||||||
assert pdfa_info['conformance'] == 'PDF/A-{}B'.format(pdfa_level)
|
assert pdfa_info['conformance'] == 'PDF/A-{}B'.format(pdfa_level)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(sys.version_info >= (3, 7, 0), reason='fixed')
|
||||||
def test_bad_locale():
|
def test_bad_locale():
|
||||||
env = os.environ.copy()
|
env = os.environ.copy()
|
||||||
env['LC_ALL'] = 'C'
|
env['LC_ALL'] = 'C'
|
||||||
|
|||||||
Reference in New Issue
Block a user