Compare commits

...
15 Commits
Author SHA1 Message Date
James R. Barlow 6a302fdb88 Travis: nevermind xenial, then
Gave a weird build error
2018-07-12 03:27:40 -07:00
James R. Barlow 1d9cc239ee Travis: Fix v6 build failures 2018-07-12 03:22:03 -07:00
James R. Barlow d240fc1ea6 Update release notes for v6.2.2 2018-07-12 03:07:47 -07:00
James R. Barlow e7d21dd826 Skip locale check on Python 3.7 2018-07-12 03:03:34 -07:00
James R. Barlow e774b4650b ocrmypdf.exec: trap FileNotFoundError too 2018-07-12 03:01:01 -07:00
James R. Barlow 8f8e6dcdd4 Fix problem iterating ruffus exceptions and rotate-pages-threshold parameter validation 2018-07-12 02:59:38 -07:00
James R. Barlow 5252b88f0f Travis: modernize with v7.0.0 updates
Travis: add 3.7 testing
Travis: remove deploy to testpypi since it's broken
Travis: cherry-pick change to declarative APT
From efb9572
Removed pngquant
Travis: Remove linux_before_install.sh
Eliminate Homebrew autobrewing
2018-07-12 02:47:29 -07:00
James R. Barlow ea69883386 Tests: Speed up a slow test (cherry-picked from v7) 2018-07-12 02:47:15 -07:00
James R. Barlow eb343b1e37 Tests: Add ability to disable use of cache (cherrypicked from v7) 2018-07-12 02:46:53 -07:00
James R. Barlow 9f02de55be main: do better parameter validation 2018-07-12 02:46:52 -07:00
James R. Barlow 7394a4cf49 Cherrypick warning about --user-words not having any effect
Might be available in full release of Tess4
2018-07-12 02:46:34 -07:00
James R. Barlow ed9fb110b1 Fix a comment about Tesseract behavior in certain versions 2018-07-12 02:46:34 -07:00
James R. Barlow 4650074428 Cherrypick Python 3.7 documentation updates from v7.0.0
From b0eacd6
2018-07-12 02:45:51 -07:00
James R. Barlow 70aa644c10 Backport Python 3.7 fix for ruffus 2.7.0 from ocrmypdf v7.0.0 2018-07-12 02:45:51 -07:00
James R. Barlow 2ccb3edc58 Ignore masks when deciding what color to rasterize at 2018-07-12 02:45:51 -07:00
17 changed files with 156 additions and 264 deletions
+1
View File
@@ -2,6 +2,7 @@
*.pyc *.pyc
*.sublime-* *.sublime-*
venv*/ venv*/
.venv/
pyvenv.cfg pyvenv.cfg
tasks.py tasks.py
.bash_history .bash_history
+31 -36
View File
@@ -1,32 +1,51 @@
dist: trusty dist: trusty
language: python language: python
cache: cache:
ccache: true
pip: true pip: true
directories: directories:
- $HOME/Library/Caches/Homebrew - $HOME/Library/Caches/Homebrew
env: addons:
global: apt:
- secure: "hsf6MT+n2x3OiDM2fQyJZdV0/PWYmv81LdVqC6cfnHBE/8N3DloJRqQ7WfO14TxhiK9PEC7MpyCj0lSabUHEO7gSH6Vks6I1asoSkt8S9/bSMlhT4hei+pwVpeGEiU5xHVATNjY+D919VC3IFvc3XmjT74h/2SLhaZ+jhEmDggM=" # HOMEBREW_OCRMYPDF_TOKEN update: true
sources:
- sourceline: 'ppa:alex-p/tesseract-ocr'
- sourceline: 'ppa:heyarje/libav-11'
- sourceline: 'ppa:vshn/ghostscript'
packages:
- ghostscript
- libavcodec56
- libavformat56
- libavutil54
- libffi-dev
- poppler-utils
- qpdf
- tesseract-ocr
- tesseract-ocr-deu
- tesseract-ocr-eng
- tesseract-ocr-fra
matrix: matrix:
include: include:
- os: linux - os: linux
sudo: required sudo: required
language: python language: python
python: 3.5 python: "3.5"
env: EXTRAS= env: EXTRAS=
- os: linux - os: linux
sudo: required sudo: required
language: python language: python
python: 3.6 python: "3.6"
env: EXTRAS= env: EXTRAS=
- os: linux - os: linux
sudo: required sudo: required
language: python language: python
python: 3.6 python: "3.6"
env: EXTRAS=[fitz] env: EXTRAS=[fitz]
- os: linux
sudo: required
language: python
python: "3.7-dev"
- os: osx - os: osx
osx_image: xcode8 osx_image: xcode8
language: generic language: generic
@@ -41,7 +60,10 @@ before_cache:
before_install: | before_install: |
if [[ "$TRAVIS_OS_NAME" == "linux" ]]; then if [[ "$TRAVIS_OS_NAME" == "linux" ]]; then
bash .travis/linux_before_install.sh pip install --upgrade pip
mkdir -p packages
wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O packages/unpaper_6.1-1.deb
sudo dpkg -i packages/unpaper_6.1-1.deb
elif [[ "$TRAVIS_OS_NAME" == "osx" ]]; then elif [[ "$TRAVIS_OS_NAME" == "osx" ]]; then
brew update && brew bundle --file=.travis/Brewfile brew update && brew bundle --file=.travis/Brewfile
pip3 install --upgrade pip pip3 install --upgrade pip
@@ -49,6 +71,7 @@ before_install: |
fi fi
install: install:
- pip3 install pycparser # py3.7 workaround for https://github.com/eliben/pycparser/issues/251
- pip3 install ".$EXTRAS" - pip3 install ".$EXTRAS"
- pip3 install -r test_requirements.txt - pip3 install -r test_requirements.txt
@@ -73,31 +96,3 @@ deploy:
tags: true tags: true
condition: $TRAVIS_PYTHON_VERSION == "3.6" && $TRAVIS_OS_NAME == "linux" && $EXTRAS == "" condition: $TRAVIS_PYTHON_VERSION == "3.6" && $TRAVIS_OS_NAME == "linux" && $EXTRAS == ""
skip_upload_docs: true skip_upload_docs: true
# test pypi
- provider: pypi
server: https://testpypi.pypi.org/legacy/
user: ocrmypdf-travis
password:
secure: "DTFOmmNL6olA0+yXvp4u9jXZlZeqrJsJ0526jzqf4a3gZ6jnGTq5UI6WzRsslSyoMMfXKtHQebqHM6ogSgCZinyZ3ufHJo8fn9brxbEc2gsiWkbj5o3bGwdWMT1vNNE7XW0VCpw87rZ1EEwjl4FJHFudMlPR1yfU5+uq0k0PACo="
distributions: "sdist"
on:
branch: develop
tags: false
condition: $TRAVIS_OS_NAME == "osx"
skip_upload_docs: true
# null deploy for osx
# we really just want to run after_deploy *after* pypi upload is done, but
# after_deploy on runs if a given box deployed
- provider: script
script: /usr/bin/true
on:
branch: master
tags: true
condition: $TRAVIS_OS_NAME == "osx"
after_deploy: |
if [[ "$TRAVIS_OS_NAME" == "osx" ]]; then
bash .travis/osx_brew.sh
fi
-93
View File
@@ -1,93 +0,0 @@
#!/usr/bin/env python3
# © 2017-18 James R. Barlow: github.com/jbarlow83
from string import Template
from subprocess import run, PIPE
import re
recipe_template = Template("""
class Ocrmypdf < Formula
include Language::Python::Virtualenv
desc "Adds an OCR text layer to scanned PDF files"
homepage "https://github.com/jbarlow83/OCRmyPDF"
${ocrmypdf_url}
${ocrmypdf_sha256}
depends_on "pkg-config" => :build
depends_on "mupdf-tools" => :build # statically links libmupdf.a
depends_on "freetype"
depends_on "ghostscript"
depends_on "jpeg"
depends_on "libpng"
depends_on "python"
depends_on "qpdf"
depends_on "tesseract"
depends_on "unpaper"
${resources}
def install
venv = virtualenv_create(libexec, "python3")
resource("Pillow").stage do
inreplace "setup.py" do |s|
sdkprefix = MacOS::CLT.installed? ? "" : MacOS.sdk_path
s.gsub! "openjpeg.h", "probably_not_a_header_called_this_eh.h"
s.gsub! "ZLIB_ROOT = None", "ZLIB_ROOT = ('#{sdkprefix}/usr/lib', '#{sdkprefix}/usr/include')"
s.gsub! "JPEG_ROOT = None", "JPEG_ROOT = ('#{Formula["jpeg"].opt_prefix}/lib', '#{Formula["jpeg"].opt_prefix}/include')"
s.gsub! "FREETYPE_ROOT = None", "FREETYPE_ROOT = ('#{Formula["freetype"].opt_prefix}/lib', '#{Formula["freetype"].opt_prefix}/include')"
end
# avoid triggering "helpful" distutils code that doesn't recognize Xcode 7 .tbd stubs
ENV.append "CFLAGS", "-I#{MacOS.sdk_path}/System/Library/Frameworks/Tk.framework/Versions/8.5/Headers" unless MacOS::CLT.installed?
venv.pip_install Pathname.pwd
end
res = resources.map(&:name).to_set - ["Pillow"]
res.each do |r|
venv.pip_install resource(r)
end
venv.pip_install_and_link buildpath
end
test do
# Since we use Python 3, we require a UTF-8 locale
ENV["LC_ALL"] = "en_US.UTF-8"
system "#{bin}/ocrmypdf", "-f", "-q", "--deskew",
test_fixtures("test.pdf"), "ocr.pdf"
assert_predicate testpath/"ocr.pdf", :exist?
end
end
""")
def main():
p = run(['poet', '--single', 'ocrmypdf'],
encoding='utf-8', stdout=PIPE, check=True)
ocrmypdf_lines = p.stdout.splitlines()
ocrmypdf_url = ocrmypdf_lines[1].strip()
ocrmypdf_sha256 = ocrmypdf_lines[2].strip()
ocrmypdf_version = re.search(
r'ocrmypdf-(.+)\.tar.*', ocrmypdf_url).group(1)
print(f"Autobrewing {ocrmypdf_version}")
p = run(['poet', '--resources', 'ocrmypdf'],
encoding='utf-8', stdout=PIPE, check=True)
poet_resources = p.stdout
# Remove the duplicate "ocrmypdf" resource block
all_resources = poet_resources.split('resource')
kept_resources = [block for block in all_resources if 'ocrmypdf' not in block]
resources = 'resource'.join(kept_resources)
with open('ocrmypdf.rb', 'w') as out:
out.write(recipe_template.substitute(**locals()))
if __name__ == '__main__':
main()
-31
View File
@@ -1,31 +0,0 @@
#!/bin/bash
# © 2017 James R. Barlow: github.com/jbarlow83
set -euo pipefail
set -x
sudo add-apt-repository ppa:vshn/ghostscript -y
sudo add-apt-repository ppa:heyarje/libav-11 -y
sudo apt-get update -qq
sudo apt-get install -y \
ghostscript \
poppler-utils \
libavformat56 \
libavcodec56 \
libavutil54 \
libffi-dev \
qpdf
sudo add-apt-repository ppa:alex-p/tesseract-ocr -y
sudo apt-get update
sudo apt-get autoremove -y
sudo apt-get install -y --no-install-recommends \
tesseract-ocr \
tesseract-ocr-eng \
tesseract-ocr-fra \
tesseract-ocr-deu
pip install --upgrade pip
mkdir -p packages
wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O packages/unpaper_6.1-1.deb
sudo dpkg -i packages/unpaper_6.1-1.deb
-23
View File
@@ -1,23 +0,0 @@
#!/bin/bash
# © 2017 James R. Barlow: github.com/jbarlow83
set -uo pipefail
set -x
pip3 install homebrew-pypi-poet
python3 .travis/autobrew.py
cat ocrmypdf.rb
# brew audit crashes Travis
#brew audit ocrmypdf.rb
# Important: disable debug output so token is hidden
set +x
git clone https://$HOMEBREW_OCRMYPDF_TOKEN@github.com/jbarlow83/homebrew-ocrmypdf.git
set -x
pushd homebrew-ocrmypdf
cp ../ocrmypdf.rb Formula/ocrmypdf.rb
git add Formula/ocrmypdf.rb
git commit -m "homebrew-ocrmypdf: automatic release $TRAVIS_BUILD_NUMBER $TRAVIS_TAG"
git push origin master
popd
+1 -3
View File
@@ -126,9 +126,7 @@ If you detect an issue, please:
Requirements Requirements
------------ ------------
Runs on CPython 3.6, and requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings. Runs on CPython 3.5, 3.6 and 3.7. Requires external program installations of Ghostscript, Tesseract OCR, QPDF, and Leptonica. ocrmypdf is pure Python, but uses CFFI to portably generate library bindings.
Python 3.5 is also supported.
Press & Media Press & Media
------------- -------------
+1 -1
View File
@@ -168,7 +168,7 @@ Install or upgrade the required Homebrew packages, if any are missing:
brew install libxml2 libffi leptonica brew install libxml2 libffi leptonica
brew install unpaper # optional brew install unpaper # optional
Python 3.5 and 3.6 are supported. Python 3.5, 3.6 and 3.7 are supported.
Install the required Tesseract OCR engine with the language packs you plan to use: Install the required Tesseract OCR engine with the language packs you plan to use:
+7
View File
@@ -10,6 +10,13 @@ The OCRmyPDF package itself does not contain a public API, although it is fairly
replace: `#$1 <https://github.com/jbarlow83/OCRmyPDF/issues/$1>`_ replace: `#$1 <https://github.com/jbarlow83/OCRmyPDF/issues/$1>`_
v6.2.2
------
- Backport compatibility fixes for Python 3.7 and ruffus 2.7.0 from v7.0.0
- Backport fix to ignore masks when deciding what colors are on a page
- Backport some minor improvements from v7.0.0: better argument validation and warnings about the Tesseract 4.0.0 ``--user-words`` regression
v6.2.1 v6.2.1
------ ------
+3 -2
View File
@@ -1,10 +1,11 @@
# requirements.txt can be used to replicate the developer's build environment # requirements.txt can be used to replicate the developer's build environment
# setup.py lists a separate set of requirements that are looser to simplify # setup.py lists a separate set of requirements that are looser to simplify
# installation # installation
ruffus == 2.6.3 ruffus == 2.7.0
Pillow == 5.1.0 Pillow == 5.2.0
reportlab == 3.4.0 reportlab == 3.4.0
PyPDF2 == 1.26.0 PyPDF2 == 1.26.0
img2pdf == 0.2.4 img2pdf == 0.2.4
cffi == 1.11.5 cffi == 1.11.5
PyMuPDF == 1.12.5 PyMuPDF == 1.12.5
defusedxml == 0.5.0
+2 -1
View File
@@ -215,6 +215,7 @@ setup(
classifiers=[ classifiers=[
"Programming Language :: Python :: 3.5", "Programming Language :: Python :: 3.5",
"Programming Language :: Python :: 3.6", "Programming Language :: Python :: 3.6",
"Programming Language :: Python :: 3.7",
"Development Status :: 5 - Production/Stable", "Development Status :: 5 - Production/Stable",
"Environment :: Console", "Environment :: Console",
"Intended Audience :: End Users/Desktop", "Intended Audience :: End Users/Desktop",
@@ -248,7 +249,7 @@ setup(
# block 5.1.0, broken wheels # block 5.1.0, broken wheels
'PyPDF2 >= 1.26', # pure Python, so track HEAD closely 'PyPDF2 >= 1.26', # pure Python, so track HEAD closely
'reportlab >= 3.3.0', # oldest released version with sane image handling 'reportlab >= 3.3.0', # oldest released version with sane image handling
'ruffus == 2.6.3', # pinned - ocrmypdf implements a 2.6.3 workaround 'ruffus >= 2.7.0',
], ],
extras_require={ extras_require={
'fitz': ['PyMuPDF >= 1.12.5'] # for table of contents bug 'fitz': ['PyMuPDF >= 1.12.5'] # for table of contents bug
+43 -26
View File
@@ -85,6 +85,20 @@ if tesseract.version() < MINIMUM_TESS_VERSION:
# ------------- # -------------
# Parser # Parser
def numeric(basetype, min_=None, max_=None):
"Validator for numeric params"
min_ = basetype(min_) if min_ is not None else None
max_ = basetype(max_) if max_ is not None else None
def _numeric(string):
value = basetype(string)
if (min_ is not None and value < min_
or max_ is not None and value > max_):
msg = "%r not in valid range %r" % (string, (min_, max_))
raise argparse.ArgumentTypeError(msg)
return value
return _numeric
parser = argparse.ArgumentParser( parser = argparse.ArgumentParser(
prog=PROGRAM_NAME, prog=PROGRAM_NAME,
fromfile_prefix_chars='@', fromfile_prefix_chars='@',
@@ -233,7 +247,7 @@ preprocessing.add_argument(
help="Clean page as above, and incorporate the cleaned image in the final " help="Clean page as above, and incorporate the cleaned image in the final "
"PDF. Might remove desired content.") "PDF. Might remove desired content.")
preprocessing.add_argument( preprocessing.add_argument(
'--oversample', metavar='DPI', type=int, default=0, '--oversample', metavar='DPI', type=numeric(int, 0, 5000), default=0,
help="Oversample images to at least the specified DPI, to improve OCR " help="Oversample images to at least the specified DPI, to improve OCR "
"results slightly") "results slightly")
@@ -255,7 +269,7 @@ ocrsettings.add_argument(
# "pages") # "pages")
ocrsettings.add_argument( ocrsettings.add_argument(
'--skip-big', type=float, metavar='MPixels', '--skip-big', type=numeric(float, 0, 5000), metavar='MPixels',
help="Skip OCR on pages larger than the specified amount of megapixels, " help="Skip OCR on pages larger than the specified amount of megapixels, "
"but include skipped pages in final output") "but include skipped pages in final output")
@@ -263,7 +277,7 @@ advanced = parser.add_argument_group(
"Advanced", "Advanced",
"Advanced options to control Tesseract's OCR behavior") "Advanced options to control Tesseract's OCR behavior")
advanced.add_argument( advanced.add_argument(
'--max-image-mpixels', action='store', type=float, metavar='MPixels', '--max-image-mpixels', action='store', type=numeric(float, 0), metavar='MPixels',
help="Set maximum number of pixels to unpack before treating an image as a " help="Set maximum number of pixels to unpack before treating an image as a "
"decompression bomb", "decompression bomb",
default=128.0) default=128.0)
@@ -296,11 +310,11 @@ advanced.add_argument(
" of Ghostscript; deprecated" " of Ghostscript; deprecated"
) )
advanced.add_argument( advanced.add_argument(
'--tesseract-timeout', default=180.0, type=float, metavar='SECONDS', '--tesseract-timeout', default=180.0, type=numeric(float, 0), metavar='SECONDS',
help='Give up on OCR after the timeout, but copy the preprocessed page ' help='Give up on OCR after the timeout, but copy the preprocessed page '
'into the final output') 'into the final output')
advanced.add_argument( advanced.add_argument(
'--rotate-pages-threshold', default=14.0, type=float, metavar='CONFIDENCE', '--rotate-pages-threshold', default=14.0, type=numeric(float, max_=1000), metavar='CONFIDENCE',
help="Only rotate pages when confidence is above this value (arbitrary " help="Only rotate pages when confidence is above this value (arbitrary "
"units reported by tesseract)") "units reported by tesseract)")
advanced.add_argument( advanced.add_argument(
@@ -491,6 +505,10 @@ def check_options_advanced(options, log):
"--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'" "--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'"
) )
if tesseract.v4() and (options.user_words or options.user_patterns):
log.warning(
'Tesseract 4.x ignores --user-words, so this has no effect')
def check_options_metadata(options, log): def check_options_metadata(options, log):
import unicodedata import unicodedata
@@ -638,33 +656,31 @@ def do_ruffus_exception(ruffus_five_tuple, options, log):
return ExitCode.other_error return ExitCode.other_error
def traverse_ruffus_exception(e_args, options, log): def traverse_ruffus_exception(exceptions, options, log):
"""Walk through a RethrownJobError and find the first exception. """Traverse a RethrownJobError and output the exceptions
Ruffus flattens exception to 5 element tuples. Because of a bug Ruffus presents exceptions as 5 element tuples. The RethrownJobException
in <= 2.6.3 it may present either the single: has a list of exceptions like
(task, job, exc, value, stack) e.job_exceptions = [(5-tuple), (5-tuple), ...]
or something like:
[[(task, job, exc, value, stack)]]
Generally cross-process exception marshalling doesn't work well ruffus < 2.7.0 had a bug with exception marshalling that would give
and ruffus doesn't support because BaseException has its own different output whether the main or child process raised the exception.
implementation of __reduce__ that attempts to reconstruct the We no longer support this.
exception based on e.__init__(e.args).
Attempting to log the exception directly marshalls it to the logger Attempting to log the exception itself will re-marshall it to the logger
which is probably in another process, so it's better to log only which is normally running in another process. It's better to avoid re-
data from the exception at this point. marshalling.
The exit code will be based on this, even if multiple exceptions occurred The exit code will be based on this, even if multiple exceptions occurred
at the same time.""" at the same time."""
if isinstance(e_args, Sequence) and isinstance(e_args[0], str) and \ exit_codes = []
len(e_args) == 5: for exc in exceptions:
return do_ruffus_exception(e_args, options, log) exit_code = do_ruffus_exception(exc, options, log)
elif is_iterable_notstr(e_args): exit_codes.append(exit_code)
for exc in e_args:
return traverse_ruffus_exception(exc, options, log) return exit_codes[0] # Multiple codes are rare so take the first one
def check_closed_streams(options): def check_closed_streams(options):
@@ -886,7 +902,8 @@ def run_pipeline():
except ruffus_exceptions.RethrownJobError as e: except ruffus_exceptions.RethrownJobError as e:
if options.verbose: if options.verbose:
_log.debug(str(e)) # stringify exception so logger doesn't have to _log.debug(str(e)) # stringify exception so logger doesn't have to
exitcode = traverse_ruffus_exception(e.args, options, _log) exceptions = e.job_exceptions
exitcode = traverse_ruffus_exception(exceptions, options, _log)
if exitcode is None: if exitcode is None:
_log.error("Unexpected ruffus exception: " + str(e)) _log.error("Unexpected ruffus exception: " + str(e))
_log.error(repr(e)) _log.error(repr(e))
+5
View File
@@ -40,6 +40,11 @@ import codecs
def verify_python3_env(): def verify_python3_env():
"""Ensures that the environment is good for unicode on Python 3.""" """Ensures that the environment is good for unicode on Python 3."""
# PEP 538 changes in Python 3.7 should make this wrangling unnecessary
if sys.version_info[0:3] >= (3, 7, 0):
return
try: try:
import locale import locale
fs_enc = codecs.lookup(locale.getpreferredencoding()).name fs_enc = codecs.lookup(locale.getpreferredencoding()).name
+4
View File
@@ -37,6 +37,10 @@ def get_version(program, *,
args_prog, close_fds=True, universal_newlines=True, args_prog, close_fds=True, universal_newlines=True,
stdout=PIPE, stderr=STDOUT, check=True) stdout=PIPE, stderr=STDOUT, check=True)
output = proc.stdout output = proc.stdout
except FileNotFoundError as e:
raise MissingDependencyError(
"Could not find program '{}' on the PATH".format(
program)) from e
except CalledProcessError as e: except CalledProcessError as e:
if e.returncode < 0: if e.returncode < 0:
raise MissingDependencyError( raise MissingDependencyError(
+1 -1
View File
@@ -160,7 +160,7 @@ def get_orientation(input_file, language: list, engine_mode, timeout: float,
assert 'Rotate' not in osd assert 'Rotate' not in osd
angle = -angle % 360 angle = -angle % 360
else: else:
# Tesseract == 3.04.01, hopefully also Tesseract > 3.04.01 # Tesseract >= 3.04.01
# reports "Orientation in degrees" as a clockwise angle # reports "Orientation in degrees" as a clockwise angle
assert 'Rotate' in osd assert 'Rotate' in osd
+17 -11
View File
@@ -496,17 +496,23 @@ def rasterize_with_ghostscript(
options = context.get_options() options = context.get_options()
pageinfo = get_pageinfo(input_file, context) pageinfo = get_pageinfo(input_file, context)
device = 'png16m' # 24-bit colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m']
if pageinfo.images: device_idx = 0
if all(image.comp == 1 for image in pageinfo.images): def at_least(cs):
if all(image.bpc == 1 for image in pageinfo.images): return max(device_idx, colorspaces.index(cs))
device = 'pngmono'
elif all(image.bpc > 1 and image.color == Colorspace.index for image in pageinfo.images:
for image in pageinfo.images): if image.type_ != 'image':
device = 'png256' continue # ignore masks
elif all(image.bpc > 1 and image.color == Colorspace.gray if image.bpc > 1:
for image in pageinfo.images): if image.color == Colorspace.index:
device = 'pnggray' device_idx = at_least('png256')
elif image.color == Colorspace.gray:
device_idx = at_least('pnggray')
else:
device_idx = at_least('png16m')
device = colorspaces[device_idx]
log.debug("Rasterize {0} with {1}".format( log.debug("Rasterize {0} with {1}".format(
os.path.basename(input_file), device)) os.path.basename(input_file), device))
+3 -1
View File
@@ -106,6 +106,8 @@ def main():
source = os.environ['_OCRMYPDF_TEST_INFILE'] # required source = os.environ['_OCRMYPDF_TEST_INFILE'] # required
args = parser.parse_args() args = parser.parse_args()
cache_disabled = os.environ.get('_OCRMYPDF_CACHE_DISABLED', False)
if args.imagename == 'stdin': if args.imagename == 'stdin':
real_tesseract() real_tesseract()
@@ -128,7 +130,7 @@ def main():
print("Tesseract cache folder {} - ".format(cache_folder), end='', print("Tesseract cache folder {} - ".format(cache_folder), end='',
file=sys.stderr) file=sys.stderr)
if (cache_folder / 'stderr.bin').exists(): if (cache_folder / 'stderr.bin').exists() and not cache_disabled:
# Cache hit # Cache hit
print("HIT", file=sys.stderr) print("HIT", file=sys.stderr)
+3 -1
View File
@@ -300,7 +300,8 @@ def test_autorotate_threshold(
@pytest.mark.parametrize('renderer',RENDERERS) @pytest.mark.parametrize('renderer',RENDERERS)
def test_ocr_timeout(renderer, resources, outpdf): def test_ocr_timeout(renderer, resources, outpdf):
out = check_ocrmypdf(resources / 'skew.pdf', outpdf, out = check_ocrmypdf(resources / 'skew.pdf', outpdf,
'--tesseract-timeout', '1.0') '--tesseract-timeout', '0.01',
'--pdf-renderer', renderer)
pdfinfo = PdfInfo(out) pdfinfo = PdfInfo(out)
assert not pdfinfo[0].has_text assert not pdfinfo[0].has_text
@@ -966,6 +967,7 @@ def test_pdfa_n(spoof_tesseract_cache, pdfa_level, resources, outpdf):
assert pdfa_info['conformance'] == 'PDF/A-{}B'.format(pdfa_level) assert pdfa_info['conformance'] == 'PDF/A-{}B'.format(pdfa_level)
@pytest.mark.skipif(sys.version_info >= (3, 7, 0), reason='fixed')
def test_bad_locale(): def test_bad_locale():
env = os.environ.copy() env = os.environ.copy()
env['LC_ALL'] = 'C' env['LC_ALL'] = 'C'