Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
5a777ee9bc | ||
|
|
7bbf6bc7f4 | ||
|
|
9bfc45702d | ||
|
|
40aa82ab41 | ||
|
|
5d2c67c62b | ||
|
|
a00ef4836d | ||
|
|
3ef766bb93 | ||
|
|
44b5a18462 | ||
|
|
4df716f0e8 | ||
|
|
fcbf34a4d3 |
@@ -23,6 +23,10 @@ class Ocrmypdf < Formula
|
|||||||
depends_on "unpaper"
|
depends_on "unpaper"
|
||||||
depends_on "qpdf"
|
depends_on "qpdf"
|
||||||
|
|
||||||
|
# mactex installs its own ghostscript by default which causes problems
|
||||||
|
# mactex users should use caskroom/cask/mactex-no-ghostscript instead
|
||||||
|
conflicts_with cask: "caskroom/cask/mactex"
|
||||||
|
|
||||||
# For Pillow source install
|
# For Pillow source install
|
||||||
depends_on "openjpeg"
|
depends_on "openjpeg"
|
||||||
depends_on "freetype"
|
depends_on "freetype"
|
||||||
|
|||||||
+4
-1
@@ -1,4 +1,7 @@
|
|||||||
Copyright (c) 2013-2016, The OCRmyPDF Authors
|
Copyright (c) 2014-2017, James R. Barlow
|
||||||
|
|
||||||
|
Copyright (c) 2013-2014, Julien Pfefferkorn
|
||||||
|
Copyright (c) 2013-2017, The OCRmyPDF Authors
|
||||||
|
|
||||||
Permission is hereby granted, free of charge, to any person obtaining a
|
Permission is hereby granted, free of charge, to any person obtaining a
|
||||||
copy of this software and associated documentation files (the
|
copy of this software and associated documentation files (the
|
||||||
|
|||||||
@@ -5,6 +5,15 @@ OCRmyPDF uses `semantic versioning <http://semver.org/>`_ for its command line i
|
|||||||
|
|
||||||
The OCRmyPDF package itself does not contain a public API, although it is fairly stable and breaking changes are usually timed with a major release. A future release will clearly define the stable public API.
|
The OCRmyPDF package itself does not contain a public API, although it is fairly stable and breaking changes are usually timed with a major release. A future release will clearly define the stable public API.
|
||||||
|
|
||||||
|
v5.4.3
|
||||||
|
------
|
||||||
|
|
||||||
|
- If a subprocess fails to report its version when queried, exit cleanly with an error instead of throwing an exception
|
||||||
|
- Added test to confirm that the system locale is Unicode-aware and fail early if it's not
|
||||||
|
- Clarified some copyright information
|
||||||
|
- Updated pinned requirements.txt so the homebrew formula captures more recent versions
|
||||||
|
|
||||||
|
|
||||||
v5.4.2
|
v5.4.2
|
||||||
------
|
------
|
||||||
|
|
||||||
|
|||||||
@@ -4,6 +4,7 @@ import pkg_resources
|
|||||||
|
|
||||||
PROGRAM_NAME = 'ocrmypdf'
|
PROGRAM_NAME = 'ocrmypdf'
|
||||||
|
|
||||||
VERSION = pkg_resources.get_distribution('ocrmypdf').version
|
# Official PEP 396
|
||||||
|
__version__ = pkg_resources.get_distribution('ocrmypdf').version
|
||||||
|
|
||||||
|
VERSION = __version__
|
||||||
|
|||||||
@@ -28,6 +28,7 @@ from . import PROGRAM_NAME, VERSION
|
|||||||
|
|
||||||
from .exceptions import *
|
from .exceptions import *
|
||||||
from . import exceptions as ocrmypdf_exceptions
|
from . import exceptions as ocrmypdf_exceptions
|
||||||
|
from . import _unicodefun
|
||||||
|
|
||||||
warnings.simplefilter('ignore', pypdf.utils.PdfReadWarning)
|
warnings.simplefilter('ignore', pypdf.utils.PdfReadWarning)
|
||||||
|
|
||||||
@@ -49,6 +50,8 @@ def complain(message):
|
|||||||
if 'IDE_PROJECT_ROOTS' in os.environ:
|
if 'IDE_PROJECT_ROOTS' in os.environ:
|
||||||
os.environ['PATH'] = '/usr/local/bin:' + os.environ['PATH']
|
os.environ['PATH'] = '/usr/local/bin:' + os.environ['PATH']
|
||||||
|
|
||||||
|
_unicodefun._verify_python3_env()
|
||||||
|
|
||||||
if tesseract.version() < MINIMUM_TESS_VERSION:
|
if tesseract.version() < MINIMUM_TESS_VERSION:
|
||||||
complain(
|
complain(
|
||||||
"Please install tesseract {0} or newer "
|
"Please install tesseract {0} or newer "
|
||||||
|
|||||||
@@ -0,0 +1,108 @@
|
|||||||
|
# Copyright (c) 2014, Armin Ronacher
|
||||||
|
#
|
||||||
|
# Copyright (c) 2017, James R Barlow
|
||||||
|
#
|
||||||
|
# Some rights reserved.
|
||||||
|
#
|
||||||
|
# Redistribution and use in source and binary forms, with or without
|
||||||
|
# modification, are permitted provided that the following conditions are
|
||||||
|
# met:
|
||||||
|
#
|
||||||
|
# * Redistributions of source code must retain the above copyright
|
||||||
|
# notice, this list of conditions and the following disclaimer.
|
||||||
|
#
|
||||||
|
# * Redistributions in binary form must reproduce the above
|
||||||
|
# copyright notice, this list of conditions and the following
|
||||||
|
# disclaimer in the documentation and/or other materials provided
|
||||||
|
# with the distribution.
|
||||||
|
#
|
||||||
|
# * The names of the contributors may not be used to endorse or
|
||||||
|
# promote products derived from this software without specific
|
||||||
|
# prior written permission.
|
||||||
|
#
|
||||||
|
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS
|
||||||
|
# "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
|
||||||
|
# LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR
|
||||||
|
# A PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT
|
||||||
|
# OWNER OR CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL,
|
||||||
|
# SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT
|
||||||
|
# LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE,
|
||||||
|
# DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY
|
||||||
|
# THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||||
|
# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||||
|
# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||||
|
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import codecs
|
||||||
|
|
||||||
|
|
||||||
|
def _verify_python3_env():
|
||||||
|
"""Ensures that the environment is good for unicode on Python 3."""
|
||||||
|
try:
|
||||||
|
import locale
|
||||||
|
fs_enc = codecs.lookup(locale.getpreferredencoding()).name
|
||||||
|
except Exception:
|
||||||
|
fs_enc = 'ascii'
|
||||||
|
if fs_enc != 'ascii':
|
||||||
|
return
|
||||||
|
|
||||||
|
extra = ''
|
||||||
|
if os.name == 'posix':
|
||||||
|
import subprocess
|
||||||
|
rv = subprocess.Popen(['locale', '-a'], stdout=subprocess.PIPE,
|
||||||
|
stderr=subprocess.PIPE).communicate()[0]
|
||||||
|
good_locales = set()
|
||||||
|
has_c_utf8 = False
|
||||||
|
|
||||||
|
# Make sure we're operating on text here.
|
||||||
|
if isinstance(rv, bytes):
|
||||||
|
rv = rv.decode('ascii', 'replace')
|
||||||
|
|
||||||
|
for line in rv.splitlines():
|
||||||
|
locale = line.strip()
|
||||||
|
if locale.lower().endswith(('.utf-8', '.utf8')):
|
||||||
|
good_locales.add(locale)
|
||||||
|
if locale.lower() in ('c.utf8', 'c.utf-8'):
|
||||||
|
has_c_utf8 = True
|
||||||
|
|
||||||
|
extra += '\n\n'
|
||||||
|
if not good_locales:
|
||||||
|
extra += (
|
||||||
|
'Additional information: on this system no suitable UTF-8\n'
|
||||||
|
'locales were discovered. This most likely requires resolving\n'
|
||||||
|
'by reconfiguring the locale system.'
|
||||||
|
)
|
||||||
|
elif has_c_utf8:
|
||||||
|
extra += (
|
||||||
|
'This system supports the C.UTF-8 locale which is recommended.\n'
|
||||||
|
'You might be able to resolve your issue by exporting the\n'
|
||||||
|
'following environment variables:\n\n'
|
||||||
|
' export LC_ALL=C.UTF-8\n'
|
||||||
|
' export LANG=C.UTF-8'
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
extra += (
|
||||||
|
'This system lists a couple of UTF-8 supporting locales that\n'
|
||||||
|
'you can pick from. The following suitable locales were\n'
|
||||||
|
'discovered: %s'
|
||||||
|
) % ', '.join(sorted(good_locales))
|
||||||
|
|
||||||
|
bad_locale = None
|
||||||
|
for locale in os.environ.get('LC_ALL'), os.environ.get('LANG'):
|
||||||
|
if locale and locale.lower().endswith(('.utf-8', '.utf8')):
|
||||||
|
bad_locale = locale
|
||||||
|
if locale is not None:
|
||||||
|
break
|
||||||
|
if bad_locale is not None:
|
||||||
|
extra += (
|
||||||
|
'\n\ocrmypdf discovered that you exported a UTF-8 locale\n'
|
||||||
|
'but the locale system could not pick up from it because\n'
|
||||||
|
'it does not exist. The exported locale is "%s" but it\n'
|
||||||
|
'is not supported'
|
||||||
|
) % bad_locale
|
||||||
|
|
||||||
|
raise RuntimeError('ocrmypdf will abort further execution because Python 3 '
|
||||||
|
'was configured to use ASCII as encoding for the '
|
||||||
|
'environment.' + extra)
|
||||||
@@ -4,9 +4,40 @@
|
|||||||
"""Wrappers to manage subprocess calls"""
|
"""Wrappers to manage subprocess calls"""
|
||||||
|
|
||||||
import os
|
import os
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from subprocess import run, STDOUT, PIPE, CalledProcessError
|
||||||
|
from ..exceptions import MissingDependencyError
|
||||||
|
|
||||||
|
|
||||||
def get_program(name):
|
def get_program(name):
|
||||||
"Check environment variables for overrides to this program"
|
"Check environment variables for overrides to this program"
|
||||||
envvar = 'OCRMYPDF_' + name.upper()
|
envvar = 'OCRMYPDF_' + name.upper()
|
||||||
return os.environ.get(envvar, name)
|
return os.environ.get(envvar, name)
|
||||||
|
|
||||||
|
|
||||||
|
def get_version(program, *,
|
||||||
|
version_arg='--version', regex=r'(\d+(\.\d+)*)'):
|
||||||
|
"Get the version of the specified program, "
|
||||||
|
args_prog = [
|
||||||
|
get_program(program),
|
||||||
|
version_arg
|
||||||
|
]
|
||||||
|
try:
|
||||||
|
proc = run(
|
||||||
|
args_prog, close_fds=True, universal_newlines=True,
|
||||||
|
stdout=PIPE, stderr=STDOUT, check=True)
|
||||||
|
output = proc.stdout
|
||||||
|
except CalledProcessError as e:
|
||||||
|
raise MissingDependencyError(
|
||||||
|
"Could not find program '{}' on the PATH".format(program)) from e
|
||||||
|
|
||||||
|
try:
|
||||||
|
version = re.match(regex, output.strip()).group(1)
|
||||||
|
except AttributeError as e:
|
||||||
|
raise MissingDependencyError(
|
||||||
|
("The program '{}' did not report its version. "
|
||||||
|
"Message was:\n{}").format(program, output)
|
||||||
|
)
|
||||||
|
|
||||||
|
return version
|
||||||
@@ -8,28 +8,14 @@ from functools import lru_cache
|
|||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
from . import get_program
|
from . import get_program, get_version
|
||||||
from ..exceptions import SubprocessOutputError, MissingDependencyError
|
from ..exceptions import SubprocessOutputError, MissingDependencyError
|
||||||
from ..helpers import fspath
|
from ..helpers import fspath
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=1)
|
@lru_cache(maxsize=1)
|
||||||
def version():
|
def version():
|
||||||
args_gs = [
|
return get_version('gs')
|
||||||
get_program('gs'),
|
|
||||||
'--version'
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
proc = run(
|
|
||||||
args_gs, close_fds=True, universal_newlines=True,
|
|
||||||
stdout=PIPE, stderr=STDOUT, check=True)
|
|
||||||
ver = proc.stdout
|
|
||||||
except CalledProcessError as e:
|
|
||||||
print("Could not find Ghostscript executable on system PATH.",
|
|
||||||
file=sys.stderr)
|
|
||||||
raise MissingDependencyError from e
|
|
||||||
|
|
||||||
return ver.strip()
|
|
||||||
|
|
||||||
|
|
||||||
def _gs_error_reported(stream):
|
def _gs_error_reported(stream):
|
||||||
|
|||||||
+2
-15
@@ -9,25 +9,12 @@ import re
|
|||||||
|
|
||||||
from ..exceptions import InputFileError, SubprocessOutputError, \
|
from ..exceptions import InputFileError, SubprocessOutputError, \
|
||||||
MissingDependencyError, EncryptedPdfError
|
MissingDependencyError, EncryptedPdfError
|
||||||
from . import get_program
|
from . import get_program, get_version
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=1)
|
@lru_cache(maxsize=1)
|
||||||
def version():
|
def version():
|
||||||
args_qpdf = [
|
return get_version('qpdf', regex=r'qpdf version (.+)')
|
||||||
get_program('qpdf'),
|
|
||||||
'--version'
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
p = run(args_qpdf, universal_newlines=True, stderr=STDOUT,
|
|
||||||
stdout=PIPE)
|
|
||||||
except CalledProcessError as e:
|
|
||||||
print("Could not find qpdf executable on system PATH.",
|
|
||||||
file=sys.stderr)
|
|
||||||
raise MissingDependencyError() from e
|
|
||||||
|
|
||||||
qpdf_version = re.match(r'qpdf version (.+)', p.stdout).group(1)
|
|
||||||
return qpdf_version
|
|
||||||
|
|
||||||
|
|
||||||
def check(input_file, log=None):
|
def check(input_file, log=None):
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ from subprocess import PIPE, CalledProcessError, \
|
|||||||
|
|
||||||
from ..exceptions import MissingDependencyError, TesseractConfigError
|
from ..exceptions import MissingDependencyError, TesseractConfigError
|
||||||
from ..helpers import page_number
|
from ..helpers import page_number
|
||||||
from . import get_program
|
from . import get_program, get_version
|
||||||
|
|
||||||
OrientationConfidence = namedtuple(
|
OrientationConfidence = namedtuple(
|
||||||
'OrientationConfidence',
|
'OrientationConfidence',
|
||||||
@@ -40,21 +40,7 @@ HOCR_TEMPLATE = """<?xml version="1.0" encoding="UTF-8"?>
|
|||||||
|
|
||||||
@lru_cache(maxsize=1)
|
@lru_cache(maxsize=1)
|
||||||
def version():
|
def version():
|
||||||
args_tess = [
|
return get_version('tesseract', regex=r'tesseract\s(.+)')
|
||||||
get_program('tesseract'),
|
|
||||||
'--version'
|
|
||||||
]
|
|
||||||
try:
|
|
||||||
versions = check_output(
|
|
||||||
args_tess, close_fds=True, universal_newlines=True,
|
|
||||||
stderr=STDOUT)
|
|
||||||
except CalledProcessError as e:
|
|
||||||
print("Could not find Tesseract executable on system PATH.",
|
|
||||||
file=sys.stderr)
|
|
||||||
raise MissingDependencyError from e
|
|
||||||
|
|
||||||
tesseract_version = re.match(r'tesseract\s(.+)', versions).group(1)
|
|
||||||
return tesseract_version
|
|
||||||
|
|
||||||
|
|
||||||
def v4():
|
def v4():
|
||||||
|
|||||||
@@ -9,19 +9,7 @@ import sys
|
|||||||
import os
|
import os
|
||||||
from functools import lru_cache
|
from functools import lru_cache
|
||||||
from ..exceptions import MissingDependencyError
|
from ..exceptions import MissingDependencyError
|
||||||
from . import get_program
|
from . import get_program, get_version
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=1)
|
|
||||||
def version():
|
|
||||||
args_unpaper = [
|
|
||||||
get_program('unpaper'),
|
|
||||||
'--version'
|
|
||||||
]
|
|
||||||
ver = check_output(
|
|
||||||
args_unpaper, close_fds=True, universal_newlines=True,
|
|
||||||
stderr=STDOUT, timeout=5)
|
|
||||||
return ver.strip()
|
|
||||||
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
@@ -31,6 +19,11 @@ except ImportError:
|
|||||||
raise
|
raise
|
||||||
|
|
||||||
|
|
||||||
|
@lru_cache(maxsize=1)
|
||||||
|
def version():
|
||||||
|
return get_version('unpaper')
|
||||||
|
|
||||||
|
|
||||||
def run(input_file, output_file, dpi, log, mode_args):
|
def run(input_file, output_file, dpi, log, mode_args):
|
||||||
args_unpaper = [
|
args_unpaper = [
|
||||||
get_program('unpaper'),
|
get_program('unpaper'),
|
||||||
|
|||||||
+1
-1
@@ -32,7 +32,7 @@ def re_symlink(input_file, soft_link_name, log=None):
|
|||||||
"%s exists and is not a link" % soft_link_name)
|
"%s exists and is not a link" % soft_link_name)
|
||||||
try:
|
try:
|
||||||
os.unlink(soft_link_name)
|
os.unlink(soft_link_name)
|
||||||
except:
|
except OSError:
|
||||||
prdebug("Can't unlink %s" % (soft_link_name))
|
prdebug("Can't unlink %s" % (soft_link_name))
|
||||||
|
|
||||||
if not os.path.exists(input_file):
|
if not os.path.exists(input_file):
|
||||||
|
|||||||
+3
-3
@@ -2,8 +2,8 @@
|
|||||||
# setup.py lists a separate set of requirements that are looser to simplify
|
# setup.py lists a separate set of requirements that are looser to simplify
|
||||||
# installation
|
# installation
|
||||||
ruffus == 2.6.3
|
ruffus == 2.6.3
|
||||||
Pillow == 4.1.1
|
Pillow == 4.3.0
|
||||||
reportlab == 3.4.0
|
reportlab == 3.4.0
|
||||||
PyPDF2 == 1.26.0
|
PyPDF2 == 1.26.0
|
||||||
img2pdf == 0.2.3
|
img2pdf == 0.2.4
|
||||||
cffi == 1.10.0
|
cffi == 1.11.2
|
||||||
|
|||||||
@@ -217,6 +217,7 @@ setup(
|
|||||||
"Topic :: Text Processing :: Indexing",
|
"Topic :: Text Processing :: Indexing",
|
||||||
"Topic :: Text Processing :: Linguistic",
|
"Topic :: Text Processing :: Linguistic",
|
||||||
],
|
],
|
||||||
|
python_requires='>=3.5',
|
||||||
setup_requires=[
|
setup_requires=[
|
||||||
'setuptools_scm', # so that version will work
|
'setuptools_scm', # so that version will work
|
||||||
'cffi>=1.9.1' # to build the leptonica module
|
'cffi>=1.9.1' # to build the leptonica module
|
||||||
|
|||||||
@@ -1005,3 +1005,15 @@ def test_pdfa_1(spoof_tesseract_cache, resources, outpdf):
|
|||||||
|
|
||||||
pdfa_info = file_claims_pdfa(outpdf)
|
pdfa_info = file_claims_pdfa(outpdf)
|
||||||
assert pdfa_info['conformance'] == 'PDF/A-1B'
|
assert pdfa_info['conformance'] == 'PDF/A-1B'
|
||||||
|
|
||||||
|
|
||||||
|
def test_bad_locale():
|
||||||
|
env = os.environ.copy()
|
||||||
|
env['LC_ALL'] = 'C'
|
||||||
|
|
||||||
|
p, out, err = run_ocrmypdf(
|
||||||
|
'a', 'b', env=env
|
||||||
|
)
|
||||||
|
assert out == '', "stdout not clean"
|
||||||
|
assert p.returncode != 0
|
||||||
|
assert 'configured to use ASCII as encoding' in err, "should whine"
|
||||||
Reference in New Issue
Block a user