Merge branch 'feature/docker-debian'
This commit is contained in:
@@ -0,0 +1,21 @@
|
||||
bin/
|
||||
build/
|
||||
dist/
|
||||
include/
|
||||
lib/
|
||||
ocrmypdf.egg-info/
|
||||
staging/
|
||||
.git/
|
||||
.ruffus_history.sqlite
|
||||
MANIFEST.in
|
||||
*.sublime*
|
||||
*.pdf
|
||||
*.rst
|
||||
*.pyc
|
||||
*/*.pyc
|
||||
*/*/*.pyc
|
||||
*/*/*/*.pyc
|
||||
*/*/*/*/*.pyc
|
||||
*/*/*/*/*/*.pyc
|
||||
*/*/*/*/*/*/*.pyc
|
||||
*/*/*/*/*/*/*/*.pyc
|
||||
+87
@@ -0,0 +1,87 @@
|
||||
# OCRmyPDF
|
||||
#
|
||||
# VERSION 3.0.0
|
||||
FROM debian:stretch
|
||||
MAINTAINER James R. Barlow <jim@purplerock.ca>
|
||||
|
||||
# Add unprivileged user
|
||||
RUN useradd docker \
|
||||
&& mkdir /home/docker \
|
||||
&& chown docker:docker /home/docker
|
||||
|
||||
# Update system and install our dependencies
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
locales \
|
||||
ghostscript \
|
||||
tesseract-ocr \
|
||||
tesseract-ocr-deu tesseract-ocr-spa tesseract-ocr-eng tesseract-ocr-fra \
|
||||
qpdf \
|
||||
poppler-utils \
|
||||
python3 \
|
||||
python3-pip \
|
||||
python3-venv \
|
||||
python3-reportlab \
|
||||
python3-pil
|
||||
|
||||
# Enforce UTF-8
|
||||
# Borrowed from https://index.docker.io/u/crosbymichael/python/
|
||||
RUN dpkg-reconfigure locales && \
|
||||
locale-gen C.UTF-8 && \
|
||||
/usr/sbin/update-locale LANG=C.UTF-8
|
||||
ENV LC_ALL C.UTF-8
|
||||
|
||||
# Build unpaper 6.1
|
||||
RUN apt-get install -y \
|
||||
wget \
|
||||
gcc \
|
||||
libavformat-dev \
|
||||
libavcodec-dev \
|
||||
libavutil-dev \
|
||||
autoconf \
|
||||
automake \
|
||||
make \
|
||||
pkg-config \
|
||||
xsltproc
|
||||
|
||||
WORKDIR /root
|
||||
RUN wget https://github.com/Flameeyes/unpaper/archive/unpaper-6.1.tar.gz
|
||||
RUN tar xf unpaper-6.1.tar.gz
|
||||
WORKDIR /root/unpaper-unpaper-6.1
|
||||
RUN autoreconf -i
|
||||
RUN ./configure CFLAGS="-O2 -march=native -pipe -flto"
|
||||
RUN make -j install
|
||||
|
||||
RUN apt-get remove -y \
|
||||
gcc \
|
||||
autoconf \
|
||||
automake \
|
||||
pkg-config \
|
||||
xsltproc \
|
||||
make
|
||||
RUN apt-get autoremove -y && apt-get clean -y
|
||||
RUN rm -rf /var/lib/apt/lists/* /tmp/* /var/tmp/*
|
||||
|
||||
# Set up a Python virtualenv and take all of the system packages, so we can
|
||||
# rely on the platform packages rather than importing GCC and compiling them
|
||||
RUN pyvenv /appenv \
|
||||
&& pyvenv --system-site-packages /appenv
|
||||
|
||||
COPY . /application/
|
||||
|
||||
# Install application and dependencies
|
||||
# In this arrangement Pillow and reportlab will be provided by the system
|
||||
RUN . /appenv/bin/activate; \
|
||||
pip install --upgrade pip \
|
||||
&& pip install --no-cache-dir /application \
|
||||
&& pip install --no-cache-dir -r /application/test_requirements.txt
|
||||
|
||||
USER docker
|
||||
WORKDIR /home/docker
|
||||
|
||||
ENV DEFAULT_RUFFUS_HISTORY_FILE=/tmp/.{basename}.ruffus_history.sqlite
|
||||
ENV OCRMYPDF_TEST_OUTPUT=/tmp/test-output
|
||||
ENV OCRMYPDF_IN_DOCKER=1
|
||||
|
||||
# Must use array form of ENTRYPOINT
|
||||
# Non-array form does not append other arguments, because that is "intuitive"
|
||||
ENTRYPOINT ["/application/docker-wrapper.sh"]
|
||||
Executable
+5
@@ -0,0 +1,5 @@
|
||||
#!/bin/bash
|
||||
|
||||
. /appenv/bin/activate
|
||||
cd /home/docker
|
||||
exec ocrmypdf "$@"
|
||||
+16
-13
@@ -5,7 +5,6 @@ from contextlib import suppress
|
||||
from tempfile import NamedTemporaryFile, mkdtemp
|
||||
import sys
|
||||
import os
|
||||
import fileinput
|
||||
import re
|
||||
import shutil
|
||||
import warnings
|
||||
@@ -212,7 +211,7 @@ if not set(options.language).issubset(tesseract.languages()):
|
||||
"The installed version of tesseract does not have language "
|
||||
"data for the following requested languages: ")
|
||||
for lang in (set(options.language) - tesseract.languages()):
|
||||
complain(lang, file=sys.stderr)
|
||||
complain(lang)
|
||||
sys.exit(ExitCode.bad_args)
|
||||
|
||||
|
||||
@@ -545,11 +544,13 @@ def ocr_tesseract_hocr(
|
||||
|
||||
pageinfo = get_pageinfo(input_file, pdfinfo, pdfinfo_lock)
|
||||
|
||||
badxml = os.path.splitext(output_file)[0] + '.badxml'
|
||||
|
||||
args_tesseract = [
|
||||
'tesseract',
|
||||
'-l', '+'.join(options.language),
|
||||
input_file,
|
||||
output_file,
|
||||
badxml,
|
||||
'hocr'
|
||||
] + options.tesseract_config
|
||||
p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE,
|
||||
@@ -575,24 +576,26 @@ def ocr_tesseract_hocr(
|
||||
if p.returncode != 0:
|
||||
raise CalledProcessError(p.returncode, args_tesseract)
|
||||
|
||||
if os.path.exists(output_file + '.html'):
|
||||
# Tesseract 3.02 appends suffix ".html" on its own (.hocr.html)
|
||||
shutil.move(output_file + '.html', output_file)
|
||||
elif os.path.exists(output_file + '.hocr'):
|
||||
# Tesseract 3.03 appends suffix ".hocr" on its own (.hocr.hocr)
|
||||
shutil.move(output_file + '.hocr', output_file)
|
||||
if os.path.exists(badxml + '.html'):
|
||||
# Tesseract 3.02 appends suffix ".html" on its own (.badxml.html)
|
||||
shutil.move(badxml + '.html', badxml)
|
||||
elif os.path.exists(badxml + '.hocr'):
|
||||
# Tesseract 3.03 appends suffix ".hocr" on its own (.badxml.hocr)
|
||||
shutil.move(badxml + '.hocr', badxml)
|
||||
|
||||
# Tesseract 3.03 inserts source filename into hocr file without
|
||||
# escaping it, creating invalid XML and breaking the parser.
|
||||
# As a workaround, rewrite the hocr file, replacing the filename
|
||||
# with a space.
|
||||
# with a space. Don't know if Tesseract 3.02 does the same.
|
||||
|
||||
regex_nested_single_quotes = re.compile(
|
||||
r"""title='image "([^"]*)";""")
|
||||
with fileinput.input(files=(output_file,), inplace=True) as f:
|
||||
for line in f:
|
||||
with open(badxml, mode='r', encoding='utf-8') as f_in, \
|
||||
open(output_file, mode='w', encoding='utf-8') as f_out:
|
||||
for line in f_in:
|
||||
line = regex_nested_single_quotes.sub(
|
||||
r"""title='image " ";""", line)
|
||||
print(line, end='') # fileinput.input redirects stdout
|
||||
f_out.write(line)
|
||||
|
||||
|
||||
@active_if(options.pdf_renderer == 'hocr')
|
||||
|
||||
@@ -7,7 +7,6 @@ from PIL import Image
|
||||
from tempfile import NamedTemporaryFile
|
||||
from contextlib import suppress
|
||||
import os
|
||||
import sys
|
||||
import shutil
|
||||
import pytest
|
||||
import img2pdf
|
||||
@@ -15,7 +14,9 @@ from pkg_resources import Requirement, resource_filename
|
||||
|
||||
req = Requirement.parse('ocrmypdf')
|
||||
|
||||
TEST_OUTPUT = os.path.join(os.path.dirname(__file__), 'output')
|
||||
TEST_OUTPUT = os.environ.get(
|
||||
'OCRMYPDF_TEST_OUTPUT',
|
||||
default=os.path.join(os.path.dirname(__file__), 'output'))
|
||||
|
||||
|
||||
def setup_module():
|
||||
|
||||
@@ -0,0 +1,4 @@
|
||||
ruffus>=2.6.3
|
||||
Pillow>=2.4.0
|
||||
reportlab>=3.1.44
|
||||
PyPDF2>=1.25.1
|
||||
@@ -158,7 +158,7 @@ if command.startswith('install') or \
|
||||
)
|
||||
check_external_program(
|
||||
program='unpaper',
|
||||
need_version='6.1',
|
||||
need_version='0.4.2',
|
||||
package='unpaper',
|
||||
optional=True
|
||||
)
|
||||
@@ -174,6 +174,9 @@ if 'upload' in sys.argv[1:]:
|
||||
print('Use twine to upload the package - setup.py upload is insecure')
|
||||
sys.exit(1)
|
||||
|
||||
install_requires = open('requirements.txt').read().splitlines()
|
||||
tests_require = open('test_requirements.txt').read().splitlines()
|
||||
|
||||
setup(
|
||||
name='ocrmypdf',
|
||||
version='3.0rc5', # also update: release notes, main.py
|
||||
@@ -200,16 +203,8 @@ setup(
|
||||
"Topic :: Text Processing :: Indexing",
|
||||
"Topic :: Text Processing :: Linguistic",
|
||||
],
|
||||
install_requires=[
|
||||
'ruffus>=2.6.3',
|
||||
'Pillow>=2.4.0',
|
||||
'reportlab>=3.1.44',
|
||||
'PyPDF2>=1.25.1'
|
||||
],
|
||||
tests_require=[
|
||||
'img2pdf>=0.1.5',
|
||||
'pytest>=2.7.2'
|
||||
],
|
||||
install_requires=install_requires,
|
||||
tests_require=tests_require,
|
||||
entry_points={
|
||||
'console_scripts': [
|
||||
'ocrmypdf = ocrmypdf.main:run_pipeline'
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
img2pdf>=0.1.5
|
||||
pytest>=2.7.2
|
||||
File diff suppressed because one or more lines are too long
+17
-2
@@ -7,7 +7,6 @@ import os
|
||||
import shutil
|
||||
from contextlib import suppress
|
||||
import sys
|
||||
from unittest.mock import patch, create_autospec
|
||||
import pytest
|
||||
from ocrmypdf.pageinfo import pdf_get_all_pageinfo
|
||||
import PyPDF2 as pypdf
|
||||
@@ -22,7 +21,9 @@ TESTS_ROOT = os.path.abspath(os.path.dirname(__file__))
|
||||
PROJECT_ROOT = os.path.dirname(TESTS_ROOT)
|
||||
OCRMYPDF = os.path.join(PROJECT_ROOT, 'OCRmyPDF.sh')
|
||||
TEST_RESOURCES = os.path.join(PROJECT_ROOT, 'tests', 'resources')
|
||||
TEST_OUTPUT = os.path.join(PROJECT_ROOT, 'tests', 'output')
|
||||
TEST_OUTPUT = os.environ.get(
|
||||
'OCRMYPDF_TEST_OUTPUT',
|
||||
default=os.path.join(PROJECT_ROOT, 'tests', 'output'))
|
||||
TEST_BINARY_PATH = os.path.join(TEST_OUTPUT, 'fakebin')
|
||||
|
||||
|
||||
@@ -264,6 +265,8 @@ def break_ghostscript_pdfa():
|
||||
return override_binary('gs', 'replace_ghostscript_nopdfa.py')
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.environ.get('OCRMYPDF_IN_DOCKER', False),
|
||||
reason="Requires writable filesystem")
|
||||
def test_ghostscript_pdfa_fails(break_ghostscript_pdfa):
|
||||
env = os.environ.copy()
|
||||
env['PATH'] = break_ghostscript_pdfa
|
||||
@@ -293,3 +296,15 @@ def test_blank_input_pdf():
|
||||
'blank.pdf', 'still_blank.pdf')
|
||||
assert p.returncode == ExitCode.ok
|
||||
|
||||
|
||||
def test_french():
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
'francais.pdf', 'francais.pdf', '-l', 'fra')
|
||||
assert p.returncode == ExitCode.ok, \
|
||||
"This test may fail if Tesseract language packs are missing"
|
||||
|
||||
|
||||
def test_klingon():
|
||||
p, out, err = run_ocrmypdf_env(
|
||||
'francais.pdf', 'francais.pdf', '-l', 'klz')
|
||||
assert p.returncode == ExitCode.bad_args
|
||||
|
||||
Reference in New Issue
Block a user