Compare commits

..
14 Commits
Author SHA1 Message Date
James R. Barlow 7982f58b2e Try tweaking Dockerfile for automated build again 2016-02-05 01:38:59 -08:00
James R. Barlow e805c1908a Minor fix for Dockerfile polyglot 2016-02-05 00:52:27 -08:00
James R. Barlow 344fc40cbc Merge branch 'release/v3.2' 2016-02-05 00:10:41 -08:00
James R. Barlow 7e5c37137b Merge branch 'develop' into release/v3.2 2016-02-04 23:42:06 -08:00
James R. Barlow 1aae11714b Update release notes for v3.2 2016-02-04 23:41:33 -08:00
James R. Barlow d82f14a7aa Update .gitignore 2016-02-04 18:51:41 -08:00
James R. Barlow 4b65e0b093 Set JPEG output quality to 95 for better transcoding 2016-02-04 18:49:09 -08:00
James R. Barlow 43b0faa830 Bug in tesseract_noop spoof: produced wrong page sizes
Now checks input image to ensure the implied page size of its .hocr file
matches the rest of the PDF.
2016-02-04 18:48:22 -08:00
James R. Barlow 8674c9fb20 Merge commit 'ccfbb54e8c26784e438ba2fcac2179f21e7d857b' into release/v3.2 2016-02-04 17:39:36 -08:00
jbarlow83 ccfbb54e8c Update release notes for v3.2
Fix the notes
2016-02-04 17:37:30 -08:00
James R. Barlow 9893ebf889 Suppress tesseract argument printout 2016-02-04 17:26:36 -08:00
James R. Barlow 303eb3e93a Merge commit 'ca546d70e5bff9e9b115371f7813f3c326822bd8' into release/v3.2 2016-02-04 17:25:56 -08:00
jbarlow83 ca546d70e5 Merge pull request #45 from spwhitton/hocrtransform-shebang-fix
fix shebang in hocrtransform.py
2016-02-04 17:21:33 -08:00
Sean Whitton 6a5ea2d64a fix shebang in hocrtransform.py 2016-02-03 17:48:35 -07:00
9 changed files with 62 additions and 39 deletions
+22 -17
View File
@@ -1,26 +1,31 @@
tmp/ # Development environment
log/
*.pyc *.pyc
tests/output/
.ruffus_history.sqlite
*.sublime-* *.sublime-*
/*.pdf
build/
dist/
*.egg-info/
venv/
venv-3.4/ venv-3.4/
venv-3.5/ venv-3.5/
*/test/output venv/
bin/
include/
lib/
pip-selfcheck.json
pyvenv.cfg pyvenv.cfg
htmlcov/
.coverage # Package building
*.egg-info/
.cache/ .cache/
.eggs/
build/
dist/
# Automatically generated files
ocrmypdf/lib/_*.py
ocrmypdf/version.py
# Code coverage
.coverage
htmlcov/
# Testing
log/
/*.pdf
.ipynb_checkpoints/ .ipynb_checkpoints/
tests/cache/ tests/cache/
tests/output/
tests/resources/private tests/resources/private
ocrmypdf/version.py tmp/
+13 -7
View File
@@ -10,20 +10,26 @@ RUN useradd docker \
&& chown docker:docker /home/docker && chown docker:docker /home/docker
# Update system and install our dependencies # Update system and install our dependencies
# If this command takes too Docker hub's automated build will timeout,
# so try it in portions
RUN apt-get update && apt-get install -y --no-install-recommends \ RUN apt-get update && apt-get install -y --no-install-recommends \
locales \ locales \
ghostscript \
tesseract-ocr \
tesseract-ocr-deu tesseract-ocr-spa tesseract-ocr-eng tesseract-ocr-fra \
qpdf \
poppler-utils \
python3 \ python3 \
python3-pip \ python3-pip \
python3-venv \ python3-venv \
python3-reportlab \ python3-reportlab \
python3-pil \ python3-pil \
python3-wheel \ python3-wheel
unpaper
RUN apt-get install -y --no-install-recommends \
unpaper \
ghostscript \
qpdf \
poppler-utils
RUN apt-get install -y --no-install-recommends \
tesseract-ocr \
tesseract-ocr-deu tesseract-ocr-spa tesseract-ocr-eng tesseract-ocr-fra
# Enforce UTF-8 # Enforce UTF-8
# Borrowed from https://index.docker.io/u/crosbymichael/python/ # Borrowed from https://index.docker.io/u/crosbymichael/python/
+3
View File
@@ -5,9 +5,12 @@ FROM jbarlow83/ocrmypdf:latest
MAINTAINER James R. Barlow <jim@purplerock.ca> MAINTAINER James R. Barlow <jim@purplerock.ca>
# Update system and install our dependencies # Update system and install our dependencies
USER root
RUN apt-get update && apt-get install -y --no-install-recommends \ RUN apt-get update && apt-get install -y --no-install-recommends \
tesseract-ocr-all tesseract-ocr-all
USER docker
# Must use array form of ENTRYPOINT # Must use array form of ENTRYPOINT
# Non-array form does not append other arguments, because that is "intuitive" # Non-array form does not append other arguments, because that is "intuitive"
ENTRYPOINT ["/application/docker-wrapper.sh"] ENTRYPOINT ["/application/docker-wrapper.sh"]
+1 -6
View File
@@ -11,12 +11,7 @@ Main features
`PDF/A <https://en.wikipedia.org/?title=PDF/A>`__ file from a regular PDF `PDF/A <https://en.wikipedia.org/?title=PDF/A>`__ file from a regular PDF
- Places OCR text accurately below the image to ease copy / paste - Places OCR text accurately below the image to ease copy / paste
- Keeps the exact resolution of the original embedded images - Keeps the exact resolution of the original embedded images
- When possible, inserts OCR information as a "lossless" operation without rendering vector information
- or if requested oversamples the images before OCRing so as to get
better results
- When possible, inserts OCR information as a "lossless" operation without transcoding
images or rendering vector information
- Keeps file size about the same - Keeps file size about the same
- If requested deskews and/or cleans the image before performing OCR - If requested deskews and/or cleans the image before performing OCR
- Validates input and output files - Validates input and output files
+15 -5
View File
@@ -6,16 +6,26 @@ Please always read this file before installing the package
Download software here: https://github.com/jbarlow83/OCRmyPDF/tags Download software here: https://github.com/jbarlow83/OCRmyPDF/tags
v3.2-rc1: v3.2:
========= =========
New features New features
------------ ------------
- Lossless reconstruction: when possible, OCRmyPDF will inject text layers without transcoding - Lossless reconstruction: when possible, OCRmyPDF will inject text layers without
images or otherwise manipulating the PDF. otherwise manipulating the content and layout of a PDF page. For example, a PDF containing a mix
- New argument --tesseract-pagesegmode allows you to pass page segmentation arguments to Tesseract OCR. of vector and raster content would see the vector content preserved. Images may still be transcoded
during PDF/A conversion. (``--deskew`` and ``--clean-final`` disable this mode, necessarily.)
- New argument ``--tesseract-pagesegmode`` allows you to pass page segmentation arguments to Tesseract OCR.
This helps for two column text and other situations that confuse Tesseract. This helps for two column text and other situations that confuse Tesseract.
- Added a new "polyglot" version of the Docker image, that generates Tesseract with all languages packs installed,
for the polyglots among us. It is much larger.
Changes
-------
- JPEG transcoding quality is now 95 instead of the default 75. Bigger file sizes for less degradation.
v3.1.1: v3.1.1:
@@ -40,7 +50,7 @@ Changes
- Fixed an issue where OCRmyPDF failed to text that certain pages contained previously OCR'ed text, - Fixed an issue where OCRmyPDF failed to text that certain pages contained previously OCR'ed text,
such as OCR text produced by Tesseract 3.04 such as OCR text produced by Tesseract 3.04
- Inserts /Creator tag into PDFs so that errors can be traced back to this project - Inserts /Creator tag into PDFs so that errors can be traced back to this project
- Added new option --pdf-renderer=auto, to let OCRmyPDF pick the best PDF renderer. - Added new option ``--pdf-renderer=auto``, to let OCRmyPDF pick the best PDF renderer.
Currently it always chooses the 'hocrtransform' renderer but that behavior may change. Currently it always chooses the 'hocrtransform' renderer but that behavior may change.
- Set up Travis CI automatic integration testing - Set up Travis CI automatic integration testing
+1
View File
@@ -45,6 +45,7 @@ def generate_pdfa(pdf_pages, output_file, threads=1):
"-sDEVICE=pdfwrite", "-sDEVICE=pdfwrite",
"-sColorConversionStrategy=/RGB", "-sColorConversionStrategy=/RGB",
"-sProcessColorModel=DeviceRGB", "-sProcessColorModel=DeviceRGB",
"-dJPEGQ=95",
"-dPDFA=2", "-dPDFA=2",
"-sPDFACompatibilityPolicy=2", "-sPDFACompatibilityPolicy=2",
"-sOutputICCProfile=srgb.icc", "-sOutputICCProfile=srgb.icc",
+1 -1
View File
@@ -1,4 +1,4 @@
#!/usr/local/bin/python3 #!/usr/bin/env python3
############################################################################## ##############################################################################
# Copyright (c) 2013-14: fritz-hh from Github # Copyright (c) 2013-14: fritz-hh from Github
# (https://github.com/fritz-hh) # (https://github.com/fritz-hh)
-1
View File
@@ -94,7 +94,6 @@ def generate_hocr(input_file, output_hocr, language: list, tessconfig: list,
badxml, badxml,
'hocr' 'hocr'
] + tessconfig) ] + tessconfig)
print(args_tesseract)
p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE, p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE,
universal_newlines=True) universal_newlines=True)
try: try:
+6 -2
View File
@@ -1,6 +1,7 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
import sys import sys
import img2pdf import img2pdf
from PIL import Image
VERSION_STRING = '''tesseract 3.04.00 VERSION_STRING = '''tesseract 3.04.00
@@ -40,9 +41,12 @@ def main():
print('List of available languages (1):\neng', file=sys.stderr) print('List of available languages (1):\neng', file=sys.stderr)
sys.exit(0) sys.exit(0)
elif sys.argv[-1] == 'hocr': elif sys.argv[-1] == 'hocr':
inputf = sys.argv[-3]
output = sys.argv[-2] output = sys.argv[-2]
with open(output + '.hocr', 'w', encoding='utf-8') as f: with Image.open(inputf) as im, \
f.write(HOCR_TEMPLATE.format('1000', '1000')) open(output + '.hocr', 'w', encoding='utf-8') as f:
w, h = im.size
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
elif sys.argv[-1] == 'pdf': elif sys.argv[-1] == 'pdf':
inputf = sys.argv[-3] inputf = sys.argv[-3]
output = sys.argv[-2] output = sys.argv[-2]