Compare commits

..
13 Commits
8 changed files with 49 additions and 32 deletions
+22 -17
View File
@@ -1,26 +1,31 @@
tmp/ # Development environment
log/
*.pyc *.pyc
tests/output/
.ruffus_history.sqlite
*.sublime-* *.sublime-*
/*.pdf
build/
dist/
*.egg-info/
venv/
venv-3.4/ venv-3.4/
venv-3.5/ venv-3.5/
*/test/output venv/
bin/
include/
lib/
pip-selfcheck.json
pyvenv.cfg pyvenv.cfg
htmlcov/
.coverage # Package building
*.egg-info/
.cache/ .cache/
.eggs/
build/
dist/
# Automatically generated files
ocrmypdf/lib/_*.py
ocrmypdf/version.py
# Code coverage
.coverage
htmlcov/
# Testing
log/
/*.pdf
.ipynb_checkpoints/ .ipynb_checkpoints/
tests/cache/ tests/cache/
tests/output/
tests/resources/private tests/resources/private
ocrmypdf/version.py tmp/
+3
View File
@@ -5,9 +5,12 @@ FROM jbarlow83/ocrmypdf:latest
MAINTAINER James R. Barlow <jim@purplerock.ca> MAINTAINER James R. Barlow <jim@purplerock.ca>
# Update system and install our dependencies # Update system and install our dependencies
USER root
RUN apt-get update && apt-get install -y --no-install-recommends \ RUN apt-get update && apt-get install -y --no-install-recommends \
tesseract-ocr-all tesseract-ocr-all
USER docker
# Must use array form of ENTRYPOINT # Must use array form of ENTRYPOINT
# Non-array form does not append other arguments, because that is "intuitive" # Non-array form does not append other arguments, because that is "intuitive"
ENTRYPOINT ["/application/docker-wrapper.sh"] ENTRYPOINT ["/application/docker-wrapper.sh"]
+1 -6
View File
@@ -11,12 +11,7 @@ Main features
`PDF/A <https://en.wikipedia.org/?title=PDF/A>`__ file from a regular PDF `PDF/A <https://en.wikipedia.org/?title=PDF/A>`__ file from a regular PDF
- Places OCR text accurately below the image to ease copy / paste - Places OCR text accurately below the image to ease copy / paste
- Keeps the exact resolution of the original embedded images - Keeps the exact resolution of the original embedded images
- When possible, inserts OCR information as a "lossless" operation without rendering vector information
- or if requested oversamples the images before OCRing so as to get
better results
- When possible, inserts OCR information as a "lossless" operation without transcoding
images or rendering vector information
- Keeps file size about the same - Keeps file size about the same
- If requested deskews and/or cleans the image before performing OCR - If requested deskews and/or cleans the image before performing OCR
- Validates input and output files - Validates input and output files
+15 -5
View File
@@ -6,16 +6,26 @@ Please always read this file before installing the package
Download software here: https://github.com/jbarlow83/OCRmyPDF/tags Download software here: https://github.com/jbarlow83/OCRmyPDF/tags
v3.2-rc1: v3.2:
========= =========
New features New features
------------ ------------
- Lossless reconstruction: when possible, OCRmyPDF will inject text layers without transcoding - Lossless reconstruction: when possible, OCRmyPDF will inject text layers without
images or otherwise manipulating the PDF. otherwise manipulating the content and layout of a PDF page. For example, a PDF containing a mix
- New argument --tesseract-pagesegmode allows you to pass page segmentation arguments to Tesseract OCR. of vector and raster content would see the vector content preserved. Images may still be transcoded
during PDF/A conversion. (``--deskew`` and ``--clean-final`` disable this mode, necessarily.)
- New argument ``--tesseract-pagesegmode`` allows you to pass page segmentation arguments to Tesseract OCR.
This helps for two column text and other situations that confuse Tesseract. This helps for two column text and other situations that confuse Tesseract.
- Added a new "polyglot" version of the Docker image, that generates Tesseract with all languages packs installed,
for the polyglots among us. It is much larger.
Changes
-------
- JPEG transcoding quality is now 95 instead of the default 75. Bigger file sizes for less degradation.
v3.1.1: v3.1.1:
@@ -40,7 +50,7 @@ Changes
- Fixed an issue where OCRmyPDF failed to text that certain pages contained previously OCR'ed text, - Fixed an issue where OCRmyPDF failed to text that certain pages contained previously OCR'ed text,
such as OCR text produced by Tesseract 3.04 such as OCR text produced by Tesseract 3.04
- Inserts /Creator tag into PDFs so that errors can be traced back to this project - Inserts /Creator tag into PDFs so that errors can be traced back to this project
- Added new option --pdf-renderer=auto, to let OCRmyPDF pick the best PDF renderer. - Added new option ``--pdf-renderer=auto``, to let OCRmyPDF pick the best PDF renderer.
Currently it always chooses the 'hocrtransform' renderer but that behavior may change. Currently it always chooses the 'hocrtransform' renderer but that behavior may change.
- Set up Travis CI automatic integration testing - Set up Travis CI automatic integration testing
+1
View File
@@ -45,6 +45,7 @@ def generate_pdfa(pdf_pages, output_file, threads=1):
"-sDEVICE=pdfwrite", "-sDEVICE=pdfwrite",
"-sColorConversionStrategy=/RGB", "-sColorConversionStrategy=/RGB",
"-sProcessColorModel=DeviceRGB", "-sProcessColorModel=DeviceRGB",
"-dJPEGQ=95",
"-dPDFA=2", "-dPDFA=2",
"-sPDFACompatibilityPolicy=2", "-sPDFACompatibilityPolicy=2",
"-sOutputICCProfile=srgb.icc", "-sOutputICCProfile=srgb.icc",
+1 -1
View File
@@ -1,4 +1,4 @@
#!/usr/local/bin/python3 #!/usr/bin/env python3
############################################################################## ##############################################################################
# Copyright (c) 2013-14: fritz-hh from Github # Copyright (c) 2013-14: fritz-hh from Github
# (https://github.com/fritz-hh) # (https://github.com/fritz-hh)
-1
View File
@@ -94,7 +94,6 @@ def generate_hocr(input_file, output_hocr, language: list, tessconfig: list,
badxml, badxml,
'hocr' 'hocr'
] + tessconfig) ] + tessconfig)
print(args_tesseract)
p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE, p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE,
universal_newlines=True) universal_newlines=True)
try: try:
+6 -2
View File
@@ -1,6 +1,7 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
import sys import sys
import img2pdf import img2pdf
from PIL import Image
VERSION_STRING = '''tesseract 3.04.00 VERSION_STRING = '''tesseract 3.04.00
@@ -40,9 +41,12 @@ def main():
print('List of available languages (1):\neng', file=sys.stderr) print('List of available languages (1):\neng', file=sys.stderr)
sys.exit(0) sys.exit(0)
elif sys.argv[-1] == 'hocr': elif sys.argv[-1] == 'hocr':
inputf = sys.argv[-3]
output = sys.argv[-2] output = sys.argv[-2]
with open(output + '.hocr', 'w', encoding='utf-8') as f: with Image.open(inputf) as im, \
f.write(HOCR_TEMPLATE.format('1000', '1000')) open(output + '.hocr', 'w', encoding='utf-8') as f:
w, h = im.size
f.write(HOCR_TEMPLATE.format(str(w), str(h)))
elif sys.argv[-1] == 'pdf': elif sys.argv[-1] == 'pdf':
inputf = sys.argv[-3] inputf = sys.argv[-3]
output = sys.argv[-2] output = sys.argv[-2]