Compare commits

..
10 Commits
Author SHA1 Message Date
James R. Barlow c48acf165a v4.3.5: Python 3.6 compatibility 2017-01-03 00:45:33 -08:00
James R. Barlow 9e004c3ec0 Another attempt at py 3.4/3.5
Revert to exactly what the previous passing build specified.
2017-01-03 00:34:26 -08:00
James R. Barlow 7be4e9c919 fix setuptools-scm for py 3.4, 3.5 2017-01-03 00:25:57 -08:00
James R. Barlow 5ec38a4bed Update requirements files and documentation for Python 3.6 - no code changes 2017-01-03 00:11:34 -08:00
James R. Barlow cc9ceaeb74 v4.3.4: release notes 2016-12-08 16:34:09 -08:00
James R. Barlow ad2fa8d1d7 Fix MANIFEST for .png 2016-12-08 16:25:04 -08:00
James R. Barlow adc1580742 Help py.test collect output in more cases 2016-12-08 16:21:07 -08:00
James R. Barlow 4d3b44d6df ghostscript: cleanup harmless error message printed for overprint
Redirect stderr->stdout to hopefully make GS output easier to work with
overall, since the previous code didn’t seem to account for mixed used
properly.
2016-12-08 16:19:15 -08:00
James R. Barlow e57aa0eee2 pageinfo: fix “decimal.InvalidOperation: quantize result has too many digits”
And add new test case for this.
2016-12-08 16:06:53 -08:00
James R. Barlow 1ae1d116c7 Make setup.py license internally consistent 2016-12-08 16:06:31 -08:00
14 changed files with 74 additions and 42 deletions
+3 -2
View File
@@ -11,8 +11,9 @@ cache:
- tests/cache - tests/cache
python: python:
- 3.4 - "3.4"
- 3.5 - "3.5"
- "3.6-dev" # 3.6 not available yet
before_cache: before_cache:
- rm -f $HOME/.cache/pip/log/debug.log - rm -f $HOME/.cache/pip/log/debug.log
+1
View File
@@ -14,6 +14,7 @@ include .dockerignore
# tests # tests
include pytest.ini include pytest.ini
recursive-include tests *.jpg recursive-include tests *.jpg
recursive-include tests *.png
recursive-include tests *.pdf recursive-include tests *.pdf
recursive-include tests *.py recursive-include tests *.py
recursive-include tests *.rst recursive-include tests *.rst
+12
View File
@@ -3,6 +3,18 @@ RELEASE NOTES
OCRmyPDF uses `semantic versioning <http://semver.org/>`_. OCRmyPDF uses `semantic versioning <http://semver.org/>`_.
v4.3.5:
=======
- Update documentation to confirm Python 3.6.0 compatibility. No code changes were needed, so many earlier versions are likely supported.
v4.3.4:
=======
- Fixed "decimal.InvalidOperation: quantize result has too many digits" for high DPI images
v4.3.3: v4.3.3:
======= =======
+9 -4
View File
@@ -1,4 +1,9 @@
check-manifest>=0.33 check-manifest >= 0.34
setuptools-scm>=1.11.1 twine >= 1.8.1
twine>=1.8.1 coverage >= 4.3.1
coverage>=4.2 pytest-xdist >= 1.15.0
# Known good versions: 1.11.1
# Known broken versions: 1.15.0
setuptools-scm == 1.11.1
setuptools-scm-git-archive == 1.0
+4
View File
@@ -78,6 +78,8 @@ In this worked example, the current working directory contains an input file cal
docker run --rm -v "$(pwd):/home/docker" ocrmypdf --skip-text test.pdf output.pdf docker run --rm -v "$(pwd):/home/docker" ocrmypdf --skip-text test.pdf output.pdf
.. note:: The working directory should be a writable local volume or Docker may not have permission to access it.
Note that ``ocrmypdf`` has its own separate ``-v VERBOSITYLEVEL`` argument to control debug verbosity. All Docker arguments should before the ``ocrmypdf`` image name and all arguments to ``ocrmypdf`` should be listed after. Note that ``ocrmypdf`` has its own separate ``-v VERBOSITYLEVEL`` argument to control debug verbosity. All Docker arguments should before the ``ocrmypdf`` image name and all arguments to ``ocrmypdf`` should be listed after.
@@ -105,6 +107,8 @@ Install or upgrade the required Homebrew packages, if any are missing:
brew install libxml2 libffi leptonica brew install libxml2 libffi leptonica
brew install unpaper # optional brew install unpaper # optional
Python 3.4, 3.5 and 3.6 are supported.
Install the required Tesseract OCR engine with the language packs you plan to use: Install the required Tesseract OCR engine with the language packs you plan to use:
.. code-block:: bash .. code-block:: bash
+22 -27
View File
@@ -2,7 +2,7 @@
# © 2015 James R. Barlow: github.com/jbarlow83 # © 2015 James R. Barlow: github.com/jbarlow83
from tempfile import NamedTemporaryFile from tempfile import NamedTemporaryFile
from subprocess import Popen, PIPE, check_call from subprocess import Popen, PIPE, STDOUT, check_call
from shutil import copy from shutil import copy
from . import get_program from . import get_program
from .pdfa import SRGB_ICC_PROFILE from .pdfa import SRGB_ICC_PROFILE
@@ -25,16 +25,13 @@ def rasterize_pdf(input_file, output_file, xres, yres, raster_device, log,
input_file input_file
] ]
p = Popen(args_gs, close_fds=True, stdout=PIPE, stderr=PIPE, p = Popen(args_gs, close_fds=True, stdout=PIPE, stderr=STDOUT,
universal_newlines=True) universal_newlines=True)
stdout, stderr = p.communicate() stdout, _ = p.communicate()
if stdout: if 'error' in stdout:
if 'error' in stdout: log.error(stdout) # Ghostscript puts errors in stdout
log.error(stdout) # Ghostscript puts errors in stdout else:
else: log.debug(stdout)
log.debug(stdout)
if stderr:
log.error(stderr)
if p.returncode == 0: if p.returncode == 0:
copy(tmp.name, output_file) copy(tmp.name, output_file)
@@ -60,24 +57,22 @@ def generate_pdfa(pdf_pages, output_file, log, threads=1):
"-sOutputFile=" + gs_pdf.name, "-sOutputFile=" + gs_pdf.name,
] ]
args_gs.extend(pdf_pages) args_gs.extend(pdf_pages)
p = Popen(args_gs, close_fds=True, stdout=PIPE, stderr=PIPE, p = Popen(args_gs, close_fds=True, stdout=PIPE, stderr=STDOUT,
universal_newlines=True) universal_newlines=True)
stdout, stderr = p.communicate() stdout, _ = p.communicate()
if stdout:
if 'error' in stdout: if 'error' in stdout:
log.error(stdout) log.error(stdout)
elif 'overprint mode not set' in stdout: elif 'overprint mode not set' in stdout:
# Unless someone is going to print PDF/A documents on a # Unless someone is going to print PDF/A documents on a
# magical sRGB printer I can't see the removal of overprinting # magical sRGB printer I can't see the removal of overprinting
# being a problem.... # being a problem....
log.debug( log.debug(
"Ghostscript had to remove PDF 'overprinting' from the " "Ghostscript had to remove PDF 'overprinting' from the "
"input file to complete PDF/A conversion. " "input file to complete PDF/A conversion. "
) )
else: else:
log.debug(stdout) log.debug(stdout)
if stderr:
log.error(stderr)
if p.returncode == 0: if p.returncode == 0:
# Ghostscript does not change return code when it fails to create # Ghostscript does not change return code when it fails to create
+2 -3
View File
@@ -2,7 +2,7 @@
# © 2015 James R. Barlow: github.com/jbarlow83 # © 2015 James R. Barlow: github.com/jbarlow83
from subprocess import Popen, PIPE from subprocess import Popen, PIPE
from decimal import Decimal, getcontext from decimal import Decimal
from math import hypot from math import hypot
import re import re
import sys import sys
@@ -330,9 +330,9 @@ def _find_page_regular_images(page, pageinfo, contentsinfo):
image['dpi_h'] = max(dpi_h, image.get('dpi_h', 0)) image['dpi_h'] = max(dpi_h, image.get('dpi_h', 0))
DPI_PREC = Decimal('1.000') DPI_PREC = Decimal('1.000')
dpi = Decimal(image['dpi_w'] * image['dpi_h']).sqrt()
image['dpi_w'] = Decimal(image['dpi_w']).quantize(DPI_PREC) image['dpi_w'] = Decimal(image['dpi_w']).quantize(DPI_PREC)
image['dpi_h'] = Decimal(image['dpi_h']).quantize(DPI_PREC) image['dpi_h'] = Decimal(image['dpi_h']).quantize(DPI_PREC)
dpi = Decimal(image['dpi_w'] * image['dpi_h']).sqrt()
image['dpi'] = dpi.quantize(DPI_PREC) image['dpi'] = dpi.quantize(DPI_PREC)
yield image yield image
@@ -407,7 +407,6 @@ def _pdf_get_pageinfo(infile, pageno: int):
def pdf_get_all_pageinfo(infile): def pdf_get_all_pageinfo(infile):
pdf = pypdf.PdfFileReader(infile) pdf = pypdf.PdfFileReader(infile)
getcontext().prec = 6
return [_pdf_get_pageinfo(infile, n) for n in range(pdf.numPages)] return [_pdf_get_pageinfo(infile, n) for n in range(pdf.numPages)]
+3 -3
View File
@@ -2,8 +2,8 @@
# setup.py lists a separate set of requirements that are looser to simplify # setup.py lists a separate set of requirements that are looser to simplify
# installation # installation
ruffus==2.6.3 ruffus==2.6.3
Pillow==3.3.0 Pillow==4.0.0
reportlab==3.2.0 reportlab==3.3.0
PyPDF2==1.26 PyPDF2==1.26
img2pdf==0.2.1 img2pdf==0.2.1
cffi==1.5.2 cffi==1.9.1
+4 -1
View File
@@ -191,11 +191,14 @@ setup(
url='https://github.com/jbarlow83/OCRmyPDF', url='https://github.com/jbarlow83/OCRmyPDF',
author='James R. Barlow', author='James R. Barlow',
author_email='jim@purplerock.ca', author_email='jim@purplerock.ca',
license='Public Domain', license='MIT',
packages=['ocrmypdf'], packages=['ocrmypdf'],
keywords=['PDF', 'OCR', 'optical character recognition', 'PDF/A', 'scanning'], keywords=['PDF', 'OCR', 'optical character recognition', 'PDF/A', 'scanning'],
classifiers=[ classifiers=[
"Programming Language :: Python :: 3", "Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.4",
"Programming Language :: Python :: 3.5",
"Programming Language :: Python :: 3.6",
"Development Status :: 5 - Production/Stable", "Development Status :: 5 - Production/Stable",
"Environment :: Console", "Environment :: Console",
"Intended Audience :: End Users/Desktop", "Intended Audience :: End Users/Desktop",
+1 -1
View File
@@ -1 +1 @@
pytest>=2.7.2 pytest >= 2.7.2
Binary file not shown.
+6 -1
View File
@@ -31,6 +31,9 @@ In some cases they were converted from one image format to another without other
* - LinnSequencer.jpg, linn.pdf, linn.txt * - LinnSequencer.jpg, linn.pdf, linn.txt
- `Wikimedia: LinnSequencer`_ - `Wikimedia: LinnSequencer`_
- Creative Commons BY-SA 3.0 - Creative Commons BY-SA 3.0
* - typewriter.png, 2400dpi.pdf
- `Wikimedia: Triumph typewrtier text Linzensoep`_
* Creative Commons BY-SA 2.5
Files generated for this project Files generated for this project
@@ -104,4 +107,6 @@ These test resources are assemblies from other previously mentioned files, relea
.. _`Wikimedia: JPEG2000 Lichtenstein`: https://en.wikipedia.org/wiki/JPEG_2000#/media/File:Jpeg2000_2-level_wavelet_transform-lichtenstein.png .. _`Wikimedia: JPEG2000 Lichtenstein`: https://en.wikipedia.org/wiki/JPEG_2000#/media/File:Jpeg2000_2-level_wavelet_transform-lichtenstein.png
.. _`Linux (Wikipedia Article)`: https://de.wikipedia.org/wiki/Linux .. _`Linux (Wikipedia Article)`: https://de.wikipedia.org/wiki/Linux
.. _`Wikimedia: Triumph typewrtier text Linzensoep`: https://commons.wikimedia.org/wiki/File:Triumph.typewriter_text_Linzensoep.gif
Binary file not shown.
+7
View File
@@ -60,6 +60,7 @@ def check_ocrmypdf(input_basename, output_basename, *args, env=None):
output_file = _outfile(output_basename) output_file = _outfile(output_basename)
p, out, err = run_ocrmypdf(input_basename, output_basename, *args, env=env) p, out, err = run_ocrmypdf(input_basename, output_basename, *args, env=env)
print(err) # ensure py.test collects the output, use -s to view
if p.returncode != 0: if p.returncode != 0:
print('stdout\n======') print('stdout\n======')
print(out) print(out)
@@ -748,3 +749,9 @@ def test_ghostscript_pdfa_failure(spoof_no_tess_no_pdfa):
def test_ghostscript_feature_elision(spoof_no_tess_pdfa_warning): def test_ghostscript_feature_elision(spoof_no_tess_pdfa_warning):
check_ocrmypdf('ccitt.pdf', 'test_feature_elision.pdf', check_ocrmypdf('ccitt.pdf', 'test_feature_elision.pdf',
env=spoof_no_tess_pdfa_warning) env=spoof_no_tess_pdfa_warning)
def test_very_high_dpi(spoof_tesseract_cache):
"Checks for a Decimal quantize error with high DPI, etc"
check_ocrmypdf('2400dpi.pdf', 'test_2400dpi.pdf',
env=spoof_tesseract_cache)