Compare commits

..
16 Commits
Author SHA1 Message Date
James R. Barlow 3fed94bb79 v4.0.7 2016-03-02 06:27:01 -08:00
James R. Barlow 8c877482bd Fix leptonica initializers 2016-03-02 06:26:25 -08:00
James R. Barlow b17d589e84 Don't set -sOutputICCProfile
Ghostscript dev advised against. It appears that this is for
creating target for a device that colors in a particular format.
2016-03-02 06:25:34 -08:00
James R. Barlow 368252a243 setuptools_scm_git_archive seems suddenly broken 2016-03-01 02:09:45 -08:00
James R. Barlow ccefda1bee v4.0.6 notes 2016-03-01 01:58:32 -08:00
James R. Barlow 3d0e8c9629 Provide our own sRGB profile instead of Ghostscript's 2016-03-01 01:27:40 -08:00
James R. Barlow 313bbbb94c setup_scm_git_archive: add additional files 2016-02-29 12:46:27 -08:00
James R. Barlow 0360f078de get_postscript_icc_path: don't check the same path multiple times 2016-02-29 12:45:58 -08:00
James R. Barlow c8901666c4 Merge branch 'master' of https://github.com/jbarlow83/OCRmyPDF 2016-02-29 00:06:07 -08:00
James R. Barlow 7430006596 Improve install instructions for OS X (unpaper) 2016-02-29 00:05:31 -08:00
James R. Barlow f3e06b2dbd Add bookmarks to file for more testing 2016-02-29 00:05:07 -08:00
jbarlow83 e97df307ff Merge pull request #54 from stweil/master
Replace broken link to c't article by permalink
2016-02-28 07:18:40 -08:00
Stefan Weil 1443354aa2 Replace broken link to c't article by permalink
Update also the 2nd article link to use a permalink, too.

Signed-off-by: Stefan Weil <sw@weilnetz.de>
2016-02-28 13:57:42 +01:00
James R. Barlow 250e68c1cd v4.0.5 release notes 2016-02-27 01:01:38 -08:00
James R. Barlow 6a380ee99c Fix temporary file placed in wrong folder 2016-02-27 00:51:47 -08:00
James R. Barlow 3c90bd96a9 Remove extraneous debug print() messages 2016-02-27 00:50:58 -08:00
12 changed files with 50 additions and 51 deletions
+1
View File
@@ -0,0 +1 @@
ref-names: $Format:%D$
+2
View File
@@ -6,3 +6,5 @@
*.jar binary *.jar binary
*.pdf binary *.pdf binary
*.PDF binary *.PDF binary
.git_archival.txt export-subst
+4 -4
View File
@@ -116,8 +116,8 @@ Install or upgrade the required Homebrew packages, if any are missing::
brew install qpdf brew install qpdf
brew install ghostscript brew install ghostscript
brew install python3 brew install python3
brew install libxml2 brew install libxml2 libffi leptonica
brew install leptonica brew install unpaper # optional
brew install tesseract brew install tesseract
Update the homebrew pip and install Pillow:: Update the homebrew pip and install Pillow::
@@ -252,11 +252,11 @@ In case you detect an issue, please:
Press & Media Press & Media
------------- -------------
- `c't 1-2014, page 59 <http://www.heise.de/ct/inhalt/2014/1/58/>`__: - `c't 1-2014, page 59 <http://heise.de/-2279695>`__:
Detailed presentation of OCRmyPDF v1.0 in the leading German IT Detailed presentation of OCRmyPDF v1.0 in the leading German IT
magazine c't magazine c't
- `heise Open Source, 09/2014: Texterkennung mit - `heise Open Source, 09/2014: Texterkennung mit
OCRmyPDF <http://www.heise.de/-2356670>`__ OCRmyPDF <http://heise.de/-2356670>`__
Disclaimer Disclaimer
---------- ----------
+21 -1
View File
@@ -6,15 +6,35 @@ Please always read this file before installing the package
Download software here: https://github.com/jbarlow83/OCRmyPDF/tags Download software here: https://github.com/jbarlow83/OCRmyPDF/tags
v4.0.4: v4.0.7:
=======
- Minor correction to Ghostscript output settings
v4.0.6:
=======
- Update install instructions
- Provide a sRGB profile instead of using Ghostscript's
v4.0.5:
======= =======
Fixes Fixes
----- -----
- Remove some verbose debug messages from v4.0.4
- Fixed temporary that wasn't being deleted
- DPI is now calculated correctly for cropped images, along with other image transformations - DPI is now calculated correctly for cropped images, along with other image transformations
- Inline images are now checked during DPI calculation instead of rejecting the image - Inline images are now checked during DPI calculation instead of rejecting the image
v4.0.4:
=======
Released with verbose debug message turned on. Do not use. Skip to v4.0.5.
v4.0.3: v4.0.3:
======= =======
Binary file not shown.
+1 -1
View File
@@ -5,6 +5,7 @@ from tempfile import NamedTemporaryFile
from subprocess import Popen, PIPE, check_call from subprocess import Popen, PIPE, check_call
from shutil import copy from shutil import copy
from . import get_program from . import get_program
from .pdfa import SRGB_ICC_PROFILE
def rasterize_pdf(input_file, output_file, xres, yres, raster_device, log, def rasterize_pdf(input_file, output_file, xres, yres, raster_device, log,
@@ -52,7 +53,6 @@ def generate_pdfa(pdf_pages, output_file, threads=1):
"-dJPEGQ=95", "-dJPEGQ=95",
"-dPDFA=2", "-dPDFA=2",
"-sPDFACompatibilityPolicy=2", "-sPDFACompatibilityPolicy=2",
"-sOutputICCProfile=srgb.icc",
"-sOutputFile=" + gs_pdf.name, "-sOutputFile=" + gs_pdf.name,
] ]
args_gs.extend(pdf_pages) args_gs.extend(pdf_pages)
+11 -6
View File
@@ -110,8 +110,8 @@ class Pix:
return "<leptonica.Pix image NULL>" return "<leptonica.Pix image NULL>"
def __getstate__(self): def __getstate__(self):
data = ffi.new('l_uint32 *[]', 1) data = ffi.new('l_uint32 **')
size = ffi.new('size_t *', 0) size = ffi.new('size_t *')
err = lept.pixSerializeToMemory(self.cpix, data, size) err = lept.pixSerializeToMemory(self.cpix, data, size)
if err != 0: if err != 0:
@@ -195,16 +195,21 @@ class Pix:
else: else:
return (None, None) return (None, None)
@staticmethod
@lru_cache(maxsize=1)
def make_pixel_sum_tab8():
return lept.makePixelSumTab8()
@staticmethod @staticmethod
def correlation_binary(pix1, pix2): def correlation_binary(pix1, pix2):
if get_leptonica_version() < 'leptonica-1.72': if get_leptonica_version() < 'leptonica-1.72':
# Older versions of Leptonica (pre-1.72) have a buggy # Older versions of Leptonica (pre-1.72) have a buggy
# implementation of pixCorrelationBinary that overflows on larger # implementation of pixCorrelationBinary that overflows on larger
# images. # images.
pix1_count = ffi.new('l_int32 *', 0) pix1_count = ffi.new('l_int32 *')
pix2_count = ffi.new('l_int32 *', 0) pix2_count = ffi.new('l_int32 *')
pixn_count = ffi.new('l_int32 *', 0) pixn_count = ffi.new('l_int32 *')
tab8 = lept.makePixelSumTab8() # Small memory leak on each call tab8 = Pix.make_pixel_sum_tab8()
lept.pixCountPixels(pix1.cpix, pix1_count, tab8) lept.pixCountPixels(pix1.cpix, pix1_count, tab8)
lept.pixCountPixels(pix2.cpix, pix2_count, tab8) lept.pixCountPixels(pix2.cpix, pix2_count, tab8)
+1 -1
View File
@@ -378,7 +378,7 @@ def cleanup_working_files(*args):
@transform( @transform(
input=options.input_file, input=options.input_file,
filter=formatter('(?i)\.pdf'), filter=formatter('(?i)\.pdf'),
output=work_folder + '{basename[0]}.repaired.pdf', output=os.path.join(work_folder, '{basename[0]}.repaired.pdf'),
extras=[_log, _pdfinfo, _pdfinfo_lock]) extras=[_log, _pdfinfo, _pdfinfo_lock])
def repair_pdf( def repair_pdf(
input_file, input_file,
-3
View File
@@ -82,7 +82,6 @@ def _interpret_contents(contentstream):
image_raster_settings = [] image_raster_settings = []
inline_images = [] inline_images = []
print(operations)
for op in operations: for op in operations:
operands, command = op operands, command = op
if command == b'q': if command == b'q':
@@ -182,7 +181,6 @@ def _get_dpi(ctm_shorthand, image_size):
def _find_page_images(page, pageinfo, contentsinfo): def _find_page_images(page, pageinfo, contentsinfo):
for n, im in enumerate(contentsinfo.inline_images): for n, im in enumerate(contentsinfo.inline_images):
print(n)
settings, shorthand = im settings, shorthand = im
image = {} image = {}
image['name'] = str('inline-%02d' % n) image['name'] = str('inline-%02d' % n)
@@ -295,7 +293,6 @@ def _pdf_get_pageinfo(infile, pageno: int):
return pageinfo return pageinfo
contentsinfo = _interpret_contents(contentstream) contentsinfo = _interpret_contents(contentstream)
print(contentsinfo)
pageinfo['images'] = [im for im in _find_page_images( pageinfo['images'] = [im for im in _find_page_images(
page, pageinfo, contentsinfo)] page, pageinfo, contentsinfo)]
+7 -33
View File
@@ -5,10 +5,13 @@
from __future__ import print_function, absolute_import, division from __future__ import print_function, absolute_import, division
from string import Template from string import Template
from subprocess import Popen, PIPE
import os
import codecs import codecs
from . import get_program import pkg_resources
ICC_PROFILE_RELPATH = 'data/sRGB_IEC61966-2-1_black_scaled.icc'
SRGB_ICC_PROFILE = pkg_resources.resource_filename(
'ocrmypdf', ICC_PROFILE_RELPATH)
# This is a template written in PostScript which is needed to create PDF/A # This is a template written in PostScript which is needed to create PDF/A
@@ -93,38 +96,9 @@ def _get_pdfa_def(icc_profile, icc_identifier, pdfmark):
return result return result
def _get_postscript_icc_path():
"Parse Ghostscript's help message to find where iccprofiles are stored"
p_gs = Popen([get_program('gs'), '--help'], close_fds=True,
universal_newlines=True,
stdout=PIPE, stderr=PIPE)
out, _ = p_gs.communicate()
lines = out.splitlines()
def search_paths(lines):
seeking = True
for line in lines:
if seeking:
if line.startswith('Search path'):
seeking = False
continue
else:
if line.strip().startswith('/'):
yield from (
path.strip() for path in line.split(':')
if path.strip() != '')
for root in search_paths(lines):
path = os.path.realpath(os.path.join(root, '../iccprofiles'))
if os.path.exists(path):
return path
raise FileNotFoundError("Could not find Ghostscript's iccprofiles")
def generate_pdfa_def(target_filename, pdfmark, icc='sRGB'): def generate_pdfa_def(target_filename, pdfmark, icc='sRGB'):
if icc == 'sRGB': if icc == 'sRGB':
icc_profile = os.path.join(_get_postscript_icc_path(), 'srgb.icc') icc_profile = SRGB_ICC_PROFILE
else: else:
raise NotImplementedError("Only supporting sRGB") raise NotImplementedError("Only supporting sRGB")
+1 -1
View File
@@ -209,7 +209,6 @@ setup(
], ],
setup_requires=[ setup_requires=[
'setuptools_scm', 'setuptools_scm',
'setuptools_scm_git_archive',
'cffi>=1.5.0', 'cffi>=1.5.0',
'pytest-runner' 'pytest-runner'
], ],
@@ -231,5 +230,6 @@ setup(
'ocrmypdf = ocrmypdf.main:run_pipeline' 'ocrmypdf = ocrmypdf.main:run_pipeline'
], ],
}, },
package_data={'ocrmypdf': ['data/sRGB_IEC61966-2-1_black_scaled.icc']},
include_package_data=True, include_package_data=True,
zip_safe=False) zip_safe=False)
Binary file not shown.