Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
aff597cef4 | ||
|
|
61b05b3dee | ||
|
|
453c4ef602 | ||
|
|
cf4b04f92d | ||
|
|
06c6999987 | ||
|
|
013c5a369f | ||
|
|
07891d994a | ||
|
|
6baf8668a6 | ||
|
|
4ba2962c56 | ||
|
|
7ad92f5db4 | ||
|
|
4dad09cc91 | ||
|
|
7b2e0c7a7a | ||
|
|
7f08f15fc9 | ||
|
|
825c0f8b2a | ||
|
|
dbe880bc41 | ||
|
|
220f1ce161 |
@@ -3,6 +3,7 @@
|
|||||||
*.sublime-*
|
*.sublime-*
|
||||||
venv-*/
|
venv-*/
|
||||||
pyvenv.cfg
|
pyvenv.cfg
|
||||||
|
tasks.py
|
||||||
|
|
||||||
# Package building
|
# Package building
|
||||||
*.egg-info/
|
*.egg-info/
|
||||||
|
|||||||
@@ -32,6 +32,7 @@ recursive-exclude .github *
|
|||||||
recursive-include ocrmypdf/data *
|
recursive-include ocrmypdf/data *
|
||||||
recursive-include share *
|
recursive-include share *
|
||||||
include *.py
|
include *.py
|
||||||
|
exclude tasks.py
|
||||||
|
|
||||||
# code
|
# code
|
||||||
recursive-include ocrmypdf *.py
|
recursive-include ocrmypdf *.py
|
||||||
|
|||||||
+2
-2
@@ -19,7 +19,7 @@ Main features
|
|||||||
- Processes pages in parallel when more than one CPU core is
|
- Processes pages in parallel when more than one CPU core is
|
||||||
available
|
available
|
||||||
- Uses `Tesseract OCR <https://github.com/tesseract-ocr/tesseract>`_ engine
|
- Uses `Tesseract OCR <https://github.com/tesseract-ocr/tesseract>`_ engine
|
||||||
- Supports the `39 languages <https://code.google.com/p/tesseract-ocr/downloads/list>`_ recognized by Tesseract
|
- Supports more than `100 languages <https://github.com/tesseract-ocr/tessdata>`_ recognized by Tesseract
|
||||||
- Battle-tested on thousands of PDFs, a test suite and continuous integration
|
- Battle-tested on thousands of PDFs, a test suite and continuous integration
|
||||||
|
|
||||||
For details: please consult the `release notes <RELEASE_NOTES.rst>`_.
|
For details: please consult the `release notes <RELEASE_NOTES.rst>`_.
|
||||||
@@ -162,7 +162,7 @@ Install the required Tesseract OCR engine with the language packs you plan to us
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
brew install tesseract # Option 1: for English, French, German, Spanish
|
brew install tesseract # Option 1: for English
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
|
|||||||
@@ -4,6 +4,13 @@ RELEASE NOTES
|
|||||||
OCRmyPDF uses `semantic versioning <http://semver.org/>`_.
|
OCRmyPDF uses `semantic versioning <http://semver.org/>`_.
|
||||||
|
|
||||||
|
|
||||||
|
v4.2.5:
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed an issue (#100) with PDFs that omit the optional /BitsPerComponent parameter on images
|
||||||
|
- Removed non-free file milk.pdf
|
||||||
|
|
||||||
|
|
||||||
v4.2.4:
|
v4.2.4:
|
||||||
=======
|
=======
|
||||||
|
|
||||||
|
|||||||
@@ -211,7 +211,10 @@ def _find_page_inline_images(page, pageinfo, contentsinfo):
|
|||||||
image['name'] = str('inline-%02d' % n)
|
image['name'] = str('inline-%02d' % n)
|
||||||
image['width'] = inline.settings['/W']
|
image['width'] = inline.settings['/W']
|
||||||
image['height'] = inline.settings['/H']
|
image['height'] = inline.settings['/H']
|
||||||
image['bpc'] = inline.settings['/BPC']
|
if '/BPC' in inline.settings:
|
||||||
|
image['bpc'] = inline.settings['/BPC']
|
||||||
|
else:
|
||||||
|
image['bpc'] = 8
|
||||||
image['color'] = FRIENDLY_COLORSPACE.get(inline.settings['/CS'], '-')
|
image['color'] = FRIENDLY_COLORSPACE.get(inline.settings['/CS'], '-')
|
||||||
image['comp'] = FRIENDLY_COMP.get(image['color'], '?')
|
image['comp'] = FRIENDLY_COMP.get(image['color'], '?')
|
||||||
if '/F' in inline.settings:
|
if '/F' in inline.settings:
|
||||||
@@ -244,7 +247,10 @@ def _find_page_regular_images(page, pageinfo, contentsinfo):
|
|||||||
image['name'] = str(xobj)
|
image['name'] = str(xobj)
|
||||||
image['width'] = pdfimage['/Width']
|
image['width'] = pdfimage['/Width']
|
||||||
image['height'] = pdfimage['/Height']
|
image['height'] = pdfimage['/Height']
|
||||||
image['bpc'] = pdfimage['/BitsPerComponent']
|
if '/BitsPerComponent' in pdfimage:
|
||||||
|
image['bpc'] = pdfimage['/BitsPerComponent']
|
||||||
|
else:
|
||||||
|
image['bpc'] = 8
|
||||||
|
|
||||||
# Fixme: this is incorrectly treats explicit masks as stencil masks,
|
# Fixme: this is incorrectly treats explicit masks as stencil masks,
|
||||||
# but good enough for now. Explicit masks have /ImageMask true but are
|
# but good enough for now. Explicit masks have /ImageMask true but are
|
||||||
|
|||||||
+1
-1
@@ -1,2 +1,2 @@
|
|||||||
[pytest]
|
[pytest]
|
||||||
norecursedirs = lib .pc
|
norecursedirs = lib .pc .git
|
||||||
|
|||||||
@@ -1,111 +0,0 @@
|
|||||||
#!/usr/bin/env python3
|
|
||||||
# -*- coding: utf-8 -*-
|
|
||||||
# © 2016 James R. Barlow: github.com/jbarlow83
|
|
||||||
# Release sanity checking
|
|
||||||
|
|
||||||
import argparse
|
|
||||||
from subprocess import run, PIPE, DEVNULL, STDOUT, CalledProcessError
|
|
||||||
from git import Repo, Remote, PushInfo
|
|
||||||
import logging
|
|
||||||
import re
|
|
||||||
import sys
|
|
||||||
import os
|
|
||||||
|
|
||||||
|
|
||||||
logging.basicConfig(level=logging.INFO)
|
|
||||||
|
|
||||||
REMOTE_ERROR_FLAGS = \
|
|
||||||
PushInfo.REJECTED | PushInfo.NO_MATCH | PushInfo.REMOTE_REJECTED | \
|
|
||||||
PushInfo.REMOTE_FAILURE | PushInfo.DELETED | PushInfo.ERROR
|
|
||||||
|
|
||||||
|
|
||||||
def test_repo(repo):
|
|
||||||
assert not repo.is_dirty(), "Repository is dirty"
|
|
||||||
if repo.untracked_files:
|
|
||||||
logging.warning('Some files are untracked:')
|
|
||||||
logging.warning('\n' + '\n'.join(repo.untracked_files))
|
|
||||||
assert repo.active_branch.name == 'master', 'Not on branch master'
|
|
||||||
|
|
||||||
|
|
||||||
def travis(args):
|
|
||||||
repo = Repo('.')
|
|
||||||
test_repo(repo)
|
|
||||||
|
|
||||||
git_describe = repo.git.describe()
|
|
||||||
|
|
||||||
try:
|
|
||||||
env = os.environ.copy()
|
|
||||||
env['SETUPTOOLS_SCM_PRETEND_VERSION'] = git_describe
|
|
||||||
proc = run(['check-manifest'], check=True, universal_newlines=True, stdout=PIPE, stderr=STDOUT, env=env)
|
|
||||||
logging.info(proc.stdout)
|
|
||||||
except CalledProcessError as e:
|
|
||||||
logging.error('MANIFEST.in error')
|
|
||||||
logging.error(e.stdout)
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
run(['python3', 'setup.py', 'build'], check=True)
|
|
||||||
|
|
||||||
origin = Remote(repo, 'jbarlow')
|
|
||||||
result = origin.push(refspec='master:master')[0]
|
|
||||||
|
|
||||||
if result.flags & REMOTE_ERROR_FLAGS:
|
|
||||||
logging.error(result.summary)
|
|
||||||
sys.exit(1)
|
|
||||||
else:
|
|
||||||
logging.info(result.summary)
|
|
||||||
|
|
||||||
logging.info("Pushed to Travis CI")
|
|
||||||
logging.info("If this passes, git tag and release")
|
|
||||||
|
|
||||||
|
|
||||||
def release(args):
|
|
||||||
repo = Repo('.')
|
|
||||||
test_repo(repo)
|
|
||||||
|
|
||||||
git_describe = repo.git.describe()
|
|
||||||
|
|
||||||
assert git_describe.startswith('v') and not '-' in git_describe and not '+ng' in git_describe, \
|
|
||||||
"Not tagged properly for release: " + git_describe
|
|
||||||
|
|
||||||
plain_version = git_describe[1:] # without 'v' prefix
|
|
||||||
|
|
||||||
with open('RELEASE_NOTES.rst') as f:
|
|
||||||
notes = f.read()
|
|
||||||
assert plain_version in notes, "Version not mentioned in release notes"
|
|
||||||
|
|
||||||
proc = run(['python3', 'setup.py', 'sdist', 'bdist_wheel'], universal_newlines=True, check=True, stdout=PIPE, stderr=STDOUT)
|
|
||||||
logging.info(proc.stdout)
|
|
||||||
|
|
||||||
|
|
||||||
origin = Remote(repo, 'jbarlow')
|
|
||||||
result = origin.push(refspec='master:master', tags=True)[0]
|
|
||||||
if result.flags & REMOTE_ERROR_FLAGS:
|
|
||||||
logging.error(result.summary)
|
|
||||||
sys.exit(1)
|
|
||||||
else:
|
|
||||||
logging.info(result.summary)
|
|
||||||
|
|
||||||
run(['twine', 'upload', '-r', 'pypitest',
|
|
||||||
'dist/ocrmypdf-{}.tar.gz'.format(plain_version),
|
|
||||||
'dist/ocrmypdf-{}-py34-none-any.whl'.format(plain_version)], check=True, universal_newlines=True, stdout=PIPE)
|
|
||||||
|
|
||||||
|
|
||||||
parser = argparse.ArgumentParser(description="ocrmypdf release tasks")
|
|
||||||
subparsers = parser.add_subparsers()
|
|
||||||
|
|
||||||
push_travis = subparsers.add_parser(
|
|
||||||
'push-travis', description="Push master to travis for testing")
|
|
||||||
push_travis.set_defaults(func=travis)
|
|
||||||
|
|
||||||
release_parser = subparsers.add_parser(
|
|
||||||
'release', description="Release to PyPI etc")
|
|
||||||
release_parser.set_defaults(func=release)
|
|
||||||
|
|
||||||
|
|
||||||
def main():
|
|
||||||
args = parser.parse_args()
|
|
||||||
args.func(args)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
|
||||||
main()
|
|
||||||
+59
-39
@@ -9,20 +9,28 @@ Files derived from free sources
|
|||||||
These test resources come from free sources, under either public domain or Creative Commons licenses.
|
These test resources come from free sources, under either public domain or Creative Commons licenses.
|
||||||
In some cases they were converted from one image format to another without other changes.
|
In some cases they were converted from one image format to another without other changes.
|
||||||
|
|
||||||
+---------------------+--------------------------------------------------------------------------------+
|
.. list-table::
|
||||||
| File | Source |
|
:widths: 20 50 30
|
||||||
+=====================+================================================================================+
|
:header-rows: 1
|
||||||
| c02-22.pdf | `Project Gutenberg`_, Adventures of Huckleberry Finn, page 22 |
|
|
||||||
+---------------------+--------------------------------------------------------------------------------+
|
* - File
|
||||||
| congress.jpg | `US Congressional Records`_ (Public Domain) |
|
- Source
|
||||||
+---------------------+--------------------------------------------------------------------------------+
|
- License
|
||||||
| graph.pdf | `Wikimedia: Pandas text analysis.png`_ (Public Domain) |
|
* - c02-22.pdf
|
||||||
+---------------------+--------------------------------------------------------------------------------+
|
- `Project Gutenberg`_, Adventures of Huckleberry Finn, page 22
|
||||||
| lichtenstein.pdf | `Wikimedia: JPEG2000 Lichtenstein`_ (Creative Commons BY-SA 3.0) |
|
- Public Domain
|
||||||
+---------------------+--------------------------------------------------------------------------------+
|
* - congress.jpg
|
||||||
| LinnSequencer.jpg, | `Wikimedia: LinnSequencer`_ (Creative Commons BY-SA 3.0) |
|
- `US Congressional Records`_
|
||||||
| linn.pdf, linn.txt | |
|
- Public Domain
|
||||||
+---------------------+--------------------------------------------------------------------------------+
|
* - graph.pdf
|
||||||
|
- `Wikimedia: Pandas text analysis.png`_
|
||||||
|
- Public Domain
|
||||||
|
* - lichtenstein.pdf
|
||||||
|
- `Wikimedia: JPEG2000 Lichtenstein`_
|
||||||
|
- Creative Commons BY-SA 3.0
|
||||||
|
* - LinnSequencer.jpg, linn.pdf, linn.txt
|
||||||
|
- `Wikimedia: LinnSequencer`_
|
||||||
|
- Creative Commons BY-SA 3.0
|
||||||
|
|
||||||
|
|
||||||
Files generated for this project
|
Files generated for this project
|
||||||
@@ -31,31 +39,43 @@ Files generated for this project
|
|||||||
The following test resources were crafted specifically for this project, and can be used
|
The following test resources were crafted specifically for this project, and can be used
|
||||||
under the terms of the license in LICENSE.rst.
|
under the terms of the license in LICENSE.rst.
|
||||||
|
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
.. list-table::
|
||||||
| File | Contributor | Purpose |
|
:widths: 20 20 60
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
:header-rows: 1
|
||||||
| aspect.pdf | @jbarlow83 | test image with 200 x 100 DPI resolution |
|
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
* - File
|
||||||
| blank.pdf | @jbarlow83 | blank PDF |
|
- Contributor
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
- Purpose
|
||||||
| cmyk.pdf | @jbarlow83 | a CMYK image created in Photoshop |
|
* - aspect.pdf
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
- @jbarlow83
|
||||||
| enormous.pdf | @jbarlow83 | very large PDF page |
|
- test image with 200 x 100 DPI resolution
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
* - blank.pdf
|
||||||
| francais.pdf | @jbarlow83 | a page containing French accents (diacritics) |
|
- @jbarlow83
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
- blank PDF
|
||||||
| hugemono.pdf | @jbarlow83 | large monochrome 35000x35000 image in JBIG2 encoding |
|
* - cmyk.pdf
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
- @jbarlow83
|
||||||
| invalid.pdf | @jbarlow83 | a PDF file header followed by EOF marker |
|
- a CMYK image created in Photoshop
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
* - enormous.pdf
|
||||||
| masks.pdf | @supergrobi | file containing explicit masks and a stencil mask |
|
- @jbarlow83
|
||||||
| | | drawn without a proper transformation matrix; printout |
|
- very large PDF page
|
||||||
| | | of a German Wikipedia article (Creative Commons BY-SA) |
|
* - epson.pdf
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
- @lowesjam
|
||||||
| milk.pdf | @lowesjam | linearized PDF containing some indirect objects |
|
- a linearized PDF containing some unusual indirect objects, created by an Epson printer; printout of a Wikipedia article (CC BY-SA)
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
* - francais.pdf
|
||||||
| missing_docinfo.pdf | @jbarlow83 | PDF file with no /DocumentInfo section |
|
- @jbarlow83
|
||||||
+---------------------+-----------------------+---------------------------------------------------------+
|
- a page containing French accents (diacritics)
|
||||||
|
* - hugemono.pdf
|
||||||
|
- @jbarlow83
|
||||||
|
- large monochrome 35000x35000 image in JBIG2 encoding
|
||||||
|
* - invalid.pdf
|
||||||
|
- @jbarlow83
|
||||||
|
- a PDF file header followed by EOF marker
|
||||||
|
* - masks.pdf
|
||||||
|
- @supergrobi
|
||||||
|
- file containing explicit masks and a stencil mask drawn without a proper transformation matrix; printout of a German Wikipedia article (CC BY-SA)
|
||||||
|
* - missing_docinfo.pdf
|
||||||
|
- @jbarlow83
|
||||||
|
- PDF file with no /DocumentInfo section
|
||||||
|
|
||||||
Assemblies
|
Assemblies
|
||||||
==========
|
==========
|
||||||
|
|||||||
Binary file not shown.
Binary file not shown.
+5
-3
@@ -124,8 +124,8 @@ def spoof_tesseract_big_image_error():
|
|||||||
return spoof('tesseract', 'tesseract_big_image_error.py')
|
return spoof('tesseract', 'tesseract_big_image_error.py')
|
||||||
|
|
||||||
|
|
||||||
def test_quick(spoof_tesseract_noop):
|
def test_quick(spoof_tesseract_cache):
|
||||||
check_ocrmypdf('c02-22.pdf', 'test_quick.pdf', env=spoof_tesseract_noop)
|
check_ocrmypdf('ccitt.pdf', 'test_quick.pdf', env=spoof_tesseract_cache)
|
||||||
|
|
||||||
|
|
||||||
def test_deskew(spoof_tesseract_noop):
|
def test_deskew(spoof_tesseract_noop):
|
||||||
@@ -635,4 +635,6 @@ def test_masks(spoof_tesseract_noop):
|
|||||||
|
|
||||||
|
|
||||||
def test_linearized_pdf_and_indirect_object(spoof_tesseract_noop):
|
def test_linearized_pdf_and_indirect_object(spoof_tesseract_noop):
|
||||||
check_ocrmypdf('milk.pdf', 'test_milk.pdf', env=spoof_tesseract_noop)
|
check_ocrmypdf(
|
||||||
|
'epson.pdf', 'test_epson.pdf',
|
||||||
|
env=spoof_tesseract_noop)
|
||||||
|
|||||||
Reference in New Issue
Block a user