Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
8be9a68c5e | ||
|
|
6c34d59836 | ||
|
|
386453d178 | ||
|
|
615a7561b5 | ||
|
|
c4c64c3ea0 | ||
|
|
21279f5784 | ||
|
|
a63a21a7fc | ||
|
|
1c4d5d79f7 | ||
|
|
644581ed3c | ||
|
|
77f7621bbc | ||
|
|
42713b77d7 | ||
|
|
690f88119d | ||
|
|
78f391536b | ||
|
|
7bdd1828a9 | ||
|
|
a8f513eeeb | ||
|
|
af18bc0684 | ||
|
|
b621df6947 | ||
|
|
313c9e7dc1 | ||
|
|
9d04795f7f | ||
|
|
9a08e71e7f | ||
|
|
790d3022f6 | ||
|
|
ec311af796 | ||
|
|
c725bf79da | ||
|
|
9559f76fae | ||
|
|
45736b7c2b | ||
|
|
5629e960b9 | ||
|
|
79fd8d01a5 | ||
|
|
79fe7a0a85 | ||
|
|
b4b32a35b5 | ||
|
|
4634b3db55 | ||
|
|
f5053158d4 | ||
|
|
dfa4ce1612 | ||
|
|
585595a98e | ||
|
|
f6396fbaac | ||
|
|
3859bae85e | ||
|
|
ee1a7baae7 | ||
|
|
a4da05b66b | ||
|
|
4d67812d51 | ||
|
|
3534742ef9 | ||
|
|
8bfd46c80d | ||
|
|
cc6e9cecc0 | ||
|
|
208657f840 | ||
|
|
f3de980447 | ||
|
|
eb8992e58b | ||
|
|
72ad618ae6 | ||
|
|
f07d0c39bb | ||
|
|
9c5c7d9be0 | ||
|
|
0b19b084e2 | ||
|
|
9b4516af7a | ||
|
|
1eb45de5c9 | ||
|
|
390b9924f5 | ||
|
|
c28858a099 | ||
|
|
f00b3c00cd | ||
|
|
4e4f0bfa1f | ||
|
|
0a31acf888 | ||
|
|
b91096c615 | ||
|
|
95d9e8d91a | ||
|
|
cb6c1939e9 | ||
|
|
3764ee872a | ||
|
|
e402d5cb4b | ||
|
|
53cd04799a | ||
|
|
f2545d4496 | ||
|
|
9b81e76ed4 | ||
|
|
0956fc81aa | ||
|
|
72279e7759 | ||
|
|
6f9b948064 | ||
|
|
4eca0a165b | ||
|
|
067e61e03a | ||
|
|
1b46481f7e | ||
|
|
d8d9c41abb | ||
|
|
a8cad72f72 | ||
|
|
86c04305f4 | ||
|
|
0a110fac55 | ||
|
|
f8970ad862 | ||
|
|
fcfc78b7ee | ||
|
|
8bb244df24 |
@@ -6,6 +6,7 @@ on:
|
||||
- master
|
||||
- ci
|
||||
- release/*
|
||||
- feature/*
|
||||
tags:
|
||||
- v*
|
||||
paths-ignore:
|
||||
@@ -18,8 +19,24 @@ jobs:
|
||||
runs-on: ${{ matrix.os }}
|
||||
strategy:
|
||||
matrix:
|
||||
os: [ubuntu-18.04] #, ubuntu-20.04]
|
||||
python: ["3.6"] #, "3.7", "3.8", "3.9"]
|
||||
include:
|
||||
- os: ubuntu-18.04
|
||||
python: 3.6
|
||||
- os: ubuntu-18.04
|
||||
python: 3.7
|
||||
- os: ubuntu-20.04
|
||||
python: 3.8
|
||||
- os: ubuntu-20.04
|
||||
python: 3.9
|
||||
- os: ubuntu-latest
|
||||
python: 3.9
|
||||
- os: ubuntu-20.04
|
||||
python: "pypy-3.6"
|
||||
- os: ubuntu-latest
|
||||
python: "pypy-3.7"
|
||||
- os: ubuntu-latest
|
||||
python: 3.9
|
||||
tesseract5: true
|
||||
|
||||
env:
|
||||
OS: ${{ matrix.os }}
|
||||
@@ -35,6 +52,11 @@ jobs:
|
||||
with:
|
||||
python-version: ${{ matrix.python }}
|
||||
|
||||
- name: Install Tesseract 5
|
||||
if: matrix.tesseract5
|
||||
run: |
|
||||
sudo add-apt-repository ppa:alex-p/tesseract-ocr-devel
|
||||
|
||||
- name: Install common packages
|
||||
run: |
|
||||
sudo apt-get update
|
||||
@@ -65,6 +87,14 @@ jobs:
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
libexempi8
|
||||
|
||||
- name: Install Ubuntu packages for PyPy
|
||||
if: startsWith(matrix.python, 'pypy')
|
||||
run: |
|
||||
sudo apt-get install -y --no-install-recommends \
|
||||
libxml2-dev \
|
||||
libxslt1-dev \
|
||||
pypy3-dev
|
||||
|
||||
- name: Install Python packages
|
||||
run: |
|
||||
python -m pip install .[test]
|
||||
|
||||
+23
-8
@@ -1,23 +1,38 @@
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v3.4.0
|
||||
rev: v4.0.1
|
||||
hooks:
|
||||
- id: check-case-conflict
|
||||
- id: check-merge-conflict
|
||||
- id: check-toml
|
||||
- id: check-yaml
|
||||
- id: debug-statements
|
||||
- repo: https://github.com/asottile/seed-isort-config
|
||||
rev: v2.2.0
|
||||
hooks:
|
||||
- id: seed-isort-config
|
||||
- repo: https://github.com/pre-commit/mirrors-isort
|
||||
rev: v5.7.0 # pick the isort version you'd like to use from https://github.com/pre-commit/mirrors-isort/releases
|
||||
- repo: https://github.com/pycqa/isort
|
||||
rev: 5.9.3
|
||||
hooks:
|
||||
- id: isort
|
||||
args: ["--profile", "black"]
|
||||
- repo: https://github.com/psf/black
|
||||
rev: 20.8b1
|
||||
rev: 21.9b0
|
||||
hooks:
|
||||
- id: black
|
||||
language_version: python
|
||||
exclude: ^src/ocrmypdf/lib/_leptonica.py
|
||||
- repo: https://github.com/asottile/setup-cfg-fmt
|
||||
rev: v1.19.0
|
||||
hooks:
|
||||
- id: setup-cfg-fmt
|
||||
- repo: https://github.com/asottile/pyupgrade
|
||||
rev: v2.29.0
|
||||
hooks:
|
||||
- id: pyupgrade
|
||||
args: ["--py36-plus"]
|
||||
- repo: https://github.com/pre-commit/mirrors-mypy
|
||||
rev: v0.910-1
|
||||
hooks:
|
||||
- id: mypy
|
||||
additional_dependencies:
|
||||
- types-toml
|
||||
- types-setuptools
|
||||
- types-requests
|
||||
- types-Pillow
|
||||
|
||||
@@ -91,6 +91,11 @@ brew install tesseract-lang
|
||||
|
||||
You can then pass the `-l LANG` argument to OCRmyPDF to give a hint as to what languages it should search for. Multiple languages can be requested.
|
||||
|
||||
OCRmyPDF supports Tesseract 4.0 and the beta versions of Tesseract 5.0. It will
|
||||
automatically use whichever version it finds first on the `PATH` environment
|
||||
variable. On Windows, if `PATH` does not provide a Tesseract binary, we use
|
||||
the highest version number that is installed according to the Windows Registry.
|
||||
|
||||
## Documentation and support
|
||||
|
||||
Once OCRmyPDF is installed, the built-in help which explains the command syntax and options can be accessed via:
|
||||
@@ -115,7 +120,7 @@ In addition to the required Python version (3.6+), OCRmyPDF requires external pr
|
||||
- [heise Open Source, 09/2014: Texterkennung mit OCRmyPDF](https://heise.de/-2356670)
|
||||
- [heise Durchsuchbare PDF-Dokumente mit OCRmyPDF erstellen](https://www.heise.de/ratgeber/Durchsuchbare-PDF-Dokumente-mit-OCRmyPDF-erstellen-4607592.html)
|
||||
- [Excellent Utilities: OCRmyPDF](https://www.linuxlinks.com/excellent-utilities-ocrmypdf-add-ocr-text-layer-scanned-pdfs/)
|
||||
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
- [LinuxUser Texterkennung mit OCRmyPDF und Scanbd automatisieren](https://www.linux-community.de/ausgaben/linuxuser/2021/06/texterkennung-mit-ocrmypdf-und-scanbd-automatisieren/)
|
||||
|
||||
## Business enquiries
|
||||
|
||||
|
||||
+8
-8
@@ -12,7 +12,7 @@ subprocess call anyway, as this provides isolation of its activities.
|
||||
Example
|
||||
=======
|
||||
|
||||
OCRmyPDF one high-level function to run its main engine from an
|
||||
OCRmyPDF provides one high-level function to run its main engine from an
|
||||
application. The parameters are symmetric to the command line arguments
|
||||
and largely have the same functions.
|
||||
|
||||
@@ -23,7 +23,7 @@ and largely have the same functions.
|
||||
if __name__ == '__main__': # To ensure correct behavior on Windows and macOS
|
||||
ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True)
|
||||
|
||||
With a few exceptions, all of the command line arguments are available
|
||||
With some exceptions, all of the command line arguments are available
|
||||
and may be passed as equivalent keywords.
|
||||
|
||||
A few differences are that ``verbose`` and ``quiet`` are not available.
|
||||
@@ -41,29 +41,29 @@ execution. To do this, it will:
|
||||
- manage the signal flags of its worker processes
|
||||
- execute other subprocesses (forking and executing other programs)
|
||||
|
||||
The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently
|
||||
The Python process that calls :func:`ocrmypdf.ocr()` must be sufficiently
|
||||
privileged to perform these actions.
|
||||
|
||||
There currently is no option to manage how jobs are scheduled other
|
||||
than the argument ``jobs=`` which will limit the number of worker
|
||||
processes.
|
||||
|
||||
Creating a child process to call ``ocrmypdf.ocr()`` is suggested. That
|
||||
Creating a child process to call :func:`ocrmypdf.ocr()` is suggested. That
|
||||
way your application will survive and remain interactive even if
|
||||
OCRmyPDF fails for any reason.
|
||||
|
||||
Programs that call ``ocrmypdf.ocr()`` should also install a SIGBUS signal
|
||||
Programs that call :func:`ocrmypdf.ocr()` should also install a SIGBUS signal
|
||||
handler (except on Windows), to raise an exception if access to a memory
|
||||
mapped file fails. OCRmyPDF may use memory mapping.
|
||||
|
||||
``ocrmypdf.ocr()`` will take a threading lock to prevent multiple runs of itself
|
||||
:func:`ocrmypdf.ocr()` will take a threading lock to prevent multiple runs of itself
|
||||
in the same Python interpreter process. This is not thread-safe, because of how
|
||||
OCRmyPDF's plugins and Python's library import system work. If you need to parallelize
|
||||
OCRmyPDF, use processes.
|
||||
|
||||
.. warning::
|
||||
|
||||
On Windows and macOS, the script that calls ``ocrmypdf.ocr()`` must be
|
||||
On Windows and macOS, the script that calls :func:`ocrmypdf.ocr()` must be
|
||||
protected by an "ifmain" guard (``if __name__ == '__main__'``). If you do
|
||||
not take at least one of these steps, process semantics will prevent
|
||||
OCRmyPDF from working correctly.
|
||||
@@ -96,7 +96,7 @@ Exceptions
|
||||
|
||||
OCRmyPDF may throw standard Python exceptions, ``ocrmypdf.exceptions.*``
|
||||
exceptions, some exceptions related to multiprocessing, and
|
||||
``KeyboardInterrupt``. The parent process should provide an exception
|
||||
:exc:`KeyboardInterrupt`. The parent process should provide an exception
|
||||
handler. OCRmyPDF will clean up its temporary files and worker processes
|
||||
automatically when an exception occurs.
|
||||
|
||||
|
||||
+12
-6
@@ -31,9 +31,16 @@
|
||||
# Add any Sphinx extension module names here, as strings. They can be
|
||||
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
|
||||
# ones.
|
||||
extensions = ['sphinx.ext.napoleon', 'sphinx_issues']
|
||||
extensions = [
|
||||
'sphinx.ext.autodoc',
|
||||
'sphinx.ext.intersphinx',
|
||||
'sphinx.ext.autosummary',
|
||||
'sphinx.ext.napoleon',
|
||||
'sphinx_issues',
|
||||
]
|
||||
|
||||
# Extension settings
|
||||
intersphinx_mapping = {'https://docs.python.org/': None}
|
||||
napoleon_use_rtype = False
|
||||
issues_github_path = "jbarlow83/OCRmyPDF"
|
||||
|
||||
@@ -56,7 +63,7 @@ master_doc = 'index'
|
||||
# General information about the project.
|
||||
project = 'ocrmypdf'
|
||||
copyright = (
|
||||
'2020, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.'
|
||||
'2021, James R. Barlow. Licensed under Creative Commons Attribution-ShareAlike 4.0.'
|
||||
)
|
||||
author = 'James R. Barlow'
|
||||
|
||||
@@ -88,11 +95,10 @@ if on_rtd:
|
||||
]
|
||||
sys.modules.update((mod_name, Mock()) for mod_name in MOCK_MODULES)
|
||||
|
||||
|
||||
from pkg_resources import get_distribution, DistributionNotFound
|
||||
from importlib_metadata import version as package_version
|
||||
|
||||
# The full version, including alpha/beta/rc tags.
|
||||
release = get_distribution('ocrmypdf').version
|
||||
release = package_version('ocrmypdf')
|
||||
version = '.'.join(release.split('.')[:2])
|
||||
|
||||
|
||||
@@ -275,7 +281,7 @@ htmlhelp_basename = 'ocrmypdfdoc'
|
||||
|
||||
# -- Options for LaTeX output ---------------------------------------------
|
||||
|
||||
latex_elements = {
|
||||
latex_elements = { # type: ignore
|
||||
# The paper size ('letterpaper' or 'a4paper').
|
||||
#
|
||||
# 'papersize': 'letterpaper',
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
OCRmyPDF documentation
|
||||
======================
|
||||
|
||||
.. figure:: images/logo.svg
|
||||
|
||||
OCRmyPDF adds an optical character recognition (OCR) text layer to scanned PDF
|
||||
files, allowing them to be searched.
|
||||
|
||||
|
||||
+14
-9
@@ -2,7 +2,12 @@
|
||||
Introduction
|
||||
============
|
||||
|
||||
OCRmyPDF is a Python 3 application and library that adds OCR layers to PDFs.
|
||||
OCRmyPDF is an application and library that adds text "layers" to images
|
||||
in PDFs, making scanned image PDFs searchable. It uses OCR to guess what text
|
||||
is contained in images. It is written in Python. OCRmyPDF supports plugins
|
||||
that allow customization of its processing steps, and is very tolerant of
|
||||
PDFs that contain scanned images and "born digital" content that needs no
|
||||
text recognition.
|
||||
|
||||
About OCR
|
||||
=========
|
||||
@@ -26,7 +31,7 @@ exactly. They contain `vector
|
||||
graphics <http://vector-conversions.com/vectorizing/raster_vs_vector.html>`__
|
||||
that can contain raster objects such as scanned images. Because PDFs can
|
||||
contain multiple pages (unlike many image formats) and can contain fonts
|
||||
and text, it is a good formats for exchanging scanned documents.
|
||||
and text, it is a good format for exchanging scanned documents.
|
||||
|
||||
|image|
|
||||
|
||||
@@ -35,9 +40,9 @@ have one image. Some scanners or scanning software will segment pages
|
||||
into monochromatic text and color regions for example, to improve the
|
||||
compression ratio and appearance of the page.
|
||||
|
||||
Rasterizing a PDF is the process of generating an image suitable for
|
||||
display or analyzing with an OCR engine. OCR engines like Tesseract work
|
||||
with images, not vector objects.
|
||||
Rasterizing a PDF is the process of generating corresponding raster images.
|
||||
OCR engines like Tesseract work with images, not scalable vector graphics
|
||||
or mixed raster-vector-text graphics such as PDF.
|
||||
|
||||
About PDF/A
|
||||
===========
|
||||
@@ -76,7 +81,7 @@ OCRmyPDF analyzes each page of a PDF to determine the colorspace and
|
||||
resolution (DPI) needed to capture all of the information on that page
|
||||
without losing content. It uses
|
||||
`Ghostscript <http://ghostscript.com/>`__ to rasterize the page, and
|
||||
then performs on OCR on the rasterized image to create an OCR "layer".
|
||||
then performs on OCR the rasterized image to create an OCR "layer".
|
||||
The layer is then grafted back onto the original PDF.
|
||||
|
||||
While one can use a program like Ghostscript or ImageMagick to get an
|
||||
@@ -84,9 +89,9 @@ image and put the image through Tesseract, that actually creates a new
|
||||
PDF and many details may be lost. OCRmyPDF can produce a minimally
|
||||
changed PDF as output.
|
||||
|
||||
OCRmyPDF also some image processing options like deskew which improve
|
||||
the appearance of files and quality of OCR. When these are used, the OCR
|
||||
layer is grafted onto the processed image instead.
|
||||
OCRmyPDF also provides some image processing options, like deskew, which
|
||||
improves the appearance of files and quality of OCR. When these are used,
|
||||
the OCR layer is grafted onto the processed image instead.
|
||||
|
||||
By default, OCRmyPDF produces archival PDFs – PDF/A, which are a
|
||||
stricter subset of PDF features designed for long term archives. If
|
||||
|
||||
+3
-3
@@ -9,11 +9,11 @@ encoding was patented for a long time. All known JBIG2 US patents have
|
||||
expired as of 2017, but it is possible that unknown patents exist.
|
||||
|
||||
JBIG2 encoding is recommended for OCRmyPDF and is used to losslessly
|
||||
create smaller PDFs. If JBIG2 encoding not available, lower quality
|
||||
create smaller PDFs. If JBIG2 encoding is not available, lower quality
|
||||
encodings will be used.
|
||||
|
||||
JBIG2 decoding is not patented and is performed automatically by most
|
||||
PDF viewers. It is widely supported has been part of the PDF
|
||||
PDF viewers. It is widely supported and has been part of the PDF
|
||||
specification since 2001.
|
||||
|
||||
On macOS, Homebrew packages jbig2enc and OCRmyPDF includes it by
|
||||
@@ -37,7 +37,7 @@ Lossy mode JBIG2
|
||||
|
||||
OCRmyPDF provides lossy mode JBIG2 as an advanced feature. Users should
|
||||
`review the technical concerns with JBIG2 in lossy
|
||||
mode <https://abbyy.technology/en:kb:tip:jbig2_compression_and_ocr>`__
|
||||
mode <https://en.wikipedia.org/wiki/JBIG2#Disadvantages>`__
|
||||
and decide if this feature is acceptable for their use case.
|
||||
|
||||
JBIG2 lossy mode does achieve higher compression ratios than any other
|
||||
|
||||
@@ -12,6 +12,79 @@ may be unreliable. Use the API to depend on precise behavior.
|
||||
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||
wish to use some of its features for working with PDFs.
|
||||
|
||||
.. note::
|
||||
|
||||
Python 3.6 reaches end of life on December 23, 2021. We will end support
|
||||
for Python 3.6 around that time. The change will be marked with a major
|
||||
release.
|
||||
|
||||
v12.7.2
|
||||
=======
|
||||
|
||||
- Fixed "invalid version number" error for Tesseract packaging with nonstandard
|
||||
version "5.0.0-rc1.20211030".
|
||||
- Fixed use of deprecated ``importlib.resources.read_binary``.
|
||||
- Replace some uses of string paths with ``pathlib.Path``.
|
||||
- Fixed a leaked file handle when using ``--output-type none``.
|
||||
- Removed shims to support versions of pikepdf that are no longer supported.
|
||||
|
||||
v12.7.1
|
||||
=======
|
||||
|
||||
- Declare support for pdfminer.six v20211012.
|
||||
|
||||
v12.7.0
|
||||
=======
|
||||
|
||||
- Fixed test suite failure when using pikepdf 3.2.0 that was compiled with pybind11
|
||||
2.8.0. :issue:`843`
|
||||
- Improve advice to user about using ``--max-image-mpixels`` if OCR fails for this
|
||||
reason.
|
||||
- Minor documentation fixes. (Thanks to @mara004.)
|
||||
- Don't require importlib-metadata and importlib-resources backports on versions of
|
||||
Python where the standard library implementation is sufficient.
|
||||
(Thanks to Marco Genasci.)
|
||||
|
||||
v12.6.0
|
||||
=======
|
||||
|
||||
- Implemented ``--output-type=none`` to skip producing PDFs for applications that
|
||||
only want sidecar files (:issue:`787`).
|
||||
- Fixed ambiguities in descriptions of behavior of ``--jbig2-lossy``.
|
||||
- Various improvements to documentation.
|
||||
|
||||
v12.5.0
|
||||
=======
|
||||
|
||||
- Fixed build failure for the combination of PyPy 3.6 and pikepdf 3.0. This
|
||||
combination can work in a source build but does not work with wheels.
|
||||
- Accepted bot that wanted to upgrade our deprecated requirements.txt.
|
||||
- Documentation updates.
|
||||
- Replace pkg_resources and install dependency on setuptools with
|
||||
importlib-metadata and importlib-resources.
|
||||
- Fixed regression in hocrtransform causing text to be omitted when this
|
||||
renderer was used.
|
||||
- Fixed some typing errors.
|
||||
|
||||
v12.4.0
|
||||
=======
|
||||
|
||||
- When grafting text layers, use pikepdf's ``unparse_content_stream`` if available.
|
||||
- Confirmed support for pluggy 1.0. (Thanks @QuLogic.)
|
||||
- Fixed some typing issues, improved pre-commit settings, and fixed issues
|
||||
flagged by linters.
|
||||
- PyPy 7.3.3 (=Python 3.6) is now supported. Note that PyPy does not necessarily
|
||||
run faster, because the vast majority of OCRmyPDF's execution time is spent
|
||||
running OCR or generally executing native code. However, PyPy may bring speed
|
||||
improvements in some areas.
|
||||
|
||||
v12.3.3
|
||||
=======
|
||||
|
||||
- watcher.py: fixed interpretation of boolean env vars (:issue:`821`).
|
||||
- Adjust CI scripts to test Tesseract 5 betas.
|
||||
- Document our support for the Tesseract 5 betas.
|
||||
|
||||
v12.3.2
|
||||
=======
|
||||
|
||||
|
||||
+15
-21
@@ -24,45 +24,39 @@
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
# pylint: disable=logging-not-lazy
|
||||
|
||||
script_dir = os.path.dirname(os.path.realpath(__file__))
|
||||
print(script_dir + '/batch.py: Start')
|
||||
script_dir = Path(__file__).parent
|
||||
|
||||
if len(sys.argv) > 1:
|
||||
start_dir = sys.argv[1]
|
||||
start_dir = Path(sys.argv[1])
|
||||
else:
|
||||
start_dir = '.'
|
||||
start_dir = Path('.')
|
||||
|
||||
if len(sys.argv) > 2:
|
||||
log_file = sys.argv[2]
|
||||
log_file = Path(sys.argv[2])
|
||||
else:
|
||||
log_file = script_dir + '/ocr-tree.log'
|
||||
log_file = script_dir.with_name('ocr-tree.log')
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format='%(asctime)s %(message)s',
|
||||
filename=log_file,
|
||||
filemode='w',
|
||||
filemode='a',
|
||||
)
|
||||
|
||||
ocrmypdf.configure_logging(ocrmypdf.Verbosity.default)
|
||||
|
||||
for dir_name, subdirs, file_list in os.walk(start_dir):
|
||||
logging.info(dir_name + '\n')
|
||||
os.chdir(dir_name)
|
||||
for filename in file_list:
|
||||
file_ext = os.path.splitext(filename)[1]
|
||||
if file_ext == '.pdf':
|
||||
full_path = dir_name + '/' + filename
|
||||
print(full_path)
|
||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||
if result == ocrmypdf.ExitCode.already_done_ocr:
|
||||
print("Skipped document because it already contained text")
|
||||
elif result == ocrmypdf.ExitCode.ok:
|
||||
print("OCR complete")
|
||||
logging.info(result)
|
||||
for filename in start_dir.glob("**/*.py"):
|
||||
logging.info(f"Processing {filename}")
|
||||
result = ocrmypdf.ocr(filename, filename, deskew=True)
|
||||
if result == ocrmypdf.ExitCode.already_done_ocr:
|
||||
logging.error("Skipped document because it already contained text")
|
||||
elif result == ocrmypdf.ExitCode.ok:
|
||||
logging.info("OCR complete")
|
||||
logging.info(result)
|
||||
|
||||
@@ -54,6 +54,7 @@ function __fish_ocrmypdf_output_type
|
||||
echo -e "pdfa-1\t"(_ "output a PDF/A-1b")
|
||||
echo -e "pdfa-2\t"(_ "output a PDF/A-2b")
|
||||
echo -e "pdfa-3\t"(_ "output a PDF/A-3b")
|
||||
echo -e "none\t"(_ "do not produce an output PDF (for example, if you only care about --sidecar)")
|
||||
end
|
||||
complete -c ocrmypdf -x -l output-type -a '(__fish_ocrmypdf_output_type)' -d "select PDF output options"
|
||||
|
||||
|
||||
+1
-1
@@ -46,7 +46,7 @@ if len(sys.argv) > 1:
|
||||
else:
|
||||
start_dir = '.'
|
||||
|
||||
for dir_name, subdirs, file_list in os.walk(start_dir):
|
||||
for dir_name, _subdirs, file_list in os.walk(start_dir):
|
||||
logging.info(dir_name)
|
||||
os.chdir(dir_name)
|
||||
for filename in file_list:
|
||||
|
||||
+9
-4
@@ -37,14 +37,19 @@ import ocrmypdf
|
||||
|
||||
# pylint: disable=logging-format-interpolation
|
||||
|
||||
|
||||
def getenv_bool(name: str, default: str = 'False'):
|
||||
return os.getenv(name, default).lower() in ('true', 'yes', 'y', '1')
|
||||
|
||||
|
||||
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
|
||||
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
|
||||
OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', ''))
|
||||
ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', ''))
|
||||
DESKEW = bool(os.getenv('OCR_DESKEW', ''))
|
||||
OUTPUT_DIRECTORY_YEAR_MONTH = getenv_bool('OCR_OUTPUT_DIRECTORY_YEAR_MONTH')
|
||||
ON_SUCCESS_DELETE = getenv_bool('OCR_ON_SUCCESS_DELETE')
|
||||
DESKEW = getenv_bool('OCR_DESKEW')
|
||||
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
||||
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
||||
USE_POLLING = bool(os.getenv('OCR_USE_POLLING', ''))
|
||||
USE_POLLING = getenv_bool('OCR_USE_POLLING')
|
||||
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
|
||||
PATTERNS = ['*.pdf', '*.PDF']
|
||||
|
||||
|
||||
+1
-1
@@ -37,7 +37,7 @@ app.secret_key = "secret"
|
||||
app.config['MAX_CONTENT_LENGTH'] = 50_000_000
|
||||
app.config.from_envvar("OCRMYPDF_WEBSERVICE_SETTINGS", silent=True)
|
||||
|
||||
ALLOWED_EXTENSIONS = set(["pdf"])
|
||||
ALLOWED_EXTENSIONS = {"pdf"}
|
||||
|
||||
|
||||
def allowed_file(filename):
|
||||
|
||||
@@ -34,3 +34,49 @@ exclude = '''
|
||||
| src/ocrmypdf/lib/_leptonica.py
|
||||
)/
|
||||
'''
|
||||
|
||||
[tool.coverage.run]
|
||||
branch = true
|
||||
parallel = true
|
||||
concurrency = ["multiprocessing"]
|
||||
|
||||
[tool.coverage.paths]
|
||||
source = ["src/ocrmypdf"]
|
||||
|
||||
[tool.coverage.report]
|
||||
# Regexes for lines to exclude from consideration
|
||||
exclude_lines = [
|
||||
# Have to re-enable the standard pragma
|
||||
"pragma: no cover",
|
||||
|
||||
# Don't complain if tests don't hit defensive assertion code:
|
||||
"raise AssertionError",
|
||||
"raise NotImplementedError",
|
||||
|
||||
# Don't complain if non-runnable code isn't run:
|
||||
"if 0:",
|
||||
"if False:",
|
||||
"if __name__ == .__main__.:",
|
||||
"if TYPE_CHECKING:"
|
||||
]
|
||||
|
||||
[tool.isort]
|
||||
profile = "black"
|
||||
known_first_party = "ocrmypdf"
|
||||
known_third_party = ["PIL", "_cffi_backend", "cffi", "flask", "img2pdf", "ocrmypdf", "pdfminer", "pikepdf", "pkg_resources", "pluggy", "pytest", "reportlab", "setuptools", "sphinx_rtd_theme", "tqdm", "watchdog", "werkzeug"]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
minversion = "6.0"
|
||||
norecursedirs = ["lib", ".pc", ".git", "venv", "output", "cache", "resources"]
|
||||
testpaths = ["tests"]
|
||||
addopts = "-n auto"
|
||||
markers = ["slow"]
|
||||
filterwarnings = ["ignore:.*XMLParser.*:DeprecationWarning"]
|
||||
|
||||
[tool.mypy]
|
||||
|
||||
[[tool.mypy.overrides]]
|
||||
module = [
|
||||
'pluggy', 'tqdm', 'coloredlogs', 'img2pdf', 'cffi', '_cffi_backend', 'pdfminer.*', 'reportlab.*'
|
||||
]
|
||||
ignore_missing_imports = true
|
||||
|
||||
@@ -5,6 +5,6 @@ img2pdf == 0.4.0
|
||||
pdfminer.six == 20201018
|
||||
pikepdf == 2.10.0
|
||||
pluggy == 0.13.1
|
||||
Pillow == 8.2.0
|
||||
Pillow == 8.3.2
|
||||
reportlab == 3.5.66
|
||||
tqdm == 4.59.0
|
||||
|
||||
@@ -2,23 +2,15 @@
|
||||
name = ocrmypdf
|
||||
description = OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched
|
||||
long_description = file: README.md
|
||||
long_description_content_type = text/markdown; charset=UTF-8
|
||||
long_description_content_type = text/markdown
|
||||
url = https://github.com/jbarlow83/OCRmyPDF
|
||||
author = James R. Barlow
|
||||
author_email = james@purplerock.ca
|
||||
license = MPL-2.0
|
||||
license_file = LICENSE
|
||||
license_files =
|
||||
LICENSE
|
||||
keywords =
|
||||
PDF
|
||||
OCR
|
||||
optical character recognition
|
||||
PDF/A
|
||||
scanning
|
||||
classifiers =
|
||||
Programming Language :: Python :: 3.6
|
||||
Programming Language :: Python :: 3.7
|
||||
Programming Language :: Python :: 3.8
|
||||
Programming Language :: Python :: 3.9
|
||||
Development Status :: 5 - Production/Stable
|
||||
Environment :: Console
|
||||
Intended Audience :: End Users/Desktop
|
||||
@@ -30,68 +22,83 @@ classifiers =
|
||||
Operating System :: POSIX
|
||||
Operating System :: POSIX :: BSD
|
||||
Operating System :: POSIX :: Linux
|
||||
Programming Language :: Python :: 3
|
||||
Programming Language :: Python :: 3 :: Only
|
||||
Programming Language :: Python :: 3.6
|
||||
Programming Language :: Python :: 3.7
|
||||
Programming Language :: Python :: 3.8
|
||||
Programming Language :: Python :: 3.9
|
||||
Programming Language :: Python :: 3.10
|
||||
Topic :: Scientific/Engineering :: Image Recognition
|
||||
Topic :: Text Processing :: Indexing
|
||||
Topic :: Text Processing :: Linguistic
|
||||
keywords =
|
||||
PDF
|
||||
OCR
|
||||
optical character recognition
|
||||
PDF/A
|
||||
scanning
|
||||
project_urls =
|
||||
Documentation = https://ocrmypdf.readthedocs.io/
|
||||
Source = https://github.com/jbarlow83/ocrmypdf
|
||||
Tracker = https://github.com/jbarlow83/ocrmypdf/issues
|
||||
|
||||
[options]
|
||||
zip_safe = False
|
||||
packages = find:
|
||||
install_requires =
|
||||
Pillow>=8.2.0
|
||||
cffi>=1.9.1 # must be a setup and install requirement
|
||||
coloredlogs>=14.0 # strictly optional
|
||||
img2pdf>=0.3.0,<0.5 # pure Python
|
||||
pdfminer.six!=20200720,>=20191110,<=20211012
|
||||
pikepdf>=2.10.0
|
||||
pluggy>=0.13.0,<2
|
||||
reportlab>=3.5.66
|
||||
tqdm>=4
|
||||
importlib-metadata>=4;python_version<'3.8' # until Python 3.8
|
||||
importlib-resources>=5;python_version<'3.9' # until Python 3.9
|
||||
pikepdf<3;implementation_name=="pypy" and python_version=='3.6'
|
||||
python_requires = >=3.6
|
||||
include_package_data = True
|
||||
package_dir =
|
||||
=src
|
||||
platforms = any
|
||||
include_package_data=True
|
||||
install_requires =
|
||||
cffi >= 1.9.1 # must be a setup and install requirement
|
||||
coloredlogs >= 14.0 # strictly optional
|
||||
img2pdf >= 0.3.0, < 0.5 # pure Python, so track HEAD closely
|
||||
pdfminer.six >= 20191110, != 20200720, <= 20201018
|
||||
pikepdf >= 2.10.0
|
||||
Pillow >= 8.2.0
|
||||
pluggy >= 0.13.0, < 1.0
|
||||
reportlab >= 3.5.66
|
||||
setuptools
|
||||
tqdm >= 4
|
||||
python_requires = >= 3.6
|
||||
setup_requires = # can be removed whenever we can drop pip 9 support
|
||||
cffi >= 1.9.1 # to build the leptonica module
|
||||
setuptools_scm # so that version will work
|
||||
setuptools_scm_git_archive # enable version from github tarballs
|
||||
setup_requires =
|
||||
cffi>=1.9.1 # to build the leptonica module
|
||||
setuptools-scm
|
||||
setuptools-scm-git-archive
|
||||
zip_safe = False
|
||||
|
||||
[options.packages.find]
|
||||
where = src
|
||||
|
||||
[options.entry_points]
|
||||
console_scripts =
|
||||
ocrmypdf = ocrmypdf.__main__:run
|
||||
|
||||
[options.extras_require]
|
||||
docs =
|
||||
sphinx
|
||||
sphinx-issues
|
||||
sphinx-rtd-theme
|
||||
extended_test =
|
||||
PyMuPDF==1.13.4
|
||||
test =
|
||||
coverage[toml]>=5
|
||||
pytest>=6.0.0
|
||||
pytest-cov>=2.11.1
|
||||
pytest-xdist>=2.2.0
|
||||
python-xmp-toolkit==2.0.1 # also requires apt-get install libexempi3
|
||||
watcher =
|
||||
watchdog>=1.0.2,<3
|
||||
webservice =
|
||||
Flask>=1,<3
|
||||
|
||||
[options.package_data]
|
||||
ocrmypdf =
|
||||
data/sRGB.icc
|
||||
py.typed
|
||||
|
||||
[options.packages.find]
|
||||
where = src
|
||||
|
||||
[options.extras_require]
|
||||
test =
|
||||
pytest >= 6.0.0
|
||||
pytest-xdist >= 2.2.0
|
||||
pytest-cov >= 2.11.1
|
||||
python-xmp-toolkit == 2.0.1 # also requires apt-get install libexempi3
|
||||
# or brew install exempi
|
||||
docs =
|
||||
sphinx
|
||||
sphinx-rtd-theme
|
||||
sphinx-issues
|
||||
extended_test =
|
||||
PyMuPDF == 1.13.4
|
||||
watcher =
|
||||
watchdog >= 1.0.2, < 3
|
||||
webservice =
|
||||
Flask >= 1, < 3
|
||||
|
||||
[options.entry_points]
|
||||
console_scripts =
|
||||
ocrmypdf = ocrmypdf.__main__:run
|
||||
|
||||
[bdist_wheel]
|
||||
python-tag = py36
|
||||
|
||||
@@ -100,48 +107,10 @@ test = pytest
|
||||
|
||||
[check-manifest]
|
||||
ignore =
|
||||
.github
|
||||
.github
|
||||
|
||||
[tool:pytest]
|
||||
norecursedirs = lib .pc .git output cache resources
|
||||
testpaths = tests
|
||||
filterwarnings =
|
||||
ignore:.*XMLParser.*:DeprecationWarning
|
||||
markers =
|
||||
slow
|
||||
addopts =
|
||||
-n auto
|
||||
|
||||
[isort]
|
||||
multi_line_output = 3
|
||||
include_trailing_comma = True
|
||||
force_grid_wrap = 0
|
||||
use_parentheses = True
|
||||
line_length = 88
|
||||
known_first_party = ocrmypdf
|
||||
known_third_party = PIL,_cffi_backend,cffi,flask,img2pdf,pdfminer,pikepdf,pkg_resources,pluggy,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug
|
||||
|
||||
[coverage:paths]
|
||||
source =
|
||||
src/ocrmypdf
|
||||
|
||||
[coverage:run]
|
||||
branch = true
|
||||
parallel = true
|
||||
concurrency = multiprocessing
|
||||
|
||||
[coverage:report]
|
||||
# Regexes for lines to exclude from consideration
|
||||
exclude_lines =
|
||||
# Have to re-enable the standard pragma
|
||||
pragma: no cover
|
||||
|
||||
# Don't complain if tests don't hit defensive assertion code:
|
||||
raise AssertionError
|
||||
raise NotImplementedError
|
||||
|
||||
# Don't complain if non-runnable code isn't run:
|
||||
if 0:
|
||||
if False:
|
||||
if __name__ == .__main__.:
|
||||
if TYPE_CHECKING:
|
||||
[flake8]
|
||||
ignore = D203,F401,W503,E501,E203,F841
|
||||
exclude = .git,__pycache__,docs/conf.py,build,dist,.venv,.venvpp,.eggs,tmp,src/ocrmypdf/lib/
|
||||
max-complexity = 10
|
||||
max-line-length = 100
|
||||
|
||||
@@ -8,9 +8,7 @@
|
||||
"""Interface to Tesseract executable"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
from collections import namedtuple
|
||||
from distutils.version import StrictVersion
|
||||
from os import fspath
|
||||
@@ -61,8 +59,8 @@ class TesseractVersion(StrictVersion):
|
||||
r'''
|
||||
^(\d+) \. (\d+) (\. (\d+))? # groups: 1/major, 2/minor, 3/[skip], 4/patch
|
||||
[-]? # optional hyphen separator
|
||||
(?:(alpha|beta|rc|dev)?[.\-\ ]?(\d+)?)? # 5/prerelease, 6/prerelease_num
|
||||
(?:-(\d+)-g[0-9a-f]+)? # untagged git version
|
||||
(?: ((?:alpha|beta|rc|dev)\d*)? [.\-\ ]? (\d+)? )? # 5/prerelease, 6/prerelease_num
|
||||
(?:(?:-\d+)?-g[0-9a-f]+)? # untagged git version
|
||||
$
|
||||
''',
|
||||
re.VERBOSE | re.ASCII,
|
||||
@@ -73,7 +71,7 @@ class TesseractVersion(StrictVersion):
|
||||
super().parse(vstring)
|
||||
except TypeError as e:
|
||||
if 'int() argument must be a string' in str(e):
|
||||
super().parse(vstring + '0')
|
||||
super().parse(vstring + '-0')
|
||||
|
||||
|
||||
def version():
|
||||
@@ -117,7 +115,7 @@ def get_languages():
|
||||
if line.startswith('Error'):
|
||||
raise MissingDependencyError(lang_error(output))
|
||||
_header, *rest = output.splitlines()
|
||||
return set(lang.strip() for lang in rest)
|
||||
return {lang.strip() for lang in rest}
|
||||
|
||||
|
||||
def tess_base_args(langs: List[str], engine_mode: Optional[int]) -> List[str]:
|
||||
@@ -250,7 +248,8 @@ def generate_hocr(
|
||||
|
||||
# Reminder: test suite tesseract test plugins will break after any changes
|
||||
# to the number of order parameters here
|
||||
args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig)
|
||||
args_tesseract.extend([fspath(input_file), fspath(prefix), 'hocr', 'txt'])
|
||||
args_tesseract.extend(tessconfig)
|
||||
try:
|
||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||
stdout = p.stdout
|
||||
@@ -272,7 +271,7 @@ def generate_hocr(
|
||||
# The sidecar text file will get the suffix .txt; rename it to
|
||||
# whatever caller wants it named
|
||||
if prefix.with_suffix('.txt').exists():
|
||||
shutil.move(prefix.with_suffix('.txt'), output_text)
|
||||
prefix.with_suffix('.txt').replace(output_text)
|
||||
|
||||
|
||||
def use_skip_page(output_pdf, output_text):
|
||||
@@ -319,17 +318,18 @@ def generate_pdf(
|
||||
if user_patterns:
|
||||
args_tesseract.extend(['--user-patterns', user_patterns])
|
||||
|
||||
prefix = os.path.splitext(output_pdf)[0] # Tesseract appends suffixes
|
||||
prefix = output_pdf.parent / Path(output_pdf.stem)
|
||||
|
||||
# Reminder: test suite tesseract test plugins might break after any changes
|
||||
# to the number of order parameters here
|
||||
|
||||
args_tesseract.extend([input_file, prefix, 'pdf', 'txt'] + tessconfig)
|
||||
args_tesseract.extend([fspath(input_file), fspath(prefix), 'pdf', 'txt'])
|
||||
args_tesseract.extend(tessconfig)
|
||||
try:
|
||||
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||
stdout = p.stdout
|
||||
if os.path.exists(prefix + '.txt'):
|
||||
shutil.move(prefix + '.txt', output_text)
|
||||
if prefix.with_suffix('.txt').exists():
|
||||
prefix.with_suffix('.txt').replace(output_text)
|
||||
except TimeoutExpired:
|
||||
page_timedout(timeout)
|
||||
use_skip_page(output_pdf, output_text)
|
||||
|
||||
@@ -45,7 +45,7 @@ def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]:
|
||||
im = im.convert(mode='1')
|
||||
else:
|
||||
im = im.convert(mode='RGB')
|
||||
except IOError as e:
|
||||
except OSError as e:
|
||||
raise MissingDependencyError(
|
||||
"Could not convert image with type " + im.mode
|
||||
) from e
|
||||
@@ -96,12 +96,12 @@ def run(
|
||||
try:
|
||||
with Image.open(output_pnm) as imout:
|
||||
imout.save(output_file, dpi=(dpi, dpi))
|
||||
except (FileNotFoundError, OSError):
|
||||
except OSError as e:
|
||||
raise SubprocessOutputError(
|
||||
"unpaper: failed to produce the expected output file. "
|
||||
+ " Called with: "
|
||||
+ str(args_unpaper)
|
||||
) from None
|
||||
) from e
|
||||
|
||||
|
||||
def validate_custom_args(args: str) -> List[str]:
|
||||
|
||||
+38
-43
@@ -11,8 +11,19 @@ from contextlib import suppress
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import pikepdf
|
||||
from pikepdf.objects import Dictionary, Name
|
||||
from pikepdf import (
|
||||
Dictionary,
|
||||
Name,
|
||||
Object,
|
||||
Operator,
|
||||
Page,
|
||||
Pdf,
|
||||
PdfError,
|
||||
PdfMatrix,
|
||||
Stream,
|
||||
parse_content_stream,
|
||||
unparse_content_stream,
|
||||
)
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
MAX_REPLACE_PAGES = 100
|
||||
@@ -47,44 +58,28 @@ def strip_invisible_text(pdf, page):
|
||||
render_mode = 0
|
||||
text_objects = []
|
||||
|
||||
rich_page = pikepdf.Page(page)
|
||||
rich_page = Page(page)
|
||||
rich_page.contents_coalesce()
|
||||
for operands, operator in pikepdf.parse_content_stream(page, ''):
|
||||
for operands, operator in parse_content_stream(page, ''):
|
||||
if not in_text_obj:
|
||||
if operator == pikepdf.Operator('BT'):
|
||||
if operator == Operator('BT'):
|
||||
in_text_obj = True
|
||||
render_mode = 0
|
||||
text_objects.append((operands, operator))
|
||||
else:
|
||||
stream.append((operands, operator))
|
||||
else:
|
||||
if operator == pikepdf.Operator('Tr'):
|
||||
if operator == Operator('Tr'):
|
||||
render_mode = operands[0]
|
||||
text_objects.append((operands, operator))
|
||||
if operator == pikepdf.Operator('ET'):
|
||||
if operator == Operator('ET'):
|
||||
in_text_obj = False
|
||||
if render_mode != 3:
|
||||
stream.extend(text_objects)
|
||||
text_objects.clear()
|
||||
|
||||
def convert(op):
|
||||
try:
|
||||
return op.unparse()
|
||||
except AttributeError:
|
||||
return str(op).encode('ascii')
|
||||
|
||||
lines = []
|
||||
|
||||
for operands, operator in stream:
|
||||
if operator == pikepdf.Operator('INLINE IMAGE'):
|
||||
iim = operands[0]
|
||||
line = iim.unparse()
|
||||
else:
|
||||
line = b' '.join(convert(op) for op in operands) + b' ' + operator.unparse()
|
||||
lines.append(line)
|
||||
|
||||
content_stream = b'\n'.join(lines)
|
||||
page.Contents = pikepdf.Stream(pdf, content_stream)
|
||||
content_stream = unparse_content_stream(stream)
|
||||
page.Contents = Stream(pdf, content_stream)
|
||||
|
||||
|
||||
class OcrGrafter:
|
||||
@@ -92,14 +87,14 @@ class OcrGrafter:
|
||||
self.context = context
|
||||
self.path_base = context.origin
|
||||
|
||||
self.pdf_base = pikepdf.open(self.path_base)
|
||||
self.pdf_base = Pdf.open(self.path_base)
|
||||
self.font, self.font_key = None, None
|
||||
|
||||
self.pdfinfo = context.pdfinfo
|
||||
self.output_file = context.get_path('graft_layers.pdf')
|
||||
|
||||
self.procset = self.pdf_base.make_indirect(
|
||||
pikepdf.Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]')
|
||||
Object.parse(b'[ /PDF /Text /ImageB /ImageC /ImageI ]')
|
||||
)
|
||||
|
||||
self.emplacements = 1
|
||||
@@ -123,7 +118,7 @@ class OcrGrafter:
|
||||
# We are updating the old page with a rasterized PDF of the new
|
||||
# page (without changing objgen, to preserve references)
|
||||
log.debug("Emplacement update")
|
||||
with pikepdf.open(image) as pdf_image:
|
||||
with Pdf.open(path_image) as pdf_image:
|
||||
self.emplacements += 1
|
||||
foreign_image_page = pdf_image.pages[0]
|
||||
self.pdf_base.pages.append(foreign_image_page)
|
||||
@@ -196,7 +191,7 @@ class OcrGrafter:
|
||||
self.pdf_base.save(next_file)
|
||||
self.pdf_base.close()
|
||||
|
||||
self.pdf_base = pikepdf.open(next_file)
|
||||
self.pdf_base = Pdf.open(next_file)
|
||||
self.procset = self.pdf_base.pages[0].Resources.ProcSet
|
||||
self.font, self.font_key = None, None # Ensure we reacquire this information
|
||||
self.interim_count += 1
|
||||
@@ -212,7 +207,7 @@ class OcrGrafter:
|
||||
font, font_key = None, None
|
||||
possible_font_names = ('/f-0-0', '/F1')
|
||||
try:
|
||||
with pikepdf.open(text) as pdf_text:
|
||||
with Pdf.open(text) as pdf_text:
|
||||
try:
|
||||
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
||||
except (AttributeError, IndexError, KeyError):
|
||||
@@ -226,7 +221,7 @@ class OcrGrafter:
|
||||
if pdf_text_font:
|
||||
font = self.pdf_base.copy_foreign(pdf_text_font)
|
||||
return font, font_key
|
||||
except (FileNotFoundError, pikepdf.PdfError):
|
||||
except (FileNotFoundError, PdfError):
|
||||
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
||||
return None, None
|
||||
|
||||
@@ -235,9 +230,9 @@ class OcrGrafter:
|
||||
*,
|
||||
page_num: int,
|
||||
textpdf: Path,
|
||||
font: pikepdf.Object,
|
||||
font_key: pikepdf.Object,
|
||||
procset: pikepdf.Object,
|
||||
font: Object,
|
||||
font_key: Object,
|
||||
procset: Object,
|
||||
text_rotation: int,
|
||||
strip_old_text: bool,
|
||||
):
|
||||
@@ -248,7 +243,7 @@ class OcrGrafter:
|
||||
return
|
||||
|
||||
# This is a pointer indicating a specific page in the base file
|
||||
with pikepdf.open(textpdf) as pdf_text:
|
||||
with Pdf.open(textpdf) as pdf_text:
|
||||
pdf_text_contents = pdf_text.pages[0].Contents.read_bytes()
|
||||
|
||||
base_page = self.pdf_base.pages.p(page_num)
|
||||
@@ -263,13 +258,13 @@ class OcrGrafter:
|
||||
mediabox = [float(base_page.MediaBox[v]) for v in range(4)]
|
||||
wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
||||
|
||||
translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2)
|
||||
untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2)
|
||||
corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1])
|
||||
translate = PdfMatrix().translated(-wt / 2, -ht / 2)
|
||||
untranslate = PdfMatrix().translated(wp / 2, hp / 2)
|
||||
corner = PdfMatrix().translated(mediabox[0], mediabox[1])
|
||||
# -rotation because the input is a clockwise angle and this formula
|
||||
# uses CCW
|
||||
text_rotation = -text_rotation % 360
|
||||
rotate = pikepdf.PdfMatrix().rotated(text_rotation)
|
||||
rotate = PdfMatrix().rotated(text_rotation)
|
||||
|
||||
# Because of rounding of DPI, we might get a text layer that is not
|
||||
# identically sized to the target page. Scale to adjust. Normally this
|
||||
@@ -280,7 +275,7 @@ class OcrGrafter:
|
||||
scale_y = hp / ht
|
||||
|
||||
# log.debug('%r', scale_x, scale_y)
|
||||
scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y)
|
||||
scale = PdfMatrix().scaled(scale_x, scale_y)
|
||||
|
||||
# Translate the text so it is centered at (0, 0), rotate it there, adjust
|
||||
# for a size different between initial and text PDF, then untranslate, and
|
||||
@@ -303,14 +298,14 @@ class OcrGrafter:
|
||||
pdf_draw_xobj = (
|
||||
(b'q %s cm\n' % ctm.encode()) + (b'%s Do\n' % text_xobj_name) + b'\nQ\n'
|
||||
)
|
||||
new_text_layer = pikepdf.Stream(self.pdf_base, pdf_draw_xobj)
|
||||
new_text_layer = Stream(self.pdf_base, pdf_draw_xobj)
|
||||
|
||||
if strip_old_text:
|
||||
strip_invisible_text(self.pdf_base, base_page)
|
||||
|
||||
if hasattr(pikepdf.Page, 'contents_add'):
|
||||
if hasattr(Page, 'contents_add'):
|
||||
# pikepdf >= 2.14 adds this method and deprecates the one below
|
||||
pikepdf.Page(base_page).contents_add(new_text_layer, prepend=True)
|
||||
Page(base_page).contents_add(new_text_layer, prepend=True)
|
||||
else:
|
||||
# pikepdf < 2.14
|
||||
base_page.page_contents_add(
|
||||
|
||||
@@ -48,7 +48,7 @@ def triage_image_file(input_file, output_file, options):
|
||||
log.info("Input file is not a PDF, checking if it is an image...")
|
||||
try:
|
||||
im = Image.open(input_file)
|
||||
except EnvironmentError as e:
|
||||
except OSError as e:
|
||||
# Recover the original filename
|
||||
log.error(str(e).replace(str(input_file), str(options.input_file)))
|
||||
raise UnsupportedImageFormatError() from e
|
||||
@@ -135,7 +135,7 @@ def triage(original_filename, input_file, output_file, options):
|
||||
# Origin file is a pdf create a symlink with pdf extension
|
||||
safe_symlink(input_file, output_file)
|
||||
return output_file
|
||||
except EnvironmentError as e:
|
||||
except OSError as e:
|
||||
log.debug(f"Temporary file was at: {input_file}")
|
||||
msg = str(e).replace(str(input_file), original_filename)
|
||||
raise InputFileError(msg) from e
|
||||
@@ -521,13 +521,12 @@ def create_ocr_image(image: Path, page_context: PageContext):
|
||||
# be None)
|
||||
bbox = [float(v) for v in textarea]
|
||||
xyscale = tuple(float(coord) / 72.0 for coord in im.info['dpi'])
|
||||
pixcoords = [
|
||||
pixcoords = (
|
||||
bbox[0] * xyscale[0],
|
||||
im.height - bbox[3] * xyscale[1],
|
||||
bbox[2] * xyscale[0],
|
||||
im.height - bbox[1] * xyscale[1],
|
||||
]
|
||||
pixcoords = [int(round(c)) for c in pixcoords]
|
||||
)
|
||||
log.debug('blanking %r', pixcoords)
|
||||
draw.rectangle(pixcoords, fill=white)
|
||||
# draw.rectangle(pixcoords, outline=pink)
|
||||
@@ -856,7 +855,7 @@ def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
||||
if frm != 1:
|
||||
stream.write('\f') # Form feed between pages
|
||||
if txt_file:
|
||||
with open(txt_file, 'r', encoding="utf-8") as in_:
|
||||
with open(txt_file, encoding="utf-8") as in_:
|
||||
txt = in_.read()
|
||||
# Some OCR engines (e.g. Tesseract v4 alpha) add form feeds
|
||||
# between pages, and some do not. For consistency, we ignore
|
||||
|
||||
+16
-8
@@ -293,12 +293,13 @@ def exec_concurrent(context: PdfContext, executor: Executor):
|
||||
# Merge layers to one single pdf
|
||||
pdf = ocrgraft.finalize()
|
||||
|
||||
# PDF/A and metadata
|
||||
log.info("Postprocessing...")
|
||||
pdf = post_process(pdf, context, executor)
|
||||
if options.output_type != 'none':
|
||||
# PDF/A and metadata
|
||||
log.info("Postprocessing...")
|
||||
pdf = post_process(pdf, context, executor)
|
||||
|
||||
# Copy PDF file to destination
|
||||
copy_final(pdf, options.output_file, context)
|
||||
# Copy PDF file to destination
|
||||
copy_final(pdf, options.output_file, context)
|
||||
|
||||
|
||||
def configure_debug_logging(log_filename: Path, prefix: str = ''):
|
||||
@@ -399,7 +400,7 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
||||
return ExitCode.invalid_output_pdf
|
||||
report_output_file_size(options, start_input_file, options.output_file)
|
||||
|
||||
except (KeyboardInterrupt if not api else NeverRaise) as e:
|
||||
except (KeyboardInterrupt if not api else NeverRaise):
|
||||
if options.verbose >= 1:
|
||||
log.exception("KeyboardInterrupt")
|
||||
else:
|
||||
@@ -413,7 +414,14 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
||||
else:
|
||||
log.error(type(e).__name__)
|
||||
return e.exit_code
|
||||
except (Exception if not api else NeverRaise) as e: # pylint: disable=broad-except
|
||||
except (PIL.Image.DecompressionBombError if not api else NeverRaise) as e:
|
||||
log.exception(
|
||||
"A decompression bomb error was encountered while executing the "
|
||||
"pipeline. Use the argument --max-image-mpixels to raise the maximum "
|
||||
"image pixel limit."
|
||||
)
|
||||
return ExitCode.other_error
|
||||
except (Exception if not api else NeverRaise): # pylint: disable=broad-except
|
||||
log.exception("An exception occurred while executing the pipeline")
|
||||
return ExitCode.other_error
|
||||
finally:
|
||||
@@ -421,7 +429,7 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
||||
try:
|
||||
debug_log_handler.close()
|
||||
log.removeHandler(debug_log_handler)
|
||||
except EnvironmentError as e:
|
||||
except OSError as e:
|
||||
print(e, file=sys.stderr)
|
||||
cleanup_working_files(work_folder, options)
|
||||
|
||||
|
||||
+23
-20
@@ -13,7 +13,7 @@ import sys
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from shutil import copyfileobj
|
||||
from typing import List, Set, Tuple, Union
|
||||
from typing import List, Set, Tuple
|
||||
|
||||
import pikepdf
|
||||
import PIL
|
||||
@@ -26,12 +26,7 @@ from ocrmypdf.exceptions import (
|
||||
MissingDependencyError,
|
||||
OutputFileAccessError,
|
||||
)
|
||||
from ocrmypdf.helpers import (
|
||||
is_file_writable,
|
||||
is_iterable_notstr,
|
||||
monotonic,
|
||||
safe_symlink,
|
||||
)
|
||||
from ocrmypdf.helpers import is_file_writable, monotonic, safe_symlink, samefile
|
||||
from ocrmypdf.hocrtransform import HOCR_OK_LANGS
|
||||
from ocrmypdf.subprocess import check_external_program
|
||||
|
||||
@@ -68,7 +63,7 @@ def check_options_languages(options, ocr_engine_languages):
|
||||
missing_languages = options.languages - ocr_engine_languages
|
||||
if missing_languages:
|
||||
msg = (
|
||||
f"OCR engine does not have language data for the following "
|
||||
"OCR engine does not have language data for the following "
|
||||
"requested languages: \n"
|
||||
)
|
||||
msg += '\n'.join(lang for lang in missing_languages)
|
||||
@@ -80,12 +75,18 @@ def check_options_output(options):
|
||||
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
||||
|
||||
if options.pdf_renderer.startswith('hocr') and not is_latin:
|
||||
msg = (
|
||||
log.warning(
|
||||
"The 'hocr' PDF renderer is known to cause problems with one "
|
||||
"or more of the languages in your document. Use "
|
||||
"--pdf-renderer auto (the default) to avoid this issue."
|
||||
"`--pdf-renderer auto` (the default) to avoid this issue."
|
||||
)
|
||||
|
||||
if options.output_type == 'none' and options.output_file != os.devnull:
|
||||
raise BadArgsError(
|
||||
"Since you specified `--pdf-renderer none`, the output file "
|
||||
f"{options.output_file} cannot be produced. Set the output file to "
|
||||
f"{os.devnull} to suppress this message."
|
||||
)
|
||||
log.warning(msg)
|
||||
|
||||
lossless_reconstruction = False
|
||||
if not any(
|
||||
@@ -112,6 +113,10 @@ def check_options_sidecar(options):
|
||||
raise BadArgsError(
|
||||
"--sidecar filename must be specified when output file is stdout."
|
||||
)
|
||||
elif options.output_file == os.devnull:
|
||||
raise BadArgsError(
|
||||
"--sidecar filename must be specified when output file is /dev/null or NUL."
|
||||
)
|
||||
options.sidecar = options.output_file + '.txt'
|
||||
if options.sidecar == options.input_file or options.sidecar == options.output_file:
|
||||
raise BadArgsError(
|
||||
@@ -142,8 +147,6 @@ def check_options_preprocessing(options):
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges: str) -> Set[int]:
|
||||
if is_iterable_notstr(ranges):
|
||||
return set(ranges)
|
||||
pages: List[int] = []
|
||||
page_groups = ranges.replace(' ', '').split(',')
|
||||
for g in page_groups:
|
||||
@@ -157,10 +160,12 @@ def _pages_from_ranges(ranges: str) -> Set[int]:
|
||||
try:
|
||||
new_pages = list(range(int(start) - 1, int(end)))
|
||||
if not new_pages:
|
||||
raise BadArgsError(f"invalid page subrange '{start}-{end}'")
|
||||
raise BadArgsError(
|
||||
f"invalid page subrange '{start}-{end}'"
|
||||
) from None
|
||||
pages.extend(new_pages)
|
||||
except ValueError:
|
||||
raise BadArgsError("invalid page range") from None
|
||||
raise BadArgsError(f"invalid page subrange '{g}'") from None
|
||||
|
||||
if not pages:
|
||||
raise BadArgsError(
|
||||
@@ -182,10 +187,8 @@ def _pages_from_ranges(ranges: str) -> Set[int]:
|
||||
|
||||
def check_options_ocr_behavior(options):
|
||||
exclusive_options = sum(
|
||||
[
|
||||
(1 if opt else 0)
|
||||
for opt in (options.force_ocr, options.skip_text, options.redo_ocr)
|
||||
]
|
||||
(1 if opt else 0)
|
||||
for opt in (options.force_ocr, options.skip_text, options.redo_ocr)
|
||||
)
|
||||
if exclusive_options >= 2:
|
||||
raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.")
|
||||
@@ -302,7 +305,7 @@ def check_closed_streams(options): # pragma: no cover
|
||||
if options.input_file == '-':
|
||||
log.error("Trying to read from stdin but stdin seems closed")
|
||||
return False
|
||||
sys.stdin = open(os.devnull, 'r')
|
||||
sys.stdin = open(os.devnull)
|
||||
|
||||
if sys.stdout is None:
|
||||
if options.output_file == '-':
|
||||
|
||||
@@ -5,9 +5,12 @@
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
|
||||
import pkg_resources
|
||||
try:
|
||||
from importlib_metadata import version as _package_version
|
||||
except ImportError:
|
||||
from importlib.metadata import version as _package_version
|
||||
|
||||
PROGRAM_NAME = 'ocrmypdf'
|
||||
|
||||
# Official PEP 396
|
||||
__version__ = pkg_resources.get_distribution('ocrmypdf').version
|
||||
__version__ = _package_version('ocrmypdf')
|
||||
|
||||
+16
-5
@@ -15,10 +15,7 @@ from pathlib import Path
|
||||
from typing import AnyStr, BinaryIO, Iterable, Optional, Union
|
||||
from warnings import warn
|
||||
|
||||
from ocrmypdf._logging import ( # pylint: disable=unused-import
|
||||
PageNumberFilter,
|
||||
TqdmConsole,
|
||||
)
|
||||
from ocrmypdf._logging import PageNumberFilter, TqdmConsole
|
||||
from ocrmypdf._plugin_manager import get_plugin_manager
|
||||
from ocrmypdf._sync import run_pipeline
|
||||
from ocrmypdf._validation import check_options
|
||||
@@ -31,7 +28,7 @@ except ModuleNotFoundError:
|
||||
coloredlogs = None
|
||||
|
||||
|
||||
StrPath = Union[os.PathLike, AnyStr]
|
||||
StrPath = Union[Path, AnyStr]
|
||||
PathOrIO = Union[BinaryIO, StrPath]
|
||||
|
||||
_api_lock = threading.Lock()
|
||||
@@ -338,3 +335,17 @@ def ocr( # pylint: disable=unused-argument
|
||||
options = create_options(**create_options_kwargs)
|
||||
check_options(options, plugin_manager)
|
||||
return run_pipeline(options=options, plugin_manager=plugin_manager, api=True)
|
||||
|
||||
|
||||
__all__ = [
|
||||
'PageNumberFilter',
|
||||
'TqdmConsole',
|
||||
'Verbosity',
|
||||
'check_options',
|
||||
'configure_logging',
|
||||
'create_options',
|
||||
'get_parser',
|
||||
'get_plugin_manager',
|
||||
'ocr',
|
||||
'run_pipeline',
|
||||
]
|
||||
|
||||
@@ -20,9 +20,8 @@ import signal
|
||||
import sys
|
||||
import threading
|
||||
from contextlib import suppress
|
||||
from multiprocessing import Pool as ProcessPool
|
||||
from multiprocessing.pool import ThreadPool
|
||||
from typing import Callable, Iterable, Union
|
||||
from multiprocessing.pool import Pool, ThreadPool
|
||||
from typing import Callable, Iterable, Type, Union
|
||||
|
||||
from tqdm import tqdm
|
||||
|
||||
@@ -31,7 +30,10 @@ from ocrmypdf._logging import TqdmConsole
|
||||
from ocrmypdf.exceptions import InputFileError
|
||||
from ocrmypdf.helpers import remove_all_log_handlers
|
||||
|
||||
ProcessPool = Pool
|
||||
Queue = Union[multiprocessing.Queue, queue.Queue]
|
||||
UserInit = Callable[[], None]
|
||||
WorkerInit = Callable[[Queue, UserInit, int], None]
|
||||
|
||||
|
||||
def log_listener(q: Queue):
|
||||
@@ -62,7 +64,7 @@ def process_sigbus(*args):
|
||||
raise InputFileError("A worker process lost access to an input file")
|
||||
|
||||
|
||||
def process_init(q: Queue, user_init: Callable[[], None], loglevel):
|
||||
def process_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
||||
"""Initialize a process pool worker"""
|
||||
|
||||
# Ignore SIGINT (our parent process will kill us gracefully)
|
||||
@@ -85,7 +87,7 @@ def process_init(q: Queue, user_init: Callable[[], None], loglevel):
|
||||
return
|
||||
|
||||
|
||||
def thread_init(_queue: Queue, user_init: Callable[[], None], _loglevel):
|
||||
def thread_init(q: Queue, user_init: UserInit, loglevel) -> None:
|
||||
# As a thread, block SIGBUS so the main thread deals with it...
|
||||
with suppress(AttributeError):
|
||||
signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS})
|
||||
@@ -107,9 +109,9 @@ class StandardExecutor(Executor):
|
||||
task_finished: Callable,
|
||||
):
|
||||
if use_threads:
|
||||
log_queue = queue.Queue(-1)
|
||||
pool_class = ThreadPool
|
||||
initializer = thread_init
|
||||
log_queue: Queue = queue.Queue(-1)
|
||||
pool_class: Type[Pool] = ThreadPool
|
||||
initializer: WorkerInit = thread_init
|
||||
else:
|
||||
log_queue = multiprocessing.Queue(-1)
|
||||
pool_class = ProcessPool
|
||||
|
||||
@@ -11,7 +11,6 @@ import os
|
||||
from ocrmypdf import hookimpl
|
||||
from ocrmypdf._exec import tesseract
|
||||
from ocrmypdf.cli import numeric
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.helpers import clamp
|
||||
from ocrmypdf.pluginspec import OcrEngine
|
||||
from ocrmypdf.subprocess import check_external_program
|
||||
|
||||
+13
-8
@@ -6,7 +6,7 @@
|
||||
|
||||
|
||||
import argparse
|
||||
from typing import Optional, Type, TypeVar
|
||||
from typing import Any, Callable, Optional, TypeVar
|
||||
|
||||
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
||||
from ocrmypdf._version import __version__ as _VERSION
|
||||
@@ -14,7 +14,9 @@ from ocrmypdf._version import __version__ as _VERSION
|
||||
T = TypeVar('T')
|
||||
|
||||
|
||||
def numeric(basetype: Type[T], min_: Optional[T] = None, max_: Optional[T] = None):
|
||||
def numeric(
|
||||
basetype: Callable[[Any], T], min_: Optional[T] = None, max_: Optional[T] = None
|
||||
):
|
||||
"""Validator for numeric params"""
|
||||
min_ = basetype(min_) if min_ is not None else None
|
||||
max_ = basetype(max_) if max_ is not None else None
|
||||
@@ -22,7 +24,7 @@ def numeric(basetype: Type[T], min_: Optional[T] = None, max_: Optional[T] = Non
|
||||
def _numeric(string):
|
||||
value = basetype(string)
|
||||
if (min_ is not None and value < min_) or (max_ is not None and value > max_):
|
||||
msg = "%r not in valid range %r" % (string, (min_, max_))
|
||||
msg = f"{string!r} not in valid range {(min_, max_)!r}"
|
||||
raise argparse.ArgumentTypeError(msg)
|
||||
return value
|
||||
|
||||
@@ -145,7 +147,7 @@ Online documentation is located at:
|
||||
)
|
||||
parser.add_argument(
|
||||
'--output-type',
|
||||
choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3'],
|
||||
choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3', 'none'],
|
||||
default='pdfa',
|
||||
help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for "
|
||||
"long term archiving (default, recommended) but may not suitable "
|
||||
@@ -153,7 +155,8 @@ Online documentation is located at:
|
||||
"also has problems with full Unicode text. 'pdf' attempts to "
|
||||
"preserve file contents as much as possible. 'pdf-a1' creates a "
|
||||
"PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a "
|
||||
"PDF/A3-b file.",
|
||||
"PDF/A3-b file. 'none' will produce no output, which may be helpful if "
|
||||
"only the --sidecar is desired.",
|
||||
)
|
||||
|
||||
# Use null string '\0' as sentinel to indicate the user supplied no argument,
|
||||
@@ -338,8 +341,9 @@ Online documentation is located at:
|
||||
"Control how PDF is optimized after processing:"
|
||||
"0 - do not optimize; "
|
||||
"1 - do safe, lossless optimizations (default); "
|
||||
"2 - do some lossy optimizations; "
|
||||
"3 - do aggressive lossy optimizations (including lossy JBIG2)"
|
||||
"2 - do lossy JPEG and JPEG2000 optimizations; "
|
||||
"3 - do more aggressive lossy JPEG and JPEG2000 optimizations. "
|
||||
"To enable lossy JBIG2, see --jbig2-lossy."
|
||||
),
|
||||
)
|
||||
optimizing.add_argument(
|
||||
@@ -377,7 +381,8 @@ Online documentation is located at:
|
||||
action='store_true',
|
||||
help=(
|
||||
"Enable JBIG2 lossy mode (better compression, not suitable for some "
|
||||
"use cases - see documentation)."
|
||||
"use cases - see documentation). Only takes effect if --optimize 1 or "
|
||||
"higher is also enabled."
|
||||
),
|
||||
)
|
||||
optimizing.add_argument(
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
# © 2021 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
|
||||
"""Data files used to generate certain PDFs."""
|
||||
@@ -28,7 +28,7 @@ from enum import Enum, auto
|
||||
from itertools import islice, repeat, takewhile, zip_longest
|
||||
from multiprocessing import Pipe, Process
|
||||
from multiprocessing.connection import Connection, wait
|
||||
from typing import Callable, Iterable, Iterator
|
||||
from typing import Callable, Iterable, Iterator, List
|
||||
|
||||
from ocrmypdf import Executor, hookimpl
|
||||
from ocrmypdf._concurrent import NullProgressBar
|
||||
@@ -60,7 +60,9 @@ def process_sigbus(*args):
|
||||
|
||||
class ConnectionLogHandler(logging.handlers.QueueHandler):
|
||||
def __init__(self, conn: Connection) -> None:
|
||||
super().__init__(None)
|
||||
# sets the parent's queue to None - parent only touches queue
|
||||
# in enqueue() which we override
|
||||
super().__init__(None) # type: ignore
|
||||
self.conn = conn
|
||||
|
||||
def enqueue(self, record):
|
||||
@@ -126,8 +128,8 @@ class LambdaExecutor(Executor):
|
||||
if not grouped_args:
|
||||
return
|
||||
|
||||
processes = []
|
||||
connections = []
|
||||
processes: List[Process] = []
|
||||
connections: List[Connection] = []
|
||||
for chunk in grouped_args:
|
||||
parent_conn, child_conn = Pipe()
|
||||
|
||||
@@ -152,6 +154,8 @@ class LambdaExecutor(Executor):
|
||||
with self.pbar_class(**tqdm_kwargs) as pbar:
|
||||
while connections:
|
||||
for r in wait(connections):
|
||||
if not isinstance(r, Connection):
|
||||
raise NotImplementedError("We only support Connection()")
|
||||
try:
|
||||
msg_type, msg = r.recv()
|
||||
except EOFError:
|
||||
|
||||
+3
-11
@@ -189,7 +189,7 @@ def is_file_writable(test_file: os.PathLike) -> bool:
|
||||
with suppress(OSError):
|
||||
p.unlink()
|
||||
return True
|
||||
except (EnvironmentError, RuntimeError) as e:
|
||||
except (OSError, RuntimeError) as e:
|
||||
log.debug(e)
|
||||
log.error(str(e))
|
||||
return False
|
||||
@@ -221,15 +221,7 @@ def check_pdf(input_file: Path) -> bool:
|
||||
# If linearization is missing entirely, we do not complain. We do
|
||||
# complain if linearization is present but incorrect.
|
||||
pdf.check_linearization(sio)
|
||||
except RuntimeError:
|
||||
pass
|
||||
except (
|
||||
# Workaround for a problematic pikepdf version
|
||||
# pragma: no cover
|
||||
getattr(pikepdf, 'ForeignObjectError')
|
||||
if pikepdf.__version__ == '2.1.0'
|
||||
else NeverRaise
|
||||
):
|
||||
except (RuntimeError, pikepdf.ForeignObjectError):
|
||||
pass
|
||||
else:
|
||||
linearize_msgs = sio.getvalue()
|
||||
@@ -273,7 +265,7 @@ def deprecated(func):
|
||||
def new_func(*args, **kwargs):
|
||||
warnings.simplefilter('always', DeprecationWarning) # turn off filter
|
||||
warnings.warn(
|
||||
"Call to deprecated function {}.".format(func.__name__),
|
||||
f"Call to deprecated function {func.__name__}.",
|
||||
category=DeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
|
||||
@@ -349,7 +349,7 @@ class HocrTransform:
|
||||
interword_spaces: bool,
|
||||
show_bounding_boxes: bool,
|
||||
):
|
||||
if not line:
|
||||
if line is None:
|
||||
return
|
||||
pxl_line_coords = self.element_coordinates(line)
|
||||
line_box = self.pt_from_pixel(pxl_line_coords)
|
||||
|
||||
@@ -1,5 +1,4 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
#
|
||||
# © 2013-16: jbarlow83 from Github (https://github.com/jbarlow83)
|
||||
#
|
||||
@@ -13,6 +12,7 @@
|
||||
import argparse
|
||||
import logging
|
||||
import os
|
||||
import platform
|
||||
import sys
|
||||
import threading
|
||||
from collections import deque
|
||||
@@ -23,6 +23,7 @@ from functools import lru_cache
|
||||
from io import BytesIO, UnsupportedOperation
|
||||
from os import fspath
|
||||
from tempfile import TemporaryFile
|
||||
from typing import ContextManager, Type
|
||||
from warnings import warn
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
@@ -67,7 +68,7 @@ if os.name == 'nt':
|
||||
# Loading zlib from other places could cause a version mismatch
|
||||
_zlib_path = os.path.join(os.path.dirname(_libpath), 'zlib1.dll')
|
||||
if not os.path.exists(_zlib_path):
|
||||
_zlib_path = find_library('zlib')
|
||||
_zlib_path = find_library('zlib') or ''
|
||||
try:
|
||||
zlib = ffi.dlopen(_zlib_path)
|
||||
except ffi.error as e:
|
||||
@@ -86,7 +87,7 @@ except ffi.error as e:
|
||||
) from e
|
||||
|
||||
|
||||
class _LeptonicaErrorTrap_Redirect:
|
||||
class _LeptonicaErrorTrap_Redirect(ContextManager):
|
||||
"""
|
||||
Context manager to trap errors reported by Leptonica < 1.79 or on Apple Silicon.
|
||||
|
||||
@@ -132,7 +133,7 @@ class _LeptonicaErrorTrap_Redirect:
|
||||
except Exception:
|
||||
self.leptonica_lock.release()
|
||||
raise
|
||||
return self
|
||||
return
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
# Restore old stderr
|
||||
@@ -172,7 +173,7 @@ tls = threading.local()
|
||||
tls.trap = None
|
||||
|
||||
|
||||
class _LeptonicaErrorTrap_Queue:
|
||||
class _LeptonicaErrorTrap_Queue(ContextManager):
|
||||
def __init__(self):
|
||||
self.queue = deque()
|
||||
|
||||
@@ -226,7 +227,7 @@ except (ffi.error, MemoryError):
|
||||
# Pre-1.79 Leptonica does not have leptSetStderrHandler
|
||||
# And some platforms, notably Apple ARM 64, do not allow the write+execute
|
||||
# memory needed to set up the callback function.
|
||||
_LeptonicaErrorTrap = _LeptonicaErrorTrap_Redirect
|
||||
_LeptonicaErrorTrap: Type[ContextManager] = _LeptonicaErrorTrap_Redirect
|
||||
else:
|
||||
# 1.79 have this new symbol
|
||||
_LeptonicaErrorTrap = _LeptonicaErrorTrap_Queue
|
||||
@@ -272,7 +273,7 @@ class LeptonicaObject:
|
||||
# Leptonica API uses double-pointers for its destroy APIs to prevent
|
||||
# dangling pointers. This means we need to put our single pointer,
|
||||
# cdata, in a temporary CDATA**.
|
||||
pp = ffi.new('{} **'.format(cls.LEPTONICA_TYPENAME), cdata)
|
||||
pp = ffi.new(f'{cls.LEPTONICA_TYPENAME} **', cdata)
|
||||
cls.cdata_destroy(pp)
|
||||
|
||||
|
||||
@@ -439,6 +440,9 @@ class Pix(LeptonicaObject):
|
||||
bio = BytesIO()
|
||||
pillow_image.save(bio, format='png', compress_level=1)
|
||||
py_buffer = bio.getbuffer()
|
||||
if platform.python_implementation() == 'PyPy':
|
||||
# PyPy complains that it cannot do from_buffer(memoryview)
|
||||
py_buffer = bytes(py_buffer)
|
||||
c_buffer = ffi.from_buffer(py_buffer)
|
||||
with _LeptonicaErrorTrap():
|
||||
pix = Pix(lept.pixReadMem(c_buffer, len(c_buffer)))
|
||||
@@ -844,7 +848,7 @@ class Box(LeptonicaObject):
|
||||
|
||||
def __repr__(self):
|
||||
if self._cdata:
|
||||
return '<leptonica.Box x={0} y={1} w={2} h={3}>'.format(
|
||||
return '<leptonica.Box x={} y={} w={} h={}>'.format(
|
||||
self.x, self.y, self.w, self.h
|
||||
)
|
||||
return '<leptonica.Box NULL>'
|
||||
@@ -916,7 +920,7 @@ class Sel(LeptonicaObject):
|
||||
lines = [line.strip() for line in selstr.split('\n') if line.strip()]
|
||||
h = len(lines)
|
||||
w = len(lines[0])
|
||||
lengths = set(len(line) for line in lines)
|
||||
lengths = {len(line) for line in lines}
|
||||
if len(lengths) != 1:
|
||||
raise ValueError("All lines in selstr must be same length")
|
||||
|
||||
|
||||
+33
-18
@@ -25,8 +25,16 @@ from typing import (
|
||||
)
|
||||
|
||||
import img2pdf
|
||||
import pikepdf
|
||||
from pikepdf import Dictionary, Name, Object, Pdf, PdfImage
|
||||
from pikepdf import (
|
||||
Dictionary,
|
||||
Name,
|
||||
Object,
|
||||
ObjectStreamMode,
|
||||
Pdf,
|
||||
PdfImage,
|
||||
Stream,
|
||||
UnsupportedImageTypeError,
|
||||
)
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf import leptonica
|
||||
@@ -63,7 +71,7 @@ def jpg_name(root: Path, xref: Xref) -> Path:
|
||||
|
||||
|
||||
def extract_image_filter(
|
||||
pike: Pdf, root: Path, image: Object, xref: Xref
|
||||
pike: Pdf, root: Path, image: Stream, xref: Xref
|
||||
) -> Optional[Tuple[PdfImage, Tuple[Name, Object]]]:
|
||||
del pike # unused args
|
||||
del root
|
||||
@@ -89,7 +97,7 @@ def extract_image_filter(
|
||||
return None # Don't mess with wide gamut images
|
||||
|
||||
if filtdp[0] == Name.JPXDecode:
|
||||
log.debug(f"Skipping JPEG2000 iamge, xref {xref}")
|
||||
log.debug(f"Skipping JPEG2000 image, xref {xref}")
|
||||
return None # Don't do JPEG2000
|
||||
|
||||
if filtdp[0] == Name.CCITTFaxDecode and filtdp[1].get('/K', 0) >= 0:
|
||||
@@ -104,7 +112,7 @@ def extract_image_filter(
|
||||
|
||||
|
||||
def extract_image_jbig2(
|
||||
*, pike: pikepdf.Pdf, root: Path, image: Object, xref: Xref, options
|
||||
*, pike: Pdf, root: Path, image: Stream, xref: Xref, options
|
||||
) -> Optional[XrefExt]:
|
||||
del options # unused arg
|
||||
|
||||
@@ -123,16 +131,16 @@ def extract_image_jbig2(
|
||||
# Showing the palette or ICC to jbig2enc will cause it to perform
|
||||
# colorspace transform to 1bpp, which will conflict the palette or
|
||||
# ICC if it exists.
|
||||
colorspace = pim.obj.get(pikepdf.Name.ColorSpace, None)
|
||||
colorspace = pim.obj.get(Name.ColorSpace, None)
|
||||
if colorspace is not None or pim.image_mask:
|
||||
try:
|
||||
# Set to DeviceGray temporarily; we already in 1 bpc.
|
||||
pim.obj.ColorSpace = pikepdf.Name.DeviceGray
|
||||
pim.obj.ColorSpace = Name.DeviceGray
|
||||
imgname = root / f'{xref:08d}'
|
||||
with imgname.open('wb') as f:
|
||||
ext = pim.extract_to(stream=f)
|
||||
imgname.rename(imgname.with_suffix(ext))
|
||||
except pikepdf.UnsupportedImageTypeError:
|
||||
except UnsupportedImageTypeError:
|
||||
return None
|
||||
finally:
|
||||
# Restore image colorspace after temporarily setting it to DeviceGray
|
||||
@@ -145,7 +153,7 @@ def extract_image_jbig2(
|
||||
|
||||
|
||||
def extract_image_generic(
|
||||
*, pike: Pdf, root: Path, image: PdfImage, xref: Xref, options
|
||||
*, pike: Pdf, root: Path, image: Stream, xref: Xref, options
|
||||
) -> Optional[XrefExt]:
|
||||
result = extract_image_filter(pike, root, image, xref)
|
||||
if result is None:
|
||||
@@ -178,7 +186,7 @@ def extract_image_generic(
|
||||
with imgname.open('wb') as f:
|
||||
ext = pim.extract_to(stream=f)
|
||||
imgname.rename(imgname.with_suffix(ext))
|
||||
except pikepdf.UnsupportedImageTypeError:
|
||||
except UnsupportedImageTypeError:
|
||||
return None
|
||||
return XrefExt(xref, ext)
|
||||
elif (
|
||||
@@ -365,6 +373,7 @@ def convert_to_jbig2(
|
||||
When the JBIG2 symbolic coder is not used, each JBIG2 stands on its own
|
||||
and needs no dictionary. Currently this must be lossless JBIG2.
|
||||
"""
|
||||
jbig2_globals_dict: Optional[Dictionary]
|
||||
|
||||
_produce_jbig2_images(jbig2_groups, root, options, executor)
|
||||
|
||||
@@ -373,7 +382,7 @@ def convert_to_jbig2(
|
||||
jbig2_symfile = root / (prefix + '.sym')
|
||||
if jbig2_symfile.exists():
|
||||
jbig2_globals_data = jbig2_symfile.read_bytes()
|
||||
jbig2_globals = pikepdf.Stream(pike, jbig2_globals_data)
|
||||
jbig2_globals = Stream(pike, jbig2_globals_data)
|
||||
jbig2_globals_dict = Dictionary(JBIG2Globals=jbig2_globals)
|
||||
elif options.jbig2_page_group_size == 1:
|
||||
jbig2_globals_dict = None
|
||||
@@ -444,8 +453,8 @@ def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool:
|
||||
with output.open('wb') as f:
|
||||
img2pdf.convert(fspath(filename), outputstream=f)
|
||||
|
||||
with pikepdf.open(output) as pdf_image:
|
||||
foreign_image = next(pdf_image.pages[0].images.values())
|
||||
with Pdf.open(output) as pdf_image:
|
||||
foreign_image = next(iter(pdf_image.pages[0].images.values()))
|
||||
local_image = pike.copy_foreign(foreign_image)
|
||||
|
||||
im_obj = pike.get_object(xref, 0)
|
||||
@@ -524,12 +533,15 @@ def transcode_pngs(
|
||||
_transcode_png(pike, filename, xref)
|
||||
|
||||
|
||||
DEFAULT_EXECUTOR = SerialExecutor()
|
||||
|
||||
|
||||
def optimize(
|
||||
input_file: Path,
|
||||
output_file: Path,
|
||||
context,
|
||||
save_settings,
|
||||
executor: Executor = SerialExecutor(),
|
||||
executor: Executor = DEFAULT_EXECUTOR,
|
||||
) -> None:
|
||||
options = context.options
|
||||
if options.optimize == 0:
|
||||
@@ -543,7 +555,7 @@ def optimize(
|
||||
if options.jbig2_page_group_size == 0:
|
||||
options.jbig2_page_group_size = 10 if options.jbig2_lossy else 1
|
||||
|
||||
with pikepdf.Pdf.open(input_file) as pike:
|
||||
with Pdf.open(input_file) as pike:
|
||||
root = output_file.parent / 'images'
|
||||
root.mkdir(exist_ok=True)
|
||||
|
||||
@@ -573,9 +585,12 @@ def optimize(
|
||||
log.info(f"Optimize ratio: {ratio:.2f} savings: {(savings):.1%}")
|
||||
|
||||
if savings < 0:
|
||||
log.info("Image optimization did not improve the file - discarded")
|
||||
log.info(
|
||||
"Image optimization did not improve the file - "
|
||||
"optimizations will not be used"
|
||||
)
|
||||
# We still need to save the file
|
||||
with pikepdf.open(input_file) as pike:
|
||||
with Pdf.open(input_file) as pike:
|
||||
pike.remove_unreferenced_resources()
|
||||
pike.save(output_file, **save_settings)
|
||||
else:
|
||||
@@ -622,7 +637,7 @@ def main(infile, outfile, level, jobs=1):
|
||||
dict(
|
||||
compress_streams=True,
|
||||
preserve_pdfa=True,
|
||||
object_stream_mode=pikepdf.ObjectStreamMode.generate,
|
||||
object_stream_mode=ObjectStreamMode.generate,
|
||||
),
|
||||
)
|
||||
copy(fspath(tmpout), fspath(outfile))
|
||||
|
||||
+12
-6
@@ -13,13 +13,21 @@ import base64
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterator, Union
|
||||
|
||||
try:
|
||||
from importlib_resources import files as package_files
|
||||
except ImportError:
|
||||
from importlib.resources import files as package_files
|
||||
|
||||
import pikepdf
|
||||
import pkg_resources
|
||||
import pkg_resources # deprecated
|
||||
|
||||
# Deprecated
|
||||
ICC_PROFILE_RELPATH = 'data/sRGB.icc'
|
||||
|
||||
# Deprecated
|
||||
SRGB_ICC_PROFILE = pkg_resources.resource_filename('ocrmypdf', ICC_PROFILE_RELPATH)
|
||||
|
||||
SRGB_ICC_PROFILE_NAME = 'sRGB.icc'
|
||||
|
||||
|
||||
def _postscript_objdef(
|
||||
alias: str,
|
||||
@@ -97,12 +105,10 @@ def generate_pdfa_ps(target_filename: Path, icc: str = 'sRGB'):
|
||||
References:
|
||||
Adobe PDFMARK Reference: https://www.adobe.com/content/dam/acom/en/devnet/acrobat/pdfs/pdfmark_reference.pdf
|
||||
"""
|
||||
if icc == 'sRGB':
|
||||
icc_profile = SRGB_ICC_PROFILE
|
||||
else:
|
||||
if icc != 'sRGB':
|
||||
raise NotImplementedError("Only supporting sRGB")
|
||||
|
||||
bytes_icc_profile = Path(icc_profile).read_bytes()
|
||||
bytes_icc_profile = (package_files('ocrmypdf.data') / SRGB_ICC_PROFILE).read_bytes()
|
||||
ps = '\n'.join(_make_postscript(icc, bytes_icc_profile, 3))
|
||||
|
||||
# We should have encoded everything to pure ASCII by this point, and
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
import atexit
|
||||
import logging
|
||||
import re
|
||||
from collections import defaultdict, namedtuple
|
||||
from collections import defaultdict
|
||||
from contextlib import ExitStack
|
||||
from decimal import Decimal
|
||||
from enum import Enum
|
||||
@@ -17,11 +17,27 @@ from functools import partial
|
||||
from math import hypot, inf, isclose
|
||||
from os import PathLike
|
||||
from pathlib import Path
|
||||
from typing import Container, Iterator, Optional, Tuple, Union
|
||||
from typing import (
|
||||
Container,
|
||||
Dict,
|
||||
Iterator,
|
||||
List,
|
||||
Mapping,
|
||||
NamedTuple,
|
||||
Optional,
|
||||
Tuple,
|
||||
Union,
|
||||
)
|
||||
from warnings import warn
|
||||
|
||||
import pikepdf
|
||||
from pikepdf import Object, Pdf, PdfMatrix
|
||||
from pikepdf import (
|
||||
Object,
|
||||
Pdf,
|
||||
PdfImage,
|
||||
PdfInlineImage,
|
||||
PdfMatrix,
|
||||
parse_content_stream,
|
||||
)
|
||||
|
||||
from ocrmypdf._concurrent import Executor, SerialExecutor
|
||||
from ocrmypdf.exceptions import EncryptedPdfError, InputFileError
|
||||
@@ -36,7 +52,7 @@ Encoding = Enum(
|
||||
'Encoding', 'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate runlength'
|
||||
)
|
||||
|
||||
FRIENDLY_COLORSPACE = {
|
||||
FRIENDLY_COLORSPACE: Dict[str, Colorspace] = {
|
||||
'/DeviceGray': Colorspace.gray,
|
||||
'/CalGray': Colorspace.gray,
|
||||
'/DeviceRGB': Colorspace.rgb,
|
||||
@@ -54,7 +70,7 @@ FRIENDLY_COLORSPACE = {
|
||||
'/I': Colorspace.index,
|
||||
}
|
||||
|
||||
FRIENDLY_ENCODING = {
|
||||
FRIENDLY_ENCODING: Dict[str, Encoding] = {
|
||||
'/CCITTFaxDecode': Encoding.ccitt,
|
||||
'/DCTDecode': Encoding.jpeg,
|
||||
'/JPXDecode': Encoding.jpeg2000,
|
||||
@@ -68,7 +84,7 @@ FRIENDLY_ENCODING = {
|
||||
'/RL': Encoding.runlength,
|
||||
}
|
||||
|
||||
FRIENDLY_COMP = {
|
||||
FRIENDLY_COMP: Dict[Colorspace, int] = {
|
||||
Colorspace.gray: 1,
|
||||
Colorspace.rgb: 3,
|
||||
Colorspace.cmyk: 4,
|
||||
@@ -86,16 +102,30 @@ def _is_unit_square(shorthand):
|
||||
return all(isclose(a, b, rel_tol=1e-3) for a, b in pairwise)
|
||||
|
||||
|
||||
XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_depth'])
|
||||
class XobjectSettings(NamedTuple):
|
||||
name: str
|
||||
shorthand: Tuple[float, float, float, float, float, float]
|
||||
stack_depth: int
|
||||
|
||||
InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth'])
|
||||
|
||||
ContentsInfo = namedtuple(
|
||||
'ContentsInfo',
|
||||
['xobject_settings', 'inline_images', 'found_vector', 'found_text', 'name_index'],
|
||||
)
|
||||
class InlineSettings(NamedTuple):
|
||||
iimage: PdfInlineImage
|
||||
shorthand: Tuple[float, float, float, float, float, float]
|
||||
stack_depth: int
|
||||
|
||||
TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt'])
|
||||
|
||||
class ContentsInfo(NamedTuple):
|
||||
xobject_settings: List[XobjectSettings]
|
||||
inline_images: List[InlineSettings]
|
||||
found_vector: bool
|
||||
found_text: bool
|
||||
name_index: Mapping[str, List[XobjectSettings]]
|
||||
|
||||
|
||||
class TextboxInfo(NamedTuple):
|
||||
bbox: Tuple[float, float, float, float]
|
||||
is_visible: bool
|
||||
is_corrupt: bool
|
||||
|
||||
|
||||
class VectorMarker:
|
||||
@@ -146,8 +176,8 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
|
||||
stack = []
|
||||
ctm = PdfMatrix(initial_shorthand)
|
||||
xobject_settings = []
|
||||
inline_images = []
|
||||
xobject_settings: List[XobjectSettings] = []
|
||||
inline_images: List[InlineSettings] = []
|
||||
name_index = defaultdict(lambda: [])
|
||||
found_vector = False
|
||||
found_text = False
|
||||
@@ -157,9 +187,7 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops)
|
||||
|
||||
for n, graphobj in enumerate(
|
||||
_normalize_stack(
|
||||
pikepdf.parse_content_stream(contentstream, operator_whitelist)
|
||||
)
|
||||
_normalize_stack(parse_content_stream(contentstream, operator_whitelist))
|
||||
):
|
||||
operands, operator = graphobj
|
||||
if operator == 'q':
|
||||
@@ -185,7 +213,7 @@ def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
name=image_name, shorthand=ctm.shorthand, stack_depth=len(stack)
|
||||
)
|
||||
xobject_settings.append(settings)
|
||||
name_index[image_name].append(settings)
|
||||
name_index[str(image_name)].append(settings)
|
||||
elif operator == 'INLINE IMAGE': # BI/ID/EI are grouped into this
|
||||
iimage = operands[0]
|
||||
inline = InlineSettings(
|
||||
@@ -271,23 +299,28 @@ def _get_dpi(ctm_shorthand, image_size) -> Resolution:
|
||||
class ImageInfo:
|
||||
DPI_PREC = Decimal('1.000')
|
||||
|
||||
_comp: Optional[int]
|
||||
_name: str
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
name='',
|
||||
pdfimage: Optional[Object] = None,
|
||||
inline: Optional[Object] = None,
|
||||
inline: Optional[PdfInlineImage] = None,
|
||||
shorthand=None,
|
||||
):
|
||||
self._name = str(name)
|
||||
self._shorthand = shorthand
|
||||
|
||||
pim: Union[PdfInlineImage, PdfImage]
|
||||
|
||||
if inline is not None:
|
||||
self._origin = 'inline'
|
||||
pim = inline.iimage
|
||||
pim = inline
|
||||
elif pdfimage is not None:
|
||||
self._origin = 'xobject'
|
||||
pim = pikepdf.PdfImage(pdfimage)
|
||||
pim = PdfImage(pdfimage)
|
||||
else:
|
||||
raise ValueError("Either pdfimage or inline must be set")
|
||||
self._width = pim.width
|
||||
@@ -303,14 +336,14 @@ class ImageInfo:
|
||||
|
||||
self._bpc = int(pim.bits_per_component)
|
||||
try:
|
||||
self._enc = FRIENDLY_ENCODING.get(pim.filters[0], 'image')
|
||||
self._enc = FRIENDLY_ENCODING.get(pim.filters[0])
|
||||
except IndexError:
|
||||
self._enc = '?'
|
||||
self._enc = None
|
||||
|
||||
try:
|
||||
self._color = FRIENDLY_COLORSPACE.get(pim.colorspace, '?')
|
||||
self._color = FRIENDLY_COLORSPACE.get(pim.colorspace or '')
|
||||
except NotImplementedError:
|
||||
self._color = '?'
|
||||
self._color = None
|
||||
if self._enc == Encoding.jpeg2000:
|
||||
self._color = Colorspace.jpeg2000
|
||||
|
||||
@@ -324,11 +357,14 @@ class ImageInfo:
|
||||
else:
|
||||
self._comp = 3
|
||||
else:
|
||||
self._comp = FRIENDLY_COMP.get(self._color, '?')
|
||||
if isinstance(self._color, Colorspace):
|
||||
self._comp = FRIENDLY_COMP.get(self._color)
|
||||
else:
|
||||
self._comp = None
|
||||
|
||||
# Bit of a hack... infer grayscale if component count is uncertain
|
||||
# but encoding only supports monochrome.
|
||||
if self._comp == '?' and self._enc in (Encoding.ccitt, Encoding.jbig2):
|
||||
if self._comp is None and self._enc in (Encoding.ccitt, Encoding.jbig2):
|
||||
self._comp = FRIENDLY_COMP[Colorspace.gray]
|
||||
|
||||
@property
|
||||
@@ -353,15 +389,15 @@ class ImageInfo:
|
||||
|
||||
@property
|
||||
def color(self):
|
||||
return self._color
|
||||
return self._color if self._color is not None else '?'
|
||||
|
||||
@property
|
||||
def comp(self):
|
||||
return self._comp
|
||||
return self._comp if self._comp is not None else '?'
|
||||
|
||||
@property
|
||||
def enc(self):
|
||||
return self._enc
|
||||
return self._enc if self._enc is not None else 'image'
|
||||
|
||||
@property
|
||||
def renderable(self):
|
||||
@@ -388,7 +424,7 @@ def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
||||
|
||||
for n, inline in enumerate(contentsinfo.inline_images):
|
||||
yield ImageInfo(
|
||||
name='inline-%02d' % n, shorthand=inline.shorthand, inline=inline
|
||||
name='inline-%02d' % n, shorthand=inline.shorthand, inline=inline.iimage
|
||||
)
|
||||
|
||||
|
||||
@@ -413,7 +449,7 @@ def _image_xobjects(container) -> Iterator[Tuple[Object, str]]:
|
||||
xobjs = resources['/XObject'].as_dict()
|
||||
for xobj in xobjs:
|
||||
candidate: Object = xobjs[xobj]
|
||||
if not '/Subtype' in candidate:
|
||||
if '/Subtype' not in candidate:
|
||||
continue
|
||||
if candidate['/Subtype'] == '/Image':
|
||||
pdfimage = candidate
|
||||
@@ -583,7 +619,7 @@ def _pdf_pageinfo_sync_init(pdf: Pdf, infile: Path, pdfminer_loglevel):
|
||||
|
||||
# If the pdf is not opened, open a copy for our worker process to use
|
||||
if pdf is None:
|
||||
worker_pdf = pikepdf.open(infile)
|
||||
worker_pdf = Pdf.open(infile)
|
||||
|
||||
def on_process_close():
|
||||
worker_pdf.close()
|
||||
@@ -597,7 +633,7 @@ def _pdf_pageinfo_sync(args):
|
||||
pdf = thread_pdf if thread_pdf is not None else worker_pdf
|
||||
with ExitStack() as stack:
|
||||
if not pdf: # When called with SerialExecutor
|
||||
pdf = stack.enter_context(pikepdf.open(infile))
|
||||
pdf = stack.enter_context(Pdf.open(infile))
|
||||
page = PageInfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
||||
return page
|
||||
|
||||
@@ -661,6 +697,10 @@ def _pdf_pageinfo_concurrent(
|
||||
|
||||
|
||||
class PageInfo:
|
||||
_has_text: Optional[bool]
|
||||
_has_vector: Optional[bool]
|
||||
_images: List[ImageInfo]
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
pdf: Pdf,
|
||||
@@ -732,7 +772,7 @@ class PageInfo:
|
||||
else:
|
||||
self._has_vector = None # i.e. "no information"
|
||||
self._has_text = None
|
||||
self._images = None
|
||||
self._images = []
|
||||
|
||||
self._dpi = None
|
||||
if self._images:
|
||||
@@ -749,7 +789,7 @@ class PageInfo:
|
||||
|
||||
@property
|
||||
def has_text(self) -> bool:
|
||||
return self._has_text
|
||||
return bool(self._has_text)
|
||||
|
||||
@property
|
||||
def has_corrupt_text(self) -> bool:
|
||||
@@ -759,7 +799,7 @@ class PageInfo:
|
||||
|
||||
@property
|
||||
def has_vector(self) -> bool:
|
||||
return self._has_vector
|
||||
return bool(self._has_vector)
|
||||
|
||||
@property
|
||||
def width_inches(self) -> Decimal:
|
||||
@@ -837,6 +877,9 @@ class PageInfo:
|
||||
)
|
||||
|
||||
|
||||
DEFAULT_EXECUTOR = SerialExecutor()
|
||||
|
||||
|
||||
class PdfInfo:
|
||||
"""Get summary information about a PDF"""
|
||||
|
||||
@@ -848,13 +891,13 @@ class PdfInfo:
|
||||
progbar: bool = False,
|
||||
max_workers: int = None,
|
||||
check_pages=None,
|
||||
executor: Executor = SerialExecutor(),
|
||||
executor: Executor = DEFAULT_EXECUTOR,
|
||||
):
|
||||
self._infile = infile
|
||||
if check_pages is None:
|
||||
check_pages = range(0, 1_000_000_000)
|
||||
|
||||
with pikepdf.open(infile) as pdf:
|
||||
with Pdf.open(infile) as pdf:
|
||||
if pdf.is_encrypted:
|
||||
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
||||
self._pages = _pdf_pageinfo_concurrent(
|
||||
|
||||
@@ -135,7 +135,7 @@ class LTStateAwareChar(LTChar):
|
||||
return self._text
|
||||
|
||||
def __repr__(self):
|
||||
return '<%s %s matrix=%s rendermode=%r font=%r adv=%s text=%r>' % (
|
||||
return '<{} {} matrix={} rendermode={!r} font={!r} adv={} text={!r}>'.format(
|
||||
self.__class__.__name__,
|
||||
bbox2str(self.bbox),
|
||||
matrix2str(self.matrix),
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
|
||||
from abc import ABC, abstractmethod, abstractstaticmethod
|
||||
from abc import ABC, abstractmethod
|
||||
from argparse import ArgumentParser, Namespace
|
||||
from collections import namedtuple
|
||||
from logging import Handler
|
||||
@@ -197,7 +197,7 @@ def rasterize_pdf_page(
|
||||
|
||||
|
||||
@hookspec(firstresult=True)
|
||||
def filter_ocr_image(page: 'PageContext', image: 'Image') -> 'Image':
|
||||
def filter_ocr_image(page: 'PageContext', image: 'Image.Image') -> 'Image.Image':
|
||||
"""Called to filter the image before it is sent to OCR.
|
||||
|
||||
This is the image that OCR sees, not what the user sees when they view the
|
||||
@@ -325,11 +325,13 @@ class OcrEngine(ABC):
|
||||
Tesseract OCR.
|
||||
"""
|
||||
|
||||
@abstractstaticmethod
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def version() -> str:
|
||||
"""Returns the version of the OCR engine."""
|
||||
|
||||
@abstractstaticmethod
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def creator_tag(options: Namespace) -> str:
|
||||
"""Returns the creator tag to identify this software's role in creating the PDF.
|
||||
|
||||
@@ -349,24 +351,28 @@ class OcrEngine(ABC):
|
||||
to the user, usually in an error message.
|
||||
"""
|
||||
|
||||
@abstractstaticmethod
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def languages(options: Namespace) -> AbstractSet[str]:
|
||||
"""Returns the set of all languages that are supported by the engine.
|
||||
|
||||
Languages are typically given in 3-letter ISO 3166-1 codes, but actually
|
||||
can be any value understood by the OCR engine."""
|
||||
|
||||
@abstractstaticmethod
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def get_orientation(input_file: Path, options: Namespace) -> OrientationConfidence:
|
||||
"""Returns the orientation of the image."""
|
||||
|
||||
@abstractstaticmethod
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def generate_hocr(
|
||||
input_file: Path, output_hocr: Path, output_text: Path, options: Namespace
|
||||
) -> None:
|
||||
"""Called to produce a hOCR file and sidecar text file."""
|
||||
|
||||
@abstractstaticmethod
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def generate_pdf(
|
||||
input_file: Path, output_pdf: Path, output_text: Path, options: Namespace
|
||||
) -> None:
|
||||
|
||||
@@ -15,7 +15,6 @@ from collections.abc import Mapping
|
||||
from contextlib import suppress
|
||||
from distutils.version import LooseVersion, Version
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||
from subprocess import run as subprocess_run
|
||||
from typing import Callable, Optional, Type, Union
|
||||
@@ -27,7 +26,9 @@ from ocrmypdf.exceptions import MissingDependencyError
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs):
|
||||
def run(
|
||||
args, *, env=None, logs_errors_to_stdout: bool = False, **kwargs
|
||||
) -> CompletedProcess:
|
||||
"""Wrapper around :py:func:`subprocess.run`
|
||||
|
||||
The main purpose of this wrapper is to log subprocess output in an orderly
|
||||
@@ -65,7 +66,9 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs):
|
||||
return proc
|
||||
|
||||
|
||||
def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs):
|
||||
def run_polling_stderr(
|
||||
args, *, callback: Callable[[str], None], check: bool = False, env=None, **kwargs
|
||||
) -> CompletedProcess:
|
||||
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
||||
|
||||
Every line of produced by stderr will be forwarded to the callback function.
|
||||
@@ -83,6 +86,8 @@ def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs):
|
||||
with Popen(args, env=env, **kwargs) as proc:
|
||||
lines = []
|
||||
while proc.poll() is None:
|
||||
if proc.stderr is None:
|
||||
continue
|
||||
for msg in iter(proc.stderr.readline, ''):
|
||||
if process_log.isEnabledFor(logging.DEBUG):
|
||||
process_log.debug(msg.strip())
|
||||
@@ -102,7 +107,7 @@ def _fix_process_args(args, env, kwargs):
|
||||
env = os.environ
|
||||
|
||||
# Search in spoof path if necessary
|
||||
program = args[0]
|
||||
program = str(args[0])
|
||||
|
||||
if os.name == 'nt':
|
||||
from ocrmypdf.subprocess._windows import fix_windows_args
|
||||
|
||||
@@ -9,9 +9,9 @@ import os
|
||||
import shutil
|
||||
import sys
|
||||
from distutils.version import LooseVersion
|
||||
from itertools import chain, filterfalse
|
||||
from itertools import chain
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Iterator, Optional, Tuple, TypeVar, cast
|
||||
from typing import Any, Callable, Iterable, Iterator, Set, Tuple, TypeVar
|
||||
|
||||
try:
|
||||
import winreg
|
||||
@@ -113,7 +113,7 @@ SHIMS = [
|
||||
]
|
||||
|
||||
|
||||
def fix_windows_args(program, args, env):
|
||||
def fix_windows_args(program: str, args, env):
|
||||
"""Adjust our desired program and command line arguments for use on Windows"""
|
||||
|
||||
if sys.version_info < (3, 8):
|
||||
@@ -137,14 +137,12 @@ def fix_windows_args(program, args, env):
|
||||
return args
|
||||
|
||||
|
||||
def unique_everseen(iterable, key=None):
|
||||
"List unique elements, preserving order. Remember all elements ever seen."
|
||||
def unique_everseen(iterable: Iterable[T], key: Callable[[T], T]) -> Iterator[T]:
|
||||
"List unique elements, preserving order."
|
||||
# unique_everseen('AAAABBBCCDAABBB') --> A B C D
|
||||
# unique_everseen('ABBCcAD', str.lower) --> A B C D
|
||||
seen = set()
|
||||
seen: Set[T] = set()
|
||||
seen_add = seen.add
|
||||
if key is None:
|
||||
key = lambda x: x
|
||||
for element in iterable:
|
||||
k = key(element)
|
||||
if k not in seen:
|
||||
|
||||
@@ -69,6 +69,11 @@ def outpdf(tmp_path):
|
||||
return tmp_path / 'out.pdf'
|
||||
|
||||
|
||||
@pytest.fixture(scope="function")
|
||||
def outtxt(tmp_path):
|
||||
return tmp_path / 'out.txt'
|
||||
|
||||
|
||||
@pytest.fixture(scope="function")
|
||||
def no_outpdf(tmp_path):
|
||||
"""This just documents the fact that a test is not expected to produce
|
||||
|
||||
@@ -87,4 +87,4 @@ def test_jpeg_in_jpeg_out(resources, outpdf):
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
with pikepdf.open(outpdf) as pdf:
|
||||
assert next(pdf.pages[0].images.values()).Filter == pikepdf.Name.DCTDecode
|
||||
assert next(iter(pdf.pages[0].images.values())).Filter == pikepdf.Name.DCTDecode
|
||||
|
||||
+33
-8
@@ -701,7 +701,7 @@ def test_sidecar_pagecount(resources, outpdf):
|
||||
pdfinfo = PdfInfo(resources / '3small.pdf')
|
||||
num_pages = len(pdfinfo)
|
||||
|
||||
with open(sidecar, 'r', encoding='utf-8') as f:
|
||||
with open(sidecar, encoding='utf-8') as f:
|
||||
ocr_text = f.read()
|
||||
|
||||
# There should a formfeed between each pair of pages, so the count of
|
||||
@@ -722,7 +722,7 @@ def test_sidecar_nonempty(resources, outpdf):
|
||||
'tests/plugins/tesseract_cache.py',
|
||||
)
|
||||
|
||||
with open(sidecar, 'r', encoding='utf-8') as f:
|
||||
with open(sidecar, encoding='utf-8') as f:
|
||||
ocr_text = f.read()
|
||||
assert 'the' in ocr_text
|
||||
|
||||
@@ -745,14 +745,13 @@ def test_pdfa_n(pdfa_level, resources, outpdf):
|
||||
assert pdfa_info['conformance'] == f'PDF/A-{pdfa_level}B'
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
PIL.__version__ < '5.0.0', reason="Pillow < 5.0.0 doesn't raise the exception"
|
||||
)
|
||||
@pytest.mark.slow
|
||||
def test_decompression_bomb(resources, outpdf):
|
||||
def test_decompression_bomb_error(resources, outpdf):
|
||||
p, _out, err = run_ocrmypdf(resources / 'hugemono.pdf', outpdf)
|
||||
assert 'decompression bomb' in err
|
||||
assert 'decompression bomb' in err and '--max-image-mpixels' in err
|
||||
|
||||
|
||||
@pytest.mark.slow
|
||||
def test_decompression_bomb_succeeds(resources, outpdf):
|
||||
p, _out, err = run_ocrmypdf(
|
||||
resources / 'hugemono.pdf', outpdf, '--max-image-mpixels', '2000'
|
||||
)
|
||||
@@ -881,3 +880,29 @@ def test_image_dpi_threshold(resources, outpdf):
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
assert outpdf.exists()
|
||||
|
||||
|
||||
def test_outputtype_none_bad_setup(resources, outpdf):
|
||||
p, _out, err = run_ocrmypdf(
|
||||
resources / 'trivial.pdf',
|
||||
outpdf,
|
||||
'--output-type=none',
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
assert p.returncode == ExitCode.bad_args
|
||||
assert 'Set the output file to' in err
|
||||
|
||||
|
||||
def test_outputtype_none(resources, outtxt):
|
||||
p, _out, err = run_ocrmypdf(
|
||||
resources / 'trivial.pdf',
|
||||
os.devnull,
|
||||
'--output-type=none',
|
||||
'--sidecar',
|
||||
outtxt,
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
assert p.returncode == ExitCode.ok
|
||||
assert outtxt.exists()
|
||||
|
||||
@@ -287,7 +287,7 @@ def test_srgb_in_unicode_path(tmp_path):
|
||||
|
||||
|
||||
def test_kodak_toc(resources, outpdf):
|
||||
_output = check_ocrmypdf(
|
||||
check_ocrmypdf(
|
||||
resources / 'kcs.pdf',
|
||||
outpdf,
|
||||
'--output-type',
|
||||
|
||||
@@ -50,10 +50,6 @@ def test_nonmonotonic_warning(caplog):
|
||||
assert 'out of order' in caplog.text
|
||||
|
||||
|
||||
def test_list_range():
|
||||
assert _pages_from_ranges([0, 1, 2]) == {0, 1, 2}
|
||||
|
||||
|
||||
def test_limited_pages(resources, outpdf):
|
||||
multi = resources / 'multipage.pdf'
|
||||
ocrmypdf.ocr(
|
||||
|
||||
@@ -79,9 +79,7 @@ def test_dpi_needed(image, text, vector, result, rgb_image, outdir):
|
||||
# Input:
|
||||
('', '', '', '', ''),
|
||||
# Output:
|
||||
(
|
||||
((1, 5), None),
|
||||
),
|
||||
(((1, 5), None),),
|
||||
),
|
||||
(
|
||||
'no_empty_values',
|
||||
@@ -147,4 +145,4 @@ def test_dpi_needed(image, text, vector, result, rgb_image, outdir):
|
||||
),
|
||||
)
|
||||
def test_enumerate_compress_ranges(name, input, output):
|
||||
assert output == tuple(_pipeline.enumerate_compress_ranges(input))
|
||||
assert output == tuple(_pipeline.enumerate_compress_ranges(input))
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
|
||||
import logging
|
||||
import os
|
||||
from unittest.mock import patch
|
||||
|
||||
import pikepdf
|
||||
@@ -237,6 +238,13 @@ def test_version_comparison():
|
||||
need_version='4.0.0',
|
||||
version_parser=TesseractVersion,
|
||||
)
|
||||
vd.check_external_program(
|
||||
program="tesseract",
|
||||
package="tesseract",
|
||||
version_checker=lambda: '5.0.0-rc1.20211030',
|
||||
need_version='4.0.0',
|
||||
version_parser=TesseractVersion,
|
||||
)
|
||||
vd.check_external_program(
|
||||
program="tesseract",
|
||||
package="tesseract",
|
||||
@@ -298,3 +306,8 @@ def test_sidecar_equals_output(resources, no_outpdf):
|
||||
op = no_outpdf
|
||||
with pytest.raises(BadArgsError, match=r'--sidecar'):
|
||||
run_ocrmypdf_api(resources / 'trivial.pdf', op, '--sidecar', op)
|
||||
|
||||
|
||||
def test_devnull_sidecar(resources):
|
||||
with pytest.raises(BadArgsError, match=r'--sidecar.*NUL'):
|
||||
run_ocrmypdf_api(resources / 'trivial.pdf', os.devnull, '--sidecar')
|
||||
|
||||
Reference in New Issue
Block a user