Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
bd0f005861 | ||
|
|
6ba4b7b3f3 | ||
|
|
2c11349ee8 | ||
|
|
b0afef09ef | ||
|
|
72fa347c38 | ||
|
|
96d68c2413 | ||
|
|
babc76fa74 | ||
|
|
dc06990e5d | ||
|
|
0ff0d2f8d1 | ||
|
|
81602cf420 | ||
|
|
607e2d7e81 | ||
|
|
b01d9e07e8 | ||
|
|
91db94cf2e | ||
|
|
416df803d4 | ||
|
|
037b96ca16 | ||
|
|
bb258fc99c | ||
|
|
4b8ccbe8cb | ||
|
|
ab1ff3331b | ||
|
|
3675ae918c | ||
|
|
0ba32b96b7 | ||
|
|
add64e4fa2 | ||
|
|
7fe2954ede | ||
|
|
ad202693b3 | ||
|
|
594ef83551 | ||
|
|
78b71618c1 | ||
|
|
b8aa89e1ec | ||
|
|
b4c1f66bc1 | ||
|
|
5172dbde8d | ||
|
|
d2908640c6 | ||
|
|
997bf7578d | ||
|
|
043258242c | ||
|
|
156d5d9a9c | ||
|
|
0b7e52fb5e | ||
|
|
a5feef07d0 | ||
|
|
f11bb53e61 | ||
|
|
68a57a7839 | ||
|
|
4194430dc1 | ||
|
|
a707c56fae | ||
|
|
3cba50bfbd | ||
|
|
ed5e17d0a4 | ||
|
|
ce0e0ecd4d | ||
|
|
7e1223c12c | ||
|
|
b83d7f6d1a | ||
|
|
80e957908a | ||
|
|
f0e7bea8ba | ||
|
|
0cdb9bd04a |
+1
-1
@@ -32,7 +32,7 @@ stages:
|
||||
choco install --yes --no-progress --pre tesseract
|
||||
choco install --yes --no-progress python3
|
||||
choco install --yes --no-progress ghostscript
|
||||
choco install --yes --no-progress pngquant
|
||||
# choco install --yes --no-progress pngquant
|
||||
displayName: "Install system packages"
|
||||
- pwsh: |
|
||||
refreshenv
|
||||
|
||||
+2
-3
@@ -125,8 +125,7 @@ include:
|
||||
.. envvar:: OMP_THREAD_LIMIT
|
||||
|
||||
Controls the number of threads Tesseract will use. OCRmyPDF will
|
||||
manage this environment if it is not already set. (Currently, it will
|
||||
set it to 1 because this gives the best results in testing.)
|
||||
manage this environment variable if it is not already set.
|
||||
|
||||
For example, if you have a development build of Tesseract don't wish to
|
||||
use the system installation, you can launch OCRmyPDF as follows:
|
||||
@@ -315,7 +314,7 @@ message is:
|
||||
.. code-block:: none
|
||||
|
||||
Temporary working files retained at:
|
||||
/tmp/com.github.ocrmypdf.u20wpz07
|
||||
/tmp/ocrmypdf.io.u20wpz07
|
||||
|
||||
The organization of this folder is an implementation detail and subject
|
||||
to change between releases. However the general organization is that
|
||||
|
||||
+2
-3
@@ -20,7 +20,7 @@ and largely have the same functions.
|
||||
|
||||
import ocrmypdf
|
||||
|
||||
if __name__ == '__main__': # To ensure correct behavior on Windows
|
||||
if __name__ == '__main__': # To ensure correct behavior on Windows and macOS
|
||||
ocrmypdf.ocr('input.pdf', 'output.pdf', deskew=True)
|
||||
|
||||
With a few exceptions, all of the command line arguments are available
|
||||
@@ -42,8 +42,7 @@ execution. To do this, it will:
|
||||
- execute other subprocesses (forking and executing other programs)
|
||||
|
||||
The Python process that calls ``ocrmypdf.ocr()`` must be sufficiently
|
||||
privileged to perform these actions. If it is not, ``ocrmypdf()`` will
|
||||
fail.
|
||||
privileged to perform these actions.
|
||||
|
||||
There is no currently no option to manage how jobs are scheduled other
|
||||
than the argument ``jobs=`` which will limit the number of worker
|
||||
|
||||
+19
-20
@@ -495,10 +495,6 @@ Installing on Windows
|
||||
Native Windows
|
||||
--------------
|
||||
|
||||
.. note::
|
||||
|
||||
It is easier to install OCRmyPDF on Windows Subsystem for Linux.
|
||||
|
||||
.. note::
|
||||
|
||||
Administrator privileges will be required for some of these steps.
|
||||
@@ -509,30 +505,33 @@ You must install the following for Windows:
|
||||
* Tesseract 4.0 or later
|
||||
* Ghostscript 9.50 or later
|
||||
|
||||
You can install these with the Chocolatey package manager:
|
||||
Using the `Chocolatey <https://chocolatey.org/>`_ package manager, install the
|
||||
following when running in an Administrator command prompt:
|
||||
|
||||
* ``choco install python3``
|
||||
* ``choco install --pre tesseract``
|
||||
* ``choco install ghostscript``
|
||||
* ``choco install pngquant`` (optional)
|
||||
|
||||
Also consider adding:
|
||||
The commands above will install Python 3.x (latest version), Tesseract, Ghostscript
|
||||
and pngquant. Chocolatey may also need to install the Windows Visual C++ Runtime
|
||||
DLLs or other Windows patches, and may require a reboot.
|
||||
|
||||
* ``choco install pngquant``
|
||||
You may then use ``pip`` to install ocrmypdf. (This can performed by a user or
|
||||
Administrator.):
|
||||
|
||||
Windows 10 64-bit and 64-bit versions of applications are recommended. Earlier
|
||||
versions of Windows and 32-bit versions of these programs are not tested, and not
|
||||
supported at this time.
|
||||
* ``pip install ocrmypdf
|
||||
|
||||
OCRmyPDF will check for Tesseract-OCR and Ghostscript in your Program Files folder.
|
||||
If they are in some other location, you may need to modify the ``PATH``
|
||||
environment variable so Tesseract, Ghostscript, and other any optional executables can
|
||||
be found. You can enter it in the command line or
|
||||
`follow these directions <https://www.computerhope.com/issues/ch000549.htm#dospath>`_
|
||||
to make the change persistent and system-wide.
|
||||
Chocolatey automatically selects appropriate versions of these applications. If you
|
||||
are installing them manually, please install 64-bit versions of all applications for
|
||||
64-bit Windows, or 32-bit versions of all applications for 32-bit Windows. Mixing
|
||||
the "bitness" of these programs will lead to errors.
|
||||
|
||||
You may then use pip to install ocrmypdf:
|
||||
|
||||
* ``pip install ocrmypdf``
|
||||
OCRmyPDF will check the Windows Registry and standard locations in your Program Files
|
||||
for third party software it needs (specifically, Tesseract and Ghostscript). To
|
||||
override the versions OCRmyPDF selects, you can modify the ``PATH`` environment
|
||||
variable. `Follow these directions <https://www.computerhope.com/issues/ch000549.htm#dospath>`_
|
||||
to change the PATH.
|
||||
|
||||
Windows Subsystem for Linux
|
||||
---------------------------
|
||||
@@ -638,7 +637,7 @@ Installing with Python pip
|
||||
OCRmyPDF is delivered by PyPI because it is a convenient way to install
|
||||
the latest version. However, PyPI and ``pip`` cannot address the fact
|
||||
that ``ocrmypdf`` depends on certain non-Python system libraries and
|
||||
programs being instsalled.
|
||||
programs being installed.
|
||||
|
||||
For best results, first install `your platform's
|
||||
version <https://repology.org/metapackage/ocrmypdf/versions>`__ of
|
||||
|
||||
@@ -12,6 +12,48 @@ may be unreliable. Use the API to depend on precise behavior.
|
||||
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||
wish to use some of its features for working with PDFs.
|
||||
|
||||
v11.4.3
|
||||
=======
|
||||
|
||||
- Removed a redundant debug message.
|
||||
- Test suite now asserts that most patched functions are called when they should be.
|
||||
- Test suite now skips a test that fails on two particular versions of piekpdf.
|
||||
|
||||
v11.4.2
|
||||
=======
|
||||
|
||||
- Fixed support for Cygwin, hopefully.
|
||||
- watcher.py: Fixed an issue with the OCR_LOGLEVEL not being interpreted.
|
||||
|
||||
v11.4.1
|
||||
=======
|
||||
|
||||
- Fixed an issue where invalid pages ranges passed using the ``pages`` argument,
|
||||
such as "1-0" would cause unhandled exceptions.
|
||||
- Accepted a user-contributed to the Synology demo script in misc/synology.py.
|
||||
- Clarified documentation about change of temporary file location ``ocrmypdf.io``.
|
||||
- Fixed Python wheel tag which was incorrectly set to py35 even though we long
|
||||
since dropped support for Python 3.5.
|
||||
|
||||
v11.4.0
|
||||
=======
|
||||
|
||||
- When looking for Tesseract and Ghostscript, we now check the Windows Registry to
|
||||
see if their installers registered the location of their executables. This should
|
||||
help Windows users who have installed these programs to non-standard
|
||||
locations.
|
||||
- We now report on the progress of PDF/A conversion, since this operation is
|
||||
sometimes slow.
|
||||
- Improved command line completions.
|
||||
- The prefix of the temporary folder OCRmyPDF creates has been changed from
|
||||
``com.github.ocrmypdf`` to ``ocrmypdf.io``. Scripts that chose to depend on this
|
||||
prefix may need to be adjusted. (This has always been an implementation detail so is
|
||||
not considered part of the semantic versioning "contract".)
|
||||
- Fixed issue #692, where a particular file with malformed fonts would flood an
|
||||
internal message cue by generating so many debug messages.
|
||||
- Fixed an exception on processing hOCR files with no page record. Tesseract
|
||||
is not known to generate such files.
|
||||
|
||||
v11.3.4
|
||||
=======
|
||||
|
||||
|
||||
@@ -59,7 +59,8 @@ complete -c ocrmypdf -x -l output-type -a '(__fish_ocrmypdf_output_type)' -d "se
|
||||
|
||||
function __fish_ocrmypdf_pdf_renderer
|
||||
echo -e "auto\t"(_ "auto select PDF renderer")
|
||||
echo -e "hocr\t"(_ "use hocr renderer")
|
||||
echo -e "hocr\t"(_ "use hOCR renderer")
|
||||
echo -e "hocrdebug\t"(_ "uses hOCR renderer in debug mode, showing recognized text")
|
||||
echo -e "sandwich\t"(_ "use sandwich renderer")
|
||||
end
|
||||
complete -c ocrmypdf -x -l pdf-renderer -a '(__fish_ocrmypdf_pdf_renderer)' -d "select PDF renderer options"
|
||||
@@ -135,4 +136,4 @@ complete -c ocrmypdf -r -l user-words -d "specify location of user words file"
|
||||
complete -c ocrmypdf -r -l user-patterns -d "specify location of user patterns file"
|
||||
complete -c ocrmypdf -x -l fast-web-view -d "if file size if above this amount in MB, linearize PDF"
|
||||
|
||||
complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf)"
|
||||
complete -c ocrmypdf -x -a "(__fish_complete_suffix .pdf; __fish_complete_suffix .PDF; __fish_complete_suffix .jpg; __fish_complete_suffix .png)"
|
||||
|
||||
+3
-1
@@ -79,8 +79,10 @@ for dir_name, subdirs, file_list in os.walk(start_dir):
|
||||
stdout=output_file,
|
||||
stderr=subprocess.PIPE,
|
||||
check=False,
|
||||
text=True,
|
||||
errors='ignore',
|
||||
)
|
||||
logging.info(proc.stderr.read())
|
||||
logging.info(proc.stderr)
|
||||
os.chmod(full_path_ocr, 0o664)
|
||||
os.chmod(full_path, 0o664)
|
||||
full_path_ocr_archive = sys.argv[2]
|
||||
|
||||
+8
-3
@@ -44,7 +44,7 @@ DESKEW = bool(os.getenv('OCR_DESKEW', ''))
|
||||
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
||||
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
||||
USE_POLLING = bool(os.getenv('OCR_USE_POLLING', ''))
|
||||
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper()
|
||||
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO')
|
||||
PATTERNS = ['*.pdf', '*.PDF']
|
||||
|
||||
log = logging.getLogger('ocrmypdf-watcher')
|
||||
@@ -117,7 +117,12 @@ class HandleObserverEvent(PatternMatchingEventHandler):
|
||||
|
||||
def main():
|
||||
ocrmypdf.configure_logging(
|
||||
verbosity=ocrmypdf.Verbosity.default, manage_root_logger=True
|
||||
verbosity=(
|
||||
ocrmypdf.Verbosity.default
|
||||
if LOGLEVEL != 'DEBUG'
|
||||
else ocrmypdf.Verbosity.debug
|
||||
),
|
||||
manage_root_logger=True,
|
||||
)
|
||||
log.setLevel(LOGLEVEL)
|
||||
log.info(
|
||||
@@ -135,7 +140,7 @@ def main():
|
||||
f"ARGS: {OCR_JSON_SETTINGS}\n"
|
||||
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
|
||||
f"USE_POLLING: {USE_POLLING}\n"
|
||||
f"LOGLEVEL: {LOGLEVEL}\n"
|
||||
f"LOGLEVEL: {LOGLEVEL}"
|
||||
)
|
||||
|
||||
if 'input_file' in OCR_JSON_SETTINGS or 'output_file' in OCR_JSON_SETTINGS:
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
[bdist_wheel]
|
||||
python-tag = py35
|
||||
python-tag = py36
|
||||
|
||||
[aliases]
|
||||
test=pytest
|
||||
|
||||
@@ -82,7 +82,7 @@ setup(
|
||||
],
|
||||
tests_require=tests_require,
|
||||
entry_points={'console_scripts': ['ocrmypdf = ocrmypdf.__main__:run']},
|
||||
package_data={'ocrmypdf': ['data/sRGB.icc']},
|
||||
package_data={'ocrmypdf': ['data/sRGB.icc', 'py.typed']},
|
||||
include_package_data=True,
|
||||
zip_safe=False,
|
||||
project_urls={
|
||||
|
||||
@@ -12,6 +12,7 @@ import os
|
||||
import signal
|
||||
import sys
|
||||
import threading
|
||||
from contextlib import suppress
|
||||
from multiprocessing import Pool as ProcessPool
|
||||
from multiprocessing.dummy import Pool as ThreadPool
|
||||
from typing import Callable, Iterable, Optional
|
||||
@@ -56,7 +57,7 @@ def process_init(queue, user_init, loglevel):
|
||||
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
||||
|
||||
# Install SIGBUS handler (so our parent process can abort somewhat gracefully)
|
||||
if hasattr(signal, 'SIGBUS'):
|
||||
with suppress(AttributeError): # Windows and Cygwin do not have SIGBUS
|
||||
signal.signal(signal.SIGBUS, process_sigbus)
|
||||
|
||||
# Reconfigure the root logger for this process to send all messages to a queue
|
||||
@@ -72,7 +73,8 @@ def process_init(queue, user_init, loglevel):
|
||||
|
||||
def thread_init(_queue, user_init, _loglevel):
|
||||
# As a thread, block SIGBUS so the main thread deals with it...
|
||||
if hasattr(signal, 'SIGBUS'):
|
||||
with suppress(AttributeError):
|
||||
# Windows and Cygwin do not have pthread_sigmask or SIGBUS
|
||||
signal.pthread_sigmask(signal.SIG_BLOCK, {signal.SIGBUS})
|
||||
if user_init:
|
||||
user_init()
|
||||
|
||||
@@ -21,7 +21,7 @@ from PIL import Image
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
||||
from ocrmypdf.helpers import Resolution
|
||||
from ocrmypdf.subprocess import get_version, run
|
||||
from ocrmypdf.subprocess import get_version, run, run_polling_stderr
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -139,12 +139,39 @@ def rasterize_pdf(
|
||||
im.save(fspath(output_file), dpi=page_dpi)
|
||||
|
||||
|
||||
class GhostscriptFollower:
|
||||
re_process = re.compile(r"Processing pages \d+ through (\d+).")
|
||||
re_page = re.compile(r"Page (\d+)")
|
||||
|
||||
def __init__(self, progressbar_class):
|
||||
self.count = 0
|
||||
self.progressbar_class = progressbar_class
|
||||
self.progressbar = None
|
||||
|
||||
def __call__(self, line):
|
||||
if not self.progressbar_class:
|
||||
return
|
||||
if not self.progressbar:
|
||||
m = self.re_process.match(line.strip())
|
||||
if m:
|
||||
self.count = int(m.group(1))
|
||||
self.progressbar = self.progressbar_class(
|
||||
total=self.count, desc="PDF/A conversion", unit='page'
|
||||
)
|
||||
return
|
||||
else:
|
||||
m = self.re_page.match(line.strip())
|
||||
if m:
|
||||
self.progressbar.update()
|
||||
|
||||
|
||||
def generate_pdfa(
|
||||
pdf_pages,
|
||||
output_file: os.PathLike,
|
||||
compression: str,
|
||||
pdf_version: str = '1.5',
|
||||
pdfa_part: str = '2',
|
||||
progressbar_class=None,
|
||||
):
|
||||
# Ghostscript's compression is all or nothing. We can either force all images
|
||||
# to JPEG, force all to Flate/PNG, or let it decide how to encode the images.
|
||||
@@ -188,7 +215,6 @@ def generate_pdfa(
|
||||
args_gs = (
|
||||
[
|
||||
GS,
|
||||
"-dQUIET",
|
||||
"-dBATCH",
|
||||
"-dNOPAUSE",
|
||||
"-dSAFER",
|
||||
@@ -208,16 +234,28 @@ def generate_pdfa(
|
||||
]
|
||||
)
|
||||
args_gs.extend(fspath(s) for s in pdf_pages) # Stringify Path objs
|
||||
|
||||
try:
|
||||
with Path(output_file).open('wb') as output:
|
||||
p = run(args_gs, stdout=output, stderr=PIPE, check=True)
|
||||
p = run_polling_stderr(
|
||||
args_gs,
|
||||
stdout=output,
|
||||
stderr=PIPE,
|
||||
check=True,
|
||||
text=True,
|
||||
encoding='utf-8',
|
||||
errors='replace',
|
||||
callback=GhostscriptFollower(progressbar_class),
|
||||
)
|
||||
except CalledProcessError as e:
|
||||
# Ghostscript does not change return code when it fails to create
|
||||
# PDF/A - check PDF/A status elsewhere
|
||||
log.error(e.stderr.decode(errors='replace'))
|
||||
raise SubprocessOutputError('Ghostscript PDF/A rendering failed')
|
||||
log.error(e.stderr)
|
||||
raise SubprocessOutputError('Ghostscript PDF/A rendering failed') from e
|
||||
else:
|
||||
stderr = p.stderr.decode('utf-8', errors='replace')
|
||||
stderr = p.stderr
|
||||
# If there is an error we log the whole stderr, except for filtering
|
||||
# duplicates.
|
||||
if _gs_error_reported(stderr):
|
||||
last_part = None
|
||||
repcount = 0
|
||||
@@ -230,11 +268,3 @@ def generate_pdfa(
|
||||
else:
|
||||
repcount += 1
|
||||
last_part = part
|
||||
elif 'overprint mode not set' in stderr:
|
||||
# Unless someone is going to print PDF/A documents on a
|
||||
# magical sRGB printer I can't see the removal of overprinting
|
||||
# being a problem....
|
||||
log.debug(
|
||||
"Ghostscript had to remove PDF 'overprinting' from the "
|
||||
"input file to complete PDF/A conversion. "
|
||||
)
|
||||
|
||||
@@ -99,7 +99,14 @@ def get_languages():
|
||||
|
||||
args_tess = ['tesseract', '--list-langs']
|
||||
try:
|
||||
proc = run(args_tess, text=True, stdout=PIPE, stderr=STDOUT, check=True)
|
||||
proc = run(
|
||||
args_tess,
|
||||
text=True,
|
||||
stdout=PIPE,
|
||||
stderr=STDOUT,
|
||||
logs_errors_to_stdout=True,
|
||||
check=True,
|
||||
)
|
||||
output = proc.stdout
|
||||
except CalledProcessError as e:
|
||||
raise MissingDependencyError(lang_error(e.output)) from e
|
||||
|
||||
+19
-16
@@ -19,6 +19,7 @@ import img2pdf
|
||||
import pikepdf
|
||||
from pikepdf.models.metadata import encode_pdf_date
|
||||
from PIL import Image, ImageColor, ImageDraw
|
||||
from tqdm import tqdm
|
||||
|
||||
from ocrmypdf import leptonica
|
||||
from ocrmypdf._exec import unpaper
|
||||
@@ -149,7 +150,7 @@ def get_pdfinfo(
|
||||
progbar=False,
|
||||
max_workers=None,
|
||||
check_pages=None,
|
||||
):
|
||||
) -> PdfInfo:
|
||||
try:
|
||||
return PdfInfo(
|
||||
input_file,
|
||||
@@ -601,14 +602,17 @@ def create_pdf_page_from_image(image: Path, page_context: PageContext):
|
||||
|
||||
|
||||
def render_hocr_page(hocr: Path, page_context: PageContext):
|
||||
options = page_context.options
|
||||
output_file = page_context.get_path('ocr_hocr.pdf')
|
||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||
dpi = get_page_square_dpi(page_context.pageinfo, options)
|
||||
debug_mode = options.pdf_renderer == 'hocrdebug'
|
||||
|
||||
hocrtransform = HocrTransform(hocr, dpi.x) # square
|
||||
hocrtransform.to_pdf(
|
||||
output_file,
|
||||
image_filename=None,
|
||||
show_bounding_boxes=False,
|
||||
invisible_text=True,
|
||||
show_bounding_boxes=False if not debug_mode else True,
|
||||
invisible_text=True if not debug_mode else False,
|
||||
interword_spaces=True,
|
||||
)
|
||||
return output_file
|
||||
@@ -709,6 +713,7 @@ def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext):
|
||||
output_file=output_file,
|
||||
compression=options.pdfa_image_compression,
|
||||
pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3
|
||||
progressbar_class=tqdm if options.progress_bar else None,
|
||||
)
|
||||
|
||||
return output_file
|
||||
@@ -751,19 +756,17 @@ def metadata_fixup(working_file: Path, context: PdfContext):
|
||||
if 'xmp:CreateDate' not in meta:
|
||||
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
||||
|
||||
# Ghostscript likes to set title to Untitled if omitted from input.
|
||||
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
||||
# and the XMP Spec do not make this recommendation.
|
||||
if meta.get('dc:title') == 'Untitled':
|
||||
with original.open_metadata(
|
||||
set_pikepdf_as_editor=False, update_docinfo=False
|
||||
) as original_meta:
|
||||
if 'dc:title' not in original_meta:
|
||||
with original.open_metadata(
|
||||
set_pikepdf_as_editor=False, update_docinfo=False, strict=False
|
||||
) as meta_original:
|
||||
if meta.get('dc:title') == 'Untitled':
|
||||
# Ghostscript likes to set title to Untitled if omitted from input.
|
||||
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
||||
# and the XMP Spec do not make this recommendation.
|
||||
if 'dc:title' not in meta_original:
|
||||
del meta['dc:title']
|
||||
|
||||
meta_original = original.open_metadata()
|
||||
missing = set(meta_original.keys()) - set(meta.keys())
|
||||
report_on_metadata(missing)
|
||||
missing = set(meta_original.keys()) - set(meta.keys())
|
||||
report_on_metadata(missing)
|
||||
|
||||
pdf.save(
|
||||
output_file,
|
||||
|
||||
@@ -207,7 +207,7 @@ def exec_page_sync(page_context: PageContext):
|
||||
visible_image_out, page_context
|
||||
)
|
||||
|
||||
if options.pdf_renderer == 'hocr':
|
||||
if options.pdf_renderer.startswith('hocr'):
|
||||
(hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context)
|
||||
ocr_out = render_hocr_page(hocr_out, page_context)
|
||||
elif options.pdf_renderer == 'sandwich':
|
||||
@@ -330,7 +330,7 @@ def run_pipeline(options, *, plugin_manager, api=False):
|
||||
if not plugin_manager:
|
||||
plugin_manager = get_plugin_manager(options.plugins)
|
||||
|
||||
work_folder = Path(mkdtemp(prefix="com.github.ocrmypdf."))
|
||||
work_folder = Path(mkdtemp(prefix="ocrmypdf.io."))
|
||||
debug_log_handler = None
|
||||
if (
|
||||
(options.keep_temporary_files or options.verbose >= 1)
|
||||
|
||||
@@ -13,7 +13,7 @@ import sys
|
||||
import unicodedata
|
||||
from pathlib import Path
|
||||
from shutil import copyfileobj
|
||||
from typing import Tuple
|
||||
from typing import List, Set, Tuple, Union
|
||||
|
||||
import pikepdf
|
||||
import PIL
|
||||
@@ -78,7 +78,7 @@ def check_options_languages(options, ocr_engine_languages):
|
||||
def check_options_output(options):
|
||||
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
||||
|
||||
if options.pdf_renderer == 'hocr' and not is_latin:
|
||||
if options.pdf_renderer.startswith('hocr') and not is_latin:
|
||||
msg = (
|
||||
"The 'hocr' PDF renderer is known to cause problems with one "
|
||||
"or more of the languages in your document. Use "
|
||||
@@ -136,10 +136,10 @@ def check_options_preprocessing(options):
|
||||
raise BadArgsError(str(e))
|
||||
|
||||
|
||||
def _pages_from_ranges(ranges):
|
||||
def _pages_from_ranges(ranges: str) -> Set[int]:
|
||||
if is_iterable_notstr(ranges):
|
||||
return set(ranges)
|
||||
pages = []
|
||||
pages: List[int] = []
|
||||
page_groups = ranges.replace(' ', '').split(',')
|
||||
for g in page_groups:
|
||||
if not g:
|
||||
@@ -150,9 +150,18 @@ def _pages_from_ranges(ranges):
|
||||
pages.append(int(g) - 1)
|
||||
else:
|
||||
try:
|
||||
pages.extend(range(int(start) - 1, int(end)))
|
||||
new_pages = list(range(int(start) - 1, int(end)))
|
||||
if not new_pages:
|
||||
raise BadArgsError(f"invalid page subrange '{start}-{end}'")
|
||||
pages.extend(new_pages)
|
||||
except ValueError:
|
||||
raise BadArgsError("invalid page range")
|
||||
raise BadArgsError("invalid page range") from None
|
||||
|
||||
if not pages:
|
||||
raise BadArgsError(
|
||||
f"The string of page ranges '{ranges}' did not contain any recognizable "
|
||||
f"page ranges."
|
||||
)
|
||||
|
||||
if not monotonic(pages):
|
||||
log.warning(
|
||||
|
||||
+7
-6
@@ -9,6 +9,7 @@ import logging
|
||||
import os
|
||||
import sys
|
||||
from enum import IntEnum
|
||||
from io import IOBase
|
||||
from pathlib import Path
|
||||
from typing import AnyStr, BinaryIO, Iterable, Optional, Union
|
||||
from warnings import warn
|
||||
@@ -174,14 +175,14 @@ def create_options(
|
||||
else:
|
||||
raise TypeError(f"{arg}: {val} ({type(val)})")
|
||||
|
||||
try:
|
||||
cmdline.append(os.fspath(input_file))
|
||||
except TypeError:
|
||||
if isinstance(input_file, (BinaryIO, IOBase)):
|
||||
cmdline.append('stream://input_file')
|
||||
try:
|
||||
cmdline.append(os.fspath(output_file))
|
||||
except TypeError:
|
||||
else:
|
||||
cmdline.append(os.fspath(input_file))
|
||||
if isinstance(output_file, (BinaryIO, IOBase)):
|
||||
cmdline.append('stream://output_file')
|
||||
else:
|
||||
cmdline.append(os.fspath(output_file))
|
||||
|
||||
parser._api_mode = True
|
||||
options = parser.parse_args(cmdline)
|
||||
|
||||
@@ -79,12 +79,21 @@ def rasterize_pdf_page(
|
||||
|
||||
|
||||
@hookimpl
|
||||
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
||||
def generate_pdfa(
|
||||
pdf_pages,
|
||||
pdfmark,
|
||||
output_file,
|
||||
compression,
|
||||
pdf_version,
|
||||
pdfa_part,
|
||||
progressbar_class,
|
||||
):
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=[*pdf_pages, pdfmark],
|
||||
output_file=output_file,
|
||||
compression=compression,
|
||||
pdf_version=pdf_version,
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=progressbar_class,
|
||||
)
|
||||
return output_file
|
||||
|
||||
+5
-2
@@ -6,12 +6,15 @@
|
||||
|
||||
|
||||
import argparse
|
||||
from typing import Optional, Type, TypeVar
|
||||
|
||||
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
||||
from ocrmypdf._version import __version__ as _VERSION
|
||||
|
||||
T = TypeVar('T')
|
||||
|
||||
def numeric(basetype, min_=None, max_=None):
|
||||
|
||||
def numeric(basetype: Type[T], min_: Optional[T] = None, max_: Optional[T] = None):
|
||||
"""Validator for numeric params"""
|
||||
min_ = basetype(min_) if min_ is not None else None
|
||||
max_ = basetype(max_) if max_ is not None else None
|
||||
@@ -407,7 +410,7 @@ Online documentation is located at:
|
||||
)
|
||||
advanced.add_argument(
|
||||
'--pdf-renderer',
|
||||
choices=['auto', 'hocr', 'sandwich'],
|
||||
choices=['auto', 'hocr', 'sandwich', 'hocrdebug'],
|
||||
default='auto',
|
||||
help="Choose OCR PDF renderer - the default option is to let OCRmyPDF "
|
||||
"choose. See documentation for discussion.",
|
||||
|
||||
@@ -216,10 +216,7 @@ def check_pdf(input_file: Path) -> bool:
|
||||
pdf.close()
|
||||
|
||||
|
||||
T = TypeVar('T')
|
||||
|
||||
|
||||
def clamp(n: T, smallest: T, largest: T) -> T:
|
||||
def clamp(n, smallest, largest): # mypy doesn't understand types for this
|
||||
"""Clamps the value of n to between smallest and largest."""
|
||||
return max(smallest, min(n, largest))
|
||||
|
||||
@@ -232,6 +229,7 @@ def pikepdf_enable_mmap():
|
||||
# log.debug("pikepdf mmap not available")
|
||||
# We found a race condition probably related to pybind issue #2252 that can
|
||||
# cause a crash. For now, disable pikepdf mmap to be on the safe side.
|
||||
# Fix is not in pybind11 2.6.0
|
||||
log.debug("pikepdf mmap disabled")
|
||||
return
|
||||
|
||||
|
||||
@@ -42,6 +42,8 @@ from reportlab.lib.colors import black, cyan, magenta, red
|
||||
from reportlab.lib.units import inch
|
||||
from reportlab.pdfgen.canvas import Canvas
|
||||
|
||||
Element = ElementTree.Element
|
||||
|
||||
Rect = namedtuple('Rect', ['x1', 'y1', 'x2', 'y2'])
|
||||
|
||||
|
||||
@@ -105,7 +107,7 @@ class HocrTransform:
|
||||
else:
|
||||
return ''
|
||||
|
||||
def _get_element_text(self, element):
|
||||
def _get_element_text(self, element: Element):
|
||||
"""
|
||||
Return the textual content of the element and its children
|
||||
"""
|
||||
@@ -119,7 +121,7 @@ class HocrTransform:
|
||||
return text
|
||||
|
||||
@classmethod
|
||||
def element_coordinates(cls, element) -> Rect:
|
||||
def element_coordinates(cls, element: Element) -> Rect:
|
||||
"""
|
||||
Returns a tuple containing the coordinates of the bounding box around
|
||||
an element
|
||||
@@ -133,7 +135,7 @@ class HocrTransform:
|
||||
return out
|
||||
|
||||
@classmethod
|
||||
def baseline(cls, element) -> Tuple[float, float]:
|
||||
def baseline(cls, element: Element) -> Tuple[float, float]:
|
||||
"""
|
||||
Returns a tuple containing the baseline slope and intercept.
|
||||
"""
|
||||
@@ -149,7 +151,7 @@ class HocrTransform:
|
||||
"""
|
||||
return Rect._make((c / self.dpi * inch) for c in pxl)
|
||||
|
||||
def _child_xpath(self, html_tag, html_class=None):
|
||||
def _child_xpath(self, html_tag: str, html_class: Optional[str] = None) -> str:
|
||||
xpath = f".//{self.xmlns}{html_tag}"
|
||||
if html_class:
|
||||
xpath += f"[@class='{html_class}']"
|
||||
@@ -280,13 +282,15 @@ class HocrTransform:
|
||||
def _do_line(
|
||||
self,
|
||||
pdf: Canvas,
|
||||
line,
|
||||
line: Optional[Element],
|
||||
elemclass: str,
|
||||
fontname: str,
|
||||
invisible_text: bool,
|
||||
interword_spaces: bool,
|
||||
show_bounding_boxes: bool,
|
||||
):
|
||||
if not line:
|
||||
return
|
||||
pxl_line_coords = self.element_coordinates(line)
|
||||
line_box = self.pt_from_pixel(pxl_line_coords)
|
||||
line_height = line_box.y2 - line_box.y1
|
||||
|
||||
@@ -27,15 +27,16 @@ from tempfile import TemporaryFile
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
from ocrmypdf.lib._leptonica import ffi
|
||||
from ocrmypdf.subprocess import shim_paths_with_program_files
|
||||
|
||||
# pylint: disable=protected-access
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
if os.name == 'nt':
|
||||
from ocrmypdf.subprocess._windows import shim_env_path
|
||||
|
||||
libname = 'liblept-5'
|
||||
os.environ['PATH'] = shim_paths_with_program_files()
|
||||
os.environ['PATH'] = shim_env_path()
|
||||
else:
|
||||
libname = 'lept'
|
||||
_libpath = find_library(libname)
|
||||
@@ -58,9 +59,9 @@ if not _libpath:
|
||||
---------------------------------------------------------------------
|
||||
"""
|
||||
)
|
||||
if os.name == 'nt':
|
||||
# On Windows, recent versions of libpng require zlib. We have to make sure
|
||||
# the zlib version being loaded is the same one that libpng was built with.
|
||||
if os.name == 'nt':
|
||||
# On Windows, recent versions of libpng require zlib. We have to make sure
|
||||
# the zlib version being loaded is the same one that libpng was built with.
|
||||
# This tries to import zlib from Tesseract's installation folder, falling back
|
||||
# to find_library() if liblept is being loaded from somewhere else.
|
||||
# Loading zlib from other places could cause a version mismatch
|
||||
|
||||
@@ -25,7 +25,6 @@ from typing import (
|
||||
Optional,
|
||||
Sequence,
|
||||
Tuple,
|
||||
Union,
|
||||
)
|
||||
|
||||
import img2pdf
|
||||
@@ -294,10 +293,6 @@ def extract_images_jbig2(pike: Pdf, root: Path, options) -> Dict[int, List[XrefE
|
||||
group = pageno // options.jbig2_page_group_size
|
||||
jbig2_groups[group].append(xref_ext)
|
||||
|
||||
# Elide empty groups
|
||||
jbig2_groups = {
|
||||
group: xrefs for group, xrefs in jbig2_groups.items() if len(xrefs) > 0
|
||||
}
|
||||
log.debug("Optimizable images: JBIG2 groups: %s", (len(jbig2_groups),))
|
||||
return jbig2_groups
|
||||
|
||||
|
||||
@@ -33,6 +33,7 @@ def _postscript_objdef(
|
||||
objtype = '/stream' if stream_name else '/dict'
|
||||
|
||||
if stream_name:
|
||||
assert stream_data is not None
|
||||
a85_data = base64.a85encode(stream_data, adobe=True).decode('ascii')
|
||||
yield f'{stream_name} ' + a85_data
|
||||
yield 'def'
|
||||
|
||||
+124
-105
@@ -15,11 +15,11 @@ from functools import partial
|
||||
from math import hypot, isclose
|
||||
from os import PathLike
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional, Union
|
||||
from typing import Any, Container, Dict, Iterator, List, Optional, Tuple, Union
|
||||
from warnings import warn
|
||||
|
||||
import pikepdf
|
||||
from pikepdf import PdfMatrix
|
||||
from pikepdf import Object, Pdf, PdfMatrix
|
||||
|
||||
from ocrmypdf._concurrent import exec_progress_pool
|
||||
from ocrmypdf.exceptions import EncryptedPdfError
|
||||
@@ -115,7 +115,7 @@ def _normalize_stack(graphobjs):
|
||||
yield (operands, operator)
|
||||
|
||||
|
||||
def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
||||
def _interpret_contents(contentstream: Object, initial_shorthand=UNIT_SQUARE):
|
||||
"""Interpret the PDF content stream.
|
||||
|
||||
The stack represents the state of the PDF graphics stack. We are only
|
||||
@@ -204,7 +204,7 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
||||
)
|
||||
|
||||
|
||||
def _get_dpi(ctm_shorthand, image_size):
|
||||
def _get_dpi(ctm_shorthand, image_size) -> Resolution:
|
||||
"""Given the transformation matrix and image size, find the image DPI.
|
||||
|
||||
PDFs do not include image resolution information within image data.
|
||||
@@ -271,8 +271,14 @@ def _get_dpi(ctm_shorthand, image_size):
|
||||
class ImageInfo:
|
||||
DPI_PREC = Decimal('1.000')
|
||||
|
||||
def __init__(self, *, name='', pdfimage=None, inline=None, shorthand=None):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
name='',
|
||||
pdfimage: Optional[Object] = None,
|
||||
inline: Optional[Object] = None,
|
||||
shorthand=None,
|
||||
):
|
||||
self._name = str(name)
|
||||
self._shorthand = shorthand
|
||||
|
||||
@@ -282,6 +288,8 @@ class ImageInfo:
|
||||
elif pdfimage is not None:
|
||||
self._origin = 'xobject'
|
||||
pim = pikepdf.PdfImage(pdfimage)
|
||||
else:
|
||||
raise ValueError("Either pdfimage or inline must be set")
|
||||
self._width = pim.width
|
||||
self._height = pim.height
|
||||
|
||||
@@ -371,7 +379,7 @@ class ImageInfo:
|
||||
).format(**class_locals)
|
||||
|
||||
|
||||
def _find_inline_images(contentsinfo):
|
||||
def _find_inline_images(contentsinfo: ContentsInfo) -> Iterator[ImageInfo]:
|
||||
"Find inline images in the contentstream"
|
||||
|
||||
for n, inline in enumerate(contentsinfo.inline_images):
|
||||
@@ -380,7 +388,7 @@ def _find_inline_images(contentsinfo):
|
||||
)
|
||||
|
||||
|
||||
def _image_xobjects(container):
|
||||
def _image_xobjects(container) -> Iterator[Tuple[Object, str]]:
|
||||
"""Search for all XObject-based images in the container
|
||||
|
||||
Usually the container is a page, but it could also be a Form XObject
|
||||
@@ -400,7 +408,7 @@ def _image_xobjects(container):
|
||||
return
|
||||
xobjs = resources['/XObject'].as_dict()
|
||||
for xobj in xobjs:
|
||||
candidate = xobjs[xobj]
|
||||
candidate: Object = xobjs[xobj]
|
||||
if not '/Subtype' in candidate:
|
||||
continue
|
||||
if candidate['/Subtype'] == '/Image':
|
||||
@@ -408,7 +416,9 @@ def _image_xobjects(container):
|
||||
yield (pdfimage, xobj)
|
||||
|
||||
|
||||
def _find_regular_images(container, contentsinfo):
|
||||
def _find_regular_images(
|
||||
container: Object, contentsinfo: ContentsInfo
|
||||
) -> Iterator[ImageInfo]:
|
||||
"""Find images stored in the container's /Resources /XObject
|
||||
|
||||
Usually the container is a page, but it could also be a Form XObject
|
||||
@@ -432,7 +442,7 @@ def _find_regular_images(container, contentsinfo):
|
||||
yield ImageInfo(name=draw.name, pdfimage=pdfimage, shorthand=draw.shorthand)
|
||||
|
||||
|
||||
def _find_form_xobject_images(pdf, container, contentsinfo):
|
||||
def _find_form_xobject_images(pdf: Pdf, container: Object, contentsinfo: ContentsInfo):
|
||||
"""Find any images that are in Form XObjects in the container
|
||||
|
||||
The container may be a page, or a parent Form XObject.
|
||||
@@ -464,7 +474,9 @@ def _find_form_xobject_images(pdf, container, contentsinfo):
|
||||
)
|
||||
|
||||
|
||||
def _process_content_streams(*, pdf, container, shorthand=None):
|
||||
def _process_content_streams(
|
||||
*, pdf: Pdf, container: Object, shorthand=None
|
||||
) -> Iterator[Union[VectorMarker, TextMarker, ImageInfo]]:
|
||||
"""Find all individual instances of images drawn in the container
|
||||
|
||||
Usually the container is a page, but it may also be a Form XObject.
|
||||
@@ -526,7 +538,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
|
||||
margin_ratio * ph, # bottom (first quadrant: bottom < top)
|
||||
)
|
||||
|
||||
def rects_intersect(a, b):
|
||||
def rects_intersect(a, b) -> bool:
|
||||
"""
|
||||
Where (a,b) are 4-tuple rects (left-0, top-1, right-2, bottom-3)
|
||||
https://stackoverflow.com/questions/306316/determine-if-two-rectangles-overlap-each-other
|
||||
@@ -542,7 +554,7 @@ def _page_has_text(text_blocks, page_width, page_height) -> bool:
|
||||
return has_text
|
||||
|
||||
|
||||
def simplify_textboxes(miner, textbox_getter):
|
||||
def simplify_textboxes(miner, textbox_getter) -> Iterator[TextboxInfo]:
|
||||
"""Extract only limited content from text boxes
|
||||
|
||||
We do this to save memory and ensure that our objects are pickleable.
|
||||
@@ -556,80 +568,15 @@ def simplify_textboxes(miner, textbox_getter):
|
||||
yield TextboxInfo(box.bbox, visible, corrupt)
|
||||
|
||||
|
||||
def _pdf_get_pageinfo(
|
||||
pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis: bool
|
||||
):
|
||||
pageinfo: Dict[str, Any] = {}
|
||||
pageinfo['pageno'] = pageno
|
||||
pageinfo['images'] = []
|
||||
|
||||
page = pdf.pages[pageno]
|
||||
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
|
||||
width_pt = mediabox[2] - mediabox[0]
|
||||
height_pt = mediabox[3] - mediabox[1]
|
||||
|
||||
check_this_page = pageno in check_pages
|
||||
|
||||
if check_this_page and detailed_analysis:
|
||||
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
|
||||
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
||||
pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes))
|
||||
bboxes = (box.bbox for box in pageinfo['textboxes'])
|
||||
|
||||
pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt)
|
||||
else:
|
||||
pageinfo['textboxes'] = []
|
||||
pageinfo['has_text'] = None # i.e. "no information"
|
||||
|
||||
userunit = page.get('/UserUnit', Decimal(1.0))
|
||||
if not isinstance(userunit, Decimal):
|
||||
userunit = Decimal(userunit)
|
||||
pageinfo['userunit'] = userunit
|
||||
pageinfo['width_inches'] = width_pt * userunit / Decimal(72.0)
|
||||
pageinfo['height_inches'] = height_pt * userunit / Decimal(72.0)
|
||||
|
||||
try:
|
||||
pageinfo['rotate'] = int(page['/Rotate'])
|
||||
except KeyError:
|
||||
pageinfo['rotate'] = 0
|
||||
|
||||
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
|
||||
|
||||
if check_this_page:
|
||||
pageinfo['has_vector'] = False
|
||||
pageinfo['has_text'] = False
|
||||
pageinfo['images'] = []
|
||||
for ci in _process_content_streams(
|
||||
pdf=pdf, container=page, shorthand=userunit_shorthand
|
||||
):
|
||||
if isinstance(ci, VectorMarker):
|
||||
pageinfo['has_vector'] = True
|
||||
elif isinstance(ci, TextMarker):
|
||||
pageinfo['has_text'] = True
|
||||
elif isinstance(ci, ImageInfo):
|
||||
pageinfo['images'].append(ci)
|
||||
else:
|
||||
raise NotImplementedError()
|
||||
else:
|
||||
pageinfo['has_vector'] = None # i.e. "no information"
|
||||
pageinfo['has_text'] = None
|
||||
pageinfo['images'] = None
|
||||
|
||||
if pageinfo['images']:
|
||||
dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo['images'])
|
||||
pageinfo['dpi'] = dpi
|
||||
pageinfo['width_pixels'] = int(round(dpi.x * float(pageinfo['width_inches'])))
|
||||
pageinfo['height_pixels'] = int(round(dpi.y * float(pageinfo['height_inches'])))
|
||||
|
||||
return pageinfo
|
||||
|
||||
|
||||
worker_pdf = None
|
||||
|
||||
|
||||
def _pdf_pageinfo_sync_init(infile):
|
||||
def _pdf_pageinfo_sync_init(infile: Path, pdfminer_loglevel):
|
||||
global worker_pdf # pylint: disable=global-statement
|
||||
pikepdf_enable_mmap()
|
||||
|
||||
logging.getLogger('pdfminer').setLevel(pdfminer_loglevel)
|
||||
|
||||
# If this function is called as a thread initializer, we need a messy hack
|
||||
# to close worker_pdf. If called as a process, it will be released when the
|
||||
# process is terminated.
|
||||
@@ -674,7 +621,9 @@ def _pdf_pageinfo_concurrent(
|
||||
tqdm_kwargs=dict(
|
||||
total=total, desc="Scanning contents", unit='page', disable=not progbar
|
||||
),
|
||||
task_initializer=partial(_pdf_pageinfo_sync_init, infile),
|
||||
task_initializer=partial(
|
||||
_pdf_pageinfo_sync_init, infile, logging.getLogger('pdfminer').level
|
||||
),
|
||||
task=_pdf_pageinfo_sync,
|
||||
task_arguments=contexts,
|
||||
task_finished=update_pageinfo,
|
||||
@@ -688,13 +637,85 @@ def _pdf_pageinfo_concurrent(
|
||||
|
||||
|
||||
class PageInfo:
|
||||
def __init__(self, pdf, pageno, infile, check_pages, detailed_analysis=False):
|
||||
def __init__(
|
||||
self,
|
||||
pdf: Pdf,
|
||||
pageno: int,
|
||||
infile: PathLike,
|
||||
check_pages: Container[int],
|
||||
detailed_analysis: bool = False,
|
||||
):
|
||||
self._pageno = pageno
|
||||
self._infile = infile
|
||||
self._detailed_analysis = detailed_analysis
|
||||
self._pageinfo = _pdf_get_pageinfo(
|
||||
pdf, pageno, infile, check_pages, detailed_analysis
|
||||
)
|
||||
self._gather_pageinfo(pdf, pageno, infile, check_pages, detailed_analysis)
|
||||
|
||||
def _gather_pageinfo(
|
||||
self,
|
||||
pdf: Pdf,
|
||||
pageno: int,
|
||||
infile: PathLike,
|
||||
check_pages: Container[int],
|
||||
detailed_analysis: bool,
|
||||
):
|
||||
page = pdf.pages[pageno]
|
||||
mediabox = [Decimal(d) for d in page.MediaBox.as_list()]
|
||||
width_pt = mediabox[2] - mediabox[0]
|
||||
height_pt = mediabox[3] - mediabox[1]
|
||||
|
||||
check_this_page = pageno in check_pages
|
||||
|
||||
if check_this_page and detailed_analysis:
|
||||
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
|
||||
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
||||
self._textboxes = list(simplify_textboxes(miner, get_text_boxes))
|
||||
bboxes = (box.bbox for box in self._textboxes)
|
||||
|
||||
self._has_text = _page_has_text(bboxes, width_pt, height_pt)
|
||||
else:
|
||||
self._textboxes = []
|
||||
self._has_text = None # i.e. "no information"
|
||||
|
||||
userunit = page.get('/UserUnit', Decimal(1.0))
|
||||
if not isinstance(userunit, Decimal):
|
||||
userunit = Decimal(userunit)
|
||||
self._userunit = userunit
|
||||
self._width_inches = width_pt * userunit / Decimal(72.0)
|
||||
self._height_inches = height_pt * userunit / Decimal(72.0)
|
||||
|
||||
try:
|
||||
self._rotate = int(page['/Rotate'])
|
||||
except KeyError:
|
||||
self._rotate = 0
|
||||
|
||||
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
|
||||
|
||||
if check_this_page:
|
||||
self._has_vector = False
|
||||
self._has_text = False
|
||||
self._images = []
|
||||
for ci in _process_content_streams(
|
||||
pdf=pdf, container=page, shorthand=userunit_shorthand
|
||||
):
|
||||
if isinstance(ci, VectorMarker):
|
||||
self._has_vector = True
|
||||
elif isinstance(ci, TextMarker):
|
||||
self._has_text = True
|
||||
elif isinstance(ci, ImageInfo):
|
||||
self._images.append(ci)
|
||||
else:
|
||||
raise NotImplementedError()
|
||||
else:
|
||||
self._has_vector = None # i.e. "no information"
|
||||
self._has_text = None
|
||||
self._images = None
|
||||
|
||||
self._dpi = None
|
||||
if self._images:
|
||||
dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in self._images)
|
||||
self._dpi = dpi
|
||||
self._width_pixels = int(round(dpi.x * float(self._width_inches)))
|
||||
self._height_pixels = int(round(dpi.y * float(self._height_inches)))
|
||||
|
||||
@property
|
||||
def pageno(self) -> int:
|
||||
@@ -702,25 +723,25 @@ class PageInfo:
|
||||
|
||||
@property
|
||||
def has_text(self) -> bool:
|
||||
return self._pageinfo['has_text']
|
||||
return self._has_text
|
||||
|
||||
@property
|
||||
def has_corrupt_text(self) -> bool:
|
||||
if not self._detailed_analysis:
|
||||
raise NotImplementedError('Did not do detailed analysis')
|
||||
return any(tbox.is_corrupt for tbox in self._pageinfo['textboxes'])
|
||||
return any(tbox.is_corrupt for tbox in self._textboxes)
|
||||
|
||||
@property
|
||||
def has_vector(self) -> bool:
|
||||
return self._pageinfo['has_vector']
|
||||
return self._has_vector
|
||||
|
||||
@property
|
||||
def width_inches(self) -> Decimal:
|
||||
return self._pageinfo['width_inches']
|
||||
return self._width_inches
|
||||
|
||||
@property
|
||||
def height_inches(self) -> Decimal:
|
||||
return self._pageinfo['height_inches']
|
||||
return self._height_inches
|
||||
|
||||
@property
|
||||
def width_pixels(self) -> int:
|
||||
@@ -732,18 +753,18 @@ class PageInfo:
|
||||
|
||||
@property
|
||||
def rotation(self) -> int:
|
||||
return self._pageinfo.get('rotate', None)
|
||||
return self._rotate
|
||||
|
||||
@rotation.setter
|
||||
def rotation(self, value):
|
||||
if value in (0, 90, 180, 270, 360, -90, -180, -270):
|
||||
self._pageinfo['rotate'] = value
|
||||
self._rotate = value
|
||||
else:
|
||||
raise ValueError("rotation must be a cardinal angle")
|
||||
|
||||
@property
|
||||
def images(self):
|
||||
return self._pageinfo['images']
|
||||
return self._images
|
||||
|
||||
def get_textareas(
|
||||
self, visible: Optional[bool] = None, corrupt: Optional[bool] = None
|
||||
@@ -758,24 +779,22 @@ class PageInfo:
|
||||
result = False
|
||||
return result
|
||||
|
||||
if 'textboxes' not in self._pageinfo:
|
||||
if not self._textboxes:
|
||||
if visible is not None and corrupt is not None:
|
||||
raise NotImplementedError('Incomplete information on textboxes')
|
||||
return self._pageinfo['bboxes']
|
||||
return self._textboxes
|
||||
|
||||
return (
|
||||
obj.bbox
|
||||
for obj in self._pageinfo['textboxes']
|
||||
if predicate(obj, visible, corrupt)
|
||||
)
|
||||
return (obj.bbox for obj in self._textboxes if predicate(obj, visible, corrupt))
|
||||
|
||||
@property
|
||||
def dpi(self) -> Resolution:
|
||||
return self._pageinfo.get('dpi', Resolution(0.0, 0.0))
|
||||
if self._dpi is None:
|
||||
return Resolution(0.0, 0.0)
|
||||
return self._dpi
|
||||
|
||||
@property
|
||||
def userunit(self) -> Decimal:
|
||||
return self._pageinfo.get('userunit', None)
|
||||
return self._userunit
|
||||
|
||||
@property
|
||||
def min_version(self) -> str:
|
||||
|
||||
@@ -223,6 +223,7 @@ def get_page_analysis(infile, pageno, pscript5_mode):
|
||||
)
|
||||
interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev)
|
||||
|
||||
patcher = None
|
||||
if pscript5_mode:
|
||||
patcher = patch.multiple(
|
||||
'pdfminer.pdffont.PDFType3Font',
|
||||
@@ -237,10 +238,10 @@ def get_page_analysis(infile, pageno, pscript5_mode):
|
||||
with Path(infile).open('rb') as f:
|
||||
page = PDFPage.get_pages(f, pagenos=[pageno], maxpages=0)
|
||||
interp.process_page(next(page))
|
||||
except PDFTextExtractionNotAllowed:
|
||||
raise EncryptedPdfError()
|
||||
except PDFTextExtractionNotAllowed as e:
|
||||
raise EncryptedPdfError() from e
|
||||
finally:
|
||||
if pscript5_mode:
|
||||
if patcher is not None:
|
||||
patcher.stop()
|
||||
|
||||
return dev.get_result()
|
||||
|
||||
@@ -59,7 +59,7 @@ def check_options(options: Namespace) -> None:
|
||||
Note:
|
||||
This hook will be called from the main process, and may modify global state
|
||||
before child worker processes are forked.
|
||||
"""
|
||||
"""
|
||||
|
||||
|
||||
@hookspec
|
||||
@@ -280,6 +280,7 @@ def generate_pdfa(
|
||||
compression: str,
|
||||
pdf_version: str,
|
||||
pdfa_part: str,
|
||||
progressbar_class,
|
||||
) -> Path:
|
||||
"""Generate a PDF/A.
|
||||
|
||||
@@ -302,10 +303,21 @@ def generate_pdfa(
|
||||
At its own discretion, the PDF/A generator may raise the version,
|
||||
but should not lower it.
|
||||
pdfa_part: The desired PDF/A compliance level, such as ``'2B'``.
|
||||
progressbar_class: The class of a progress bar with a tqdm-like API. An
|
||||
instance of this class will be initialized when PDF/A conversion
|
||||
begins, using
|
||||
``instance = progressbar_class(total: int, desc: str, unit:str)``,
|
||||
defining the number of work units, a user-visible description,
|
||||
and the name of the work units ("page"). Then ``instance.update()``
|
||||
will be called when a work unit is completed. If ``None``, no
|
||||
progress information is reported.
|
||||
|
||||
Returns:
|
||||
Path: If successful, the hook should return ``output_file``.
|
||||
|
||||
Note:
|
||||
This is a :ref:`firstresult hook<firstresult>`.
|
||||
|
||||
See also:
|
||||
https://github.com/tqdm/tqdm
|
||||
"""
|
||||
|
||||
@@ -0,0 +1 @@
|
||||
# ocrmypdf is typed
|
||||
@@ -10,14 +10,13 @@
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
from collections.abc import Mapping
|
||||
from contextlib import suppress
|
||||
from distutils.version import LooseVersion
|
||||
from functools import lru_cache
|
||||
from pathlib import Path
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
||||
from subprocess import PIPE, STDOUT, CalledProcessError, CompletedProcess, Popen
|
||||
from subprocess import run as subprocess_run
|
||||
|
||||
from ocrmypdf.exceptions import MissingDependencyError
|
||||
@@ -41,26 +40,7 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs):
|
||||
if there is an error. If False, stderr is logged. Could be used with
|
||||
stderr=STDOUT, stdout=PIPE for example.
|
||||
"""
|
||||
if not env:
|
||||
env = os.environ
|
||||
|
||||
# Search in spoof path if necessary
|
||||
program = args[0]
|
||||
|
||||
if os.name == 'nt':
|
||||
args = _fix_windows_args(program, args, env)
|
||||
|
||||
log.debug("Running: %s", args)
|
||||
process_log = log.getChild(os.path.basename(program))
|
||||
if sys.version_info < (3, 7):
|
||||
if os.name == 'nt':
|
||||
# Can't use close_fds=True on Windows with Python 3.6 or older
|
||||
# https://bugs.python.org/issue19575, etc.
|
||||
kwargs['close_fds'] = False
|
||||
if 'text' in kwargs:
|
||||
# Convert run(...text=) to run(...universal_newlines=) for Python 3.6
|
||||
kwargs['universal_newlines'] = kwargs['text']
|
||||
del kwargs['text']
|
||||
args, env, process_log, _text = _fix_process_args(args, env, kwargs)
|
||||
|
||||
stderr = None
|
||||
stderr_name = 'stderr' if not logs_errors_to_stdout else 'stdout'
|
||||
@@ -82,28 +62,64 @@ def run(args, *, env=None, logs_errors_to_stdout=False, **kwargs):
|
||||
return proc
|
||||
|
||||
|
||||
def _fix_windows_args(program, args, env):
|
||||
"""Adjust our desired program and command line arguments for use on Windows"""
|
||||
def run_polling_stderr(args, *, callback, check=False, env=None, **kwargs):
|
||||
"""Run a process like ``ocrmypdf.subprocess.run``, and poll stderr.
|
||||
|
||||
if sys.version_info < (3, 8):
|
||||
# bpo-33617 - Windows needs manual Path -> str conversion
|
||||
args = [os.fspath(arg) for arg in args]
|
||||
program = os.fspath(program)
|
||||
Every line of produced by stderr will be forwarded to the callback function.
|
||||
The intended use is monitoring progress of subprocesses that output their
|
||||
own progress indicators. In addition, each line will be logged if debug
|
||||
logging is enabled.
|
||||
|
||||
# If we are running a .py on Windows, ensure we call it with this Python
|
||||
# (to support test suite shims)
|
||||
if program.lower().endswith('.py'):
|
||||
args = [sys.executable] + args
|
||||
Requires stderr to be opened in text mode for ease of handling errors. In
|
||||
addition the expected encoding= and errors= arguments should be set. Note
|
||||
that if stdout is already set up, it need not be binary.
|
||||
"""
|
||||
args, env, process_log, text = _fix_process_args(args, env, kwargs)
|
||||
assert text, "Must use text=True"
|
||||
|
||||
paths = os.pathsep.join(os.get_exec_path(env))
|
||||
if not shutil.which(args[0], path=paths):
|
||||
# If the program we want is not on the PATH, add some interesting
|
||||
# locations in %PROGRAMFILES% to the PATH and try again
|
||||
shimmed_path = shim_paths_with_program_files(env)
|
||||
new_args0 = shutil.which(args[0], path=shimmed_path)
|
||||
if new_args0:
|
||||
args[0] = new_args0
|
||||
return args
|
||||
proc = Popen(args, env=env, **kwargs)
|
||||
|
||||
lines = []
|
||||
while proc.poll() is None:
|
||||
for msg in iter(proc.stderr.readline, ''):
|
||||
if process_log.isEnabledFor(logging.DEBUG):
|
||||
process_log.debug(msg.strip())
|
||||
callback(msg)
|
||||
lines.append(msg)
|
||||
stderr = ''.join(lines)
|
||||
|
||||
if check and proc.returncode != 0:
|
||||
raise CalledProcessError(proc.returncode, args, output=None, stderr=stderr)
|
||||
return CompletedProcess(args, proc.returncode, None, stderr=stderr)
|
||||
|
||||
|
||||
def _fix_process_args(args, env, kwargs):
|
||||
assert 'universal_newlines' not in kwargs, "Use text= instead of universal_newlines"
|
||||
|
||||
if not env:
|
||||
env = os.environ
|
||||
|
||||
# Search in spoof path if necessary
|
||||
program = args[0]
|
||||
|
||||
if os.name == 'nt':
|
||||
from ocrmypdf.subprocess._windows import fix_windows_args
|
||||
|
||||
args = fix_windows_args(program, args, env)
|
||||
|
||||
log.debug("Running: %s", args)
|
||||
process_log = log.getChild(os.path.basename(program))
|
||||
text = kwargs.get('text', False)
|
||||
if sys.version_info < (3, 7):
|
||||
if os.name == 'nt':
|
||||
# Can't use close_fds=True on Windows with Python 3.6 or older
|
||||
# https://bugs.python.org/issue19575, etc.
|
||||
kwargs['close_fds'] = False
|
||||
if 'text' in kwargs:
|
||||
# Convert run(...text=) to run(...universal_newlines=) for Python 3.6
|
||||
kwargs['universal_newlines'] = kwargs['text']
|
||||
del kwargs['text']
|
||||
return args, env, process_log, text
|
||||
|
||||
|
||||
@lru_cache(maxsize=None)
|
||||
@@ -143,44 +159,18 @@ def get_version(
|
||||
raise MissingDependencyError(
|
||||
f"Could not find program '{program}' on the PATH"
|
||||
) from e
|
||||
try:
|
||||
version = re.match(regex, output.strip()).group(1)
|
||||
except AttributeError as e:
|
||||
|
||||
match = re.match(regex, output.strip())
|
||||
if not match:
|
||||
raise MissingDependencyError(
|
||||
f"The program '{program}' did not report its version. "
|
||||
f"Message was:\n{output}"
|
||||
)
|
||||
version = match.group(1)
|
||||
|
||||
return version
|
||||
|
||||
|
||||
def shim_paths_with_program_files(env=None):
|
||||
if not env:
|
||||
env = os.environ
|
||||
program_files = env.get('PROGRAMFILES', '')
|
||||
if not program_files:
|
||||
return env.get('PATH', '')
|
||||
|
||||
def path_walker():
|
||||
for path in Path(program_files).iterdir():
|
||||
if not path.is_dir():
|
||||
continue
|
||||
if path.name.lower() == 'tesseract-ocr':
|
||||
yield path
|
||||
elif path.name.lower() == 'gs':
|
||||
yield from (p for p in path.glob('**/bin') if p.is_dir())
|
||||
|
||||
paths = sorted(
|
||||
(p for p in path_walker()), key=lambda p: (p.name, p.parent.name), reverse=True
|
||||
)
|
||||
paths.extend(
|
||||
Path(str_path)
|
||||
for str_path in os.get_exec_path(env)
|
||||
if Path(str_path) not in set(paths)
|
||||
)
|
||||
return os.pathsep.join(str(p) for p in paths)
|
||||
|
||||
|
||||
missing_program = '''
|
||||
The program '{program}' could not be executed or was not found on your
|
||||
system PATH.
|
||||
@@ -0,0 +1,162 @@
|
||||
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||
#
|
||||
# This Source Code Form is subject to the terms of the Mozilla Public
|
||||
# License, v. 2.0. If a copy of the MPL was not distributed with this
|
||||
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
from distutils.version import LooseVersion
|
||||
from itertools import chain, filterfalse
|
||||
from pathlib import Path
|
||||
from typing import Any, Callable, Iterator, Optional, Tuple, TypeVar, cast
|
||||
|
||||
try:
|
||||
import winreg
|
||||
except ModuleNotFoundError as e:
|
||||
raise ModuleNotFoundError("This module is for Windows only") from e
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
T = TypeVar('T')
|
||||
|
||||
|
||||
def registry_enum(
|
||||
key: winreg.HKEYType, enum_fn: Callable[[winreg.HKEYType, int], T]
|
||||
) -> Iterator[T]:
|
||||
LIMIT = 999
|
||||
n = 0
|
||||
while n < LIMIT:
|
||||
try:
|
||||
yield enum_fn(key, n)
|
||||
n += 1
|
||||
except OSError:
|
||||
break
|
||||
if n == LIMIT:
|
||||
raise ValueError(f"Too many registry keys under {key}")
|
||||
|
||||
|
||||
def registry_subkeys(key: winreg.HKEYType) -> Iterator[str]:
|
||||
return registry_enum(key, winreg.EnumKey)
|
||||
|
||||
|
||||
def registry_values(key: winreg.HKEYType) -> Iterator[Tuple[str, Any, int]]:
|
||||
return registry_enum(key, winreg.EnumValue)
|
||||
|
||||
|
||||
def registry_path_ghostscript(env=None) -> Iterator[Path]:
|
||||
try:
|
||||
with winreg.OpenKey(
|
||||
winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Artifex\GPL Ghostscript"
|
||||
) as k:
|
||||
latest_gs = max(registry_subkeys(k), key=LooseVersion)
|
||||
with winreg.OpenKey(
|
||||
winreg.HKEY_LOCAL_MACHINE, fr"SOFTWARE\Artifex\GPL Ghostscript\{latest_gs}"
|
||||
) as k:
|
||||
_, gs_path, _ = next(registry_values(k))
|
||||
yield Path(gs_path) / 'bin'
|
||||
except OSError as e:
|
||||
log.warning(e)
|
||||
|
||||
|
||||
def registry_path_tesseract(env=None) -> Iterator[Path]:
|
||||
try:
|
||||
with winreg.OpenKey(winreg.HKEY_LOCAL_MACHINE, r"SOFTWARE\Tesseract-OCR") as k:
|
||||
for subkey, val, _valtype in registry_values(k):
|
||||
if subkey == 'InstallDir':
|
||||
tesseract_path = Path(val)
|
||||
yield tesseract_path
|
||||
except OSError as e:
|
||||
log.warning(e)
|
||||
|
||||
|
||||
def program_files_paths(env=None) -> Iterator[Path]:
|
||||
if not env:
|
||||
env = os.environ
|
||||
program_files = env.get('PROGRAMFILES', '')
|
||||
|
||||
def path_walker() -> Iterator[Path]:
|
||||
for path in Path(program_files).iterdir():
|
||||
if not path.is_dir():
|
||||
continue
|
||||
if path.name.lower() == 'tesseract-ocr':
|
||||
yield path
|
||||
elif path.name.lower() == 'gs':
|
||||
yield from (p for p in path.glob('**/bin') if p.is_dir())
|
||||
|
||||
return iter(
|
||||
sorted(
|
||||
(p for p in path_walker()),
|
||||
key=lambda p: (p.name, p.parent.name),
|
||||
reverse=True,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def paths_from_env(env=None) -> Iterator[Path]:
|
||||
return (Path(p) for p in os.get_exec_path(env) if p)
|
||||
|
||||
|
||||
def shim_path(new_paths: Callable[[Any], Iterator[Path]], env=None) -> str:
|
||||
if not env:
|
||||
env = os.environ
|
||||
return os.pathsep.join(str(p) for p in new_paths(env) if p)
|
||||
|
||||
|
||||
SHIMS = [
|
||||
paths_from_env,
|
||||
registry_path_ghostscript,
|
||||
registry_path_tesseract,
|
||||
program_files_paths,
|
||||
]
|
||||
|
||||
|
||||
def fix_windows_args(program, args, env):
|
||||
"""Adjust our desired program and command line arguments for use on Windows"""
|
||||
|
||||
if sys.version_info < (3, 8):
|
||||
# bpo-33617 - Windows needs manual Path -> str conversion
|
||||
args = [os.fspath(arg) for arg in args]
|
||||
program = os.fspath(program)
|
||||
|
||||
# If we are running a .py on Windows, ensure we call it with this Python
|
||||
# (to support test suite shims)
|
||||
if program.lower().endswith('.py'):
|
||||
args = [sys.executable] + args
|
||||
|
||||
# If the program we want is not on the PATH, check elsewhere
|
||||
for shim in SHIMS:
|
||||
shimmed_path = shim_path(shim, env)
|
||||
new_args0 = shutil.which(args[0], path=shimmed_path)
|
||||
if new_args0:
|
||||
args[0] = new_args0
|
||||
break
|
||||
|
||||
return args
|
||||
|
||||
|
||||
def unique_everseen(iterable, key=None):
|
||||
"List unique elements, preserving order. Remember all elements ever seen."
|
||||
# unique_everseen('AAAABBBCCDAABBB') --> A B C D
|
||||
# unique_everseen('ABBCcAD', str.lower) --> A B C D
|
||||
seen = set()
|
||||
seen_add = seen.add
|
||||
if key is None:
|
||||
key = lambda x: x
|
||||
for element in iterable:
|
||||
k = key(element)
|
||||
if k not in seen:
|
||||
seen_add(k)
|
||||
yield element
|
||||
|
||||
|
||||
def shim_env_path(env=None):
|
||||
if env is None:
|
||||
env = os.environ
|
||||
|
||||
shim_paths = chain.from_iterable(shim(env) for shim in SHIMS)
|
||||
return os.pathsep.join(
|
||||
str(p) for p in unique_everseen(shim_paths, key=lambda p: str.casefold(str(p)))
|
||||
)
|
||||
@@ -23,21 +23,22 @@ from unittest.mock import patch
|
||||
|
||||
from ocrmypdf import hookimpl
|
||||
from ocrmypdf.builtin_plugins import ghostscript
|
||||
from ocrmypdf.subprocess import run
|
||||
from ocrmypdf.subprocess import run_polling_stderr
|
||||
|
||||
elision_warning = """GPL Ghostscript 9.20: Setting Overprint Mode to 1
|
||||
not permitted in PDF/A-2, overprint mode not set"""
|
||||
|
||||
|
||||
def run_append_stderr(*args, **kwargs):
|
||||
proc = run(*args, **kwargs)
|
||||
proc.stderr = b'\n'.join([proc.stderr, elision_warning.encode('utf-8')])
|
||||
proc = run_polling_stderr(*args, **kwargs)
|
||||
proc.stderr += '\n' + elision_warning + '\n'
|
||||
return proc
|
||||
|
||||
|
||||
@hookimpl
|
||||
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
||||
with patch('ocrmypdf._exec.ghostscript.run', new=run_append_stderr):
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
|
||||
mock.side_effect = run_append_stderr
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=pdf_pages,
|
||||
pdfmark=pdfmark,
|
||||
@@ -45,5 +46,7 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
|
||||
compression=compression,
|
||||
pdf_version=pdf_version,
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=None,
|
||||
)
|
||||
return output_file
|
||||
mock.assert_called_once()
|
||||
return output_file
|
||||
|
||||
@@ -23,7 +23,7 @@ from unittest.mock import patch
|
||||
|
||||
from ocrmypdf import hookimpl
|
||||
from ocrmypdf.builtin_plugins import ghostscript
|
||||
from ocrmypdf.subprocess import run
|
||||
from ocrmypdf.subprocess import run_polling_stderr
|
||||
|
||||
|
||||
def run_rig_args(args, **kwargs):
|
||||
@@ -33,13 +33,14 @@ def run_rig_args(args, **kwargs):
|
||||
new_args = [
|
||||
arg for arg in args if not arg.startswith('-dPDFA') and not arg.endswith('.ps')
|
||||
]
|
||||
proc = run(new_args, **kwargs)
|
||||
proc = run_polling_stderr(new_args, **kwargs)
|
||||
return proc
|
||||
|
||||
|
||||
@hookimpl
|
||||
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
||||
with patch('ocrmypdf._exec.ghostscript.run', new=run_rig_args):
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
|
||||
mock.side_effect = run_rig_args
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=pdf_pages,
|
||||
pdfmark=pdfmark,
|
||||
@@ -47,5 +48,7 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
|
||||
compression=compression,
|
||||
pdf_version=pdf_version,
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=None,
|
||||
)
|
||||
mock.assert_called()
|
||||
return output_file
|
||||
|
||||
@@ -44,7 +44,8 @@ def rasterize_pdf_page(
|
||||
rotation=None,
|
||||
filter_vector=False,
|
||||
) -> Path:
|
||||
with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail):
|
||||
with patch('ocrmypdf._exec.ghostscript.run') as mock:
|
||||
mock.side_effect = raise_gs_fail
|
||||
ghostscript.rasterize_pdf_page(
|
||||
input_file=input_file,
|
||||
output_file=output_file,
|
||||
@@ -55,4 +56,5 @@ def rasterize_pdf_page(
|
||||
rotation=rotation,
|
||||
filter_vector=filter_vector,
|
||||
)
|
||||
mock.assert_called()
|
||||
return output_file
|
||||
|
||||
@@ -34,7 +34,8 @@ def raise_gs_fail(*args, **kwargs):
|
||||
|
||||
@hookimpl
|
||||
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
||||
with patch('ocrmypdf._exec.ghostscript.run', new=raise_gs_fail):
|
||||
with patch('ocrmypdf._exec.ghostscript.run_polling_stderr') as mock:
|
||||
mock.side_effect = raise_gs_fail
|
||||
ghostscript.generate_pdfa(
|
||||
pdf_pages=pdf_pages,
|
||||
pdfmark=pdfmark,
|
||||
@@ -42,5 +43,7 @@ def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdf
|
||||
compression=compression,
|
||||
pdf_version=pdf_version,
|
||||
pdfa_part=pdfa_part,
|
||||
progressbar_class=None,
|
||||
)
|
||||
mock.assert_called()
|
||||
return output_file
|
||||
|
||||
@@ -26,6 +26,7 @@ that is not UTF-8 compatible, so we are forced to check that we can convert it
|
||||
and present it to the user.
|
||||
"""
|
||||
|
||||
from contextlib import contextmanager
|
||||
from subprocess import CalledProcessError
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -42,17 +43,25 @@ def bad_utf8(*args, **kwargs):
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def patch_tesseract_run():
|
||||
with patch('ocrmypdf._exec.tesseract.run') as mock:
|
||||
mock.side_effect = bad_utf8
|
||||
yield
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
class BadUtf8OcrEngine(TesseractOcrEngine):
|
||||
@staticmethod
|
||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_hocr(
|
||||
input_file, output_hocr, output_text, options
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def generate_pdf(input_file, output_pdf, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=bad_utf8):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_pdf(
|
||||
input_file, output_pdf, output_text, options
|
||||
)
|
||||
|
||||
@@ -19,6 +19,7 @@
|
||||
# TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE
|
||||
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
from contextlib import contextmanager
|
||||
from subprocess import CalledProcessError
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -35,22 +36,30 @@ def raise_size_exception(*args, **kwargs):
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def patch_tesseract_run():
|
||||
with patch('ocrmypdf._exec.tesseract.run') as mock:
|
||||
mock.side_effect = raise_size_exception
|
||||
yield
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
class BigImageErrorOcrEngine(TesseractOcrEngine):
|
||||
@staticmethod
|
||||
def get_orientation(input_file, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
|
||||
with patch_tesseract_run():
|
||||
return TesseractOcrEngine.get_orientation(input_file, options)
|
||||
|
||||
@staticmethod
|
||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_hocr(
|
||||
input_file, output_hocr, output_text, options
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def generate_pdf(input_file, output_pdf, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_size_exception):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_pdf(
|
||||
input_file, output_pdf, output_text, options
|
||||
)
|
||||
|
||||
@@ -165,7 +165,7 @@ def cached_run(options, run_args, **run_kwargs):
|
||||
|
||||
def clean_sys_argv():
|
||||
for arg in run_args[1:]:
|
||||
yield re.sub(r'.*/com.github.ocrmypdf[^/]+[/](.*)', r'$TMPDIR/\1', arg)
|
||||
yield re.sub(r'.*/ocrmypdf[.]io[.][^/]+[/](.*)', r'$TMPDIR/\1', arg)
|
||||
|
||||
manifest['args'] = list(clean_sys_argv())
|
||||
with (Path(CACHE_ROOT) / 'manifest.jsonl').open('a') as f:
|
||||
|
||||
@@ -20,6 +20,7 @@
|
||||
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
|
||||
import signal
|
||||
from contextlib import contextmanager
|
||||
from subprocess import CalledProcessError
|
||||
from unittest.mock import patch
|
||||
|
||||
@@ -37,22 +38,30 @@ def raise_crash(*args, **kwargs):
|
||||
)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def patch_tesseract_run():
|
||||
with patch('ocrmypdf._exec.tesseract.run') as mock:
|
||||
mock.side_effect = raise_crash
|
||||
yield
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
class CrashOcrEngine(TesseractOcrEngine):
|
||||
@staticmethod
|
||||
def get_orientation(input_file, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
|
||||
with patch_tesseract_run():
|
||||
return TesseractOcrEngine.get_orientation(input_file, options)
|
||||
|
||||
@staticmethod
|
||||
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_hocr(
|
||||
input_file, output_hocr, output_text, options
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def generate_pdf(input_file, output_pdf, output_text, options):
|
||||
with patch('ocrmypdf._exec.tesseract.run', new=raise_crash):
|
||||
with patch_tesseract_run():
|
||||
TesseractOcrEngine.generate_pdf(
|
||||
input_file, output_pdf, output_text, options
|
||||
)
|
||||
|
||||
+13
-6
@@ -13,7 +13,6 @@ from unittest.mock import MagicMock
|
||||
import pytest
|
||||
|
||||
from ocrmypdf import helpers as helpers
|
||||
from ocrmypdf.subprocess import shim_paths_with_program_files
|
||||
|
||||
|
||||
class TestSafeSymlink:
|
||||
@@ -37,12 +36,17 @@ class TestSafeSymlink:
|
||||
|
||||
|
||||
def test_no_cpu_count(monkeypatch):
|
||||
invoked = False
|
||||
|
||||
def cpu_count_raises():
|
||||
nonlocal invoked
|
||||
invoked = True
|
||||
raise NotImplementedError()
|
||||
|
||||
monkeypatch.setattr(multiprocessing, 'cpu_count', cpu_count_raises)
|
||||
with pytest.warns(expected_warning=UserWarning):
|
||||
assert helpers.available_cpu_count() == 1
|
||||
assert invoked, "Patched function called during test"
|
||||
|
||||
|
||||
def test_deprecated():
|
||||
@@ -94,7 +98,10 @@ class TestFileIsWritable:
|
||||
assert not helpers.is_file_writable(pathmock)
|
||||
|
||||
|
||||
@pytest.mark.skipif(os.name != 'nt', reason="Windows test")
|
||||
def test_shim_paths(tmp_path):
|
||||
from ocrmypdf.subprocess._windows import shim_env_path
|
||||
|
||||
progfiles = tmp_path / 'Program Files'
|
||||
progfiles.mkdir()
|
||||
(progfiles / 'tesseract-ocr').mkdir()
|
||||
@@ -103,9 +110,9 @@ def test_shim_paths(tmp_path):
|
||||
syspath = tmp_path / 'bin'
|
||||
env = {'PROGRAMFILES': str(progfiles), 'PATH': str(syspath)}
|
||||
|
||||
result_str = shim_paths_with_program_files(env=env)
|
||||
result_str = shim_env_path(env=env)
|
||||
results = result_str.split(os.pathsep)
|
||||
assert results[0].endswith('tesseract-ocr')
|
||||
assert results[1].endswith(os.path.join('gs', '9.52', 'bin'))
|
||||
assert results[2].endswith(os.path.join('gs', '9.51', 'bin'))
|
||||
assert results[3] == str(syspath)
|
||||
assert results[0] == str(syspath), results
|
||||
assert results[-3].endswith('tesseract-ocr'), results
|
||||
assert results[-2].endswith(os.path.join('gs', '9.52', 'bin')), results
|
||||
assert results[-1].endswith(os.path.join('gs', '9.51', 'bin')), results
|
||||
|
||||
@@ -65,11 +65,12 @@ def test_cmyk_no_icc(caplog, resources, no_outpdf):
|
||||
def test_img2pdf_fails(resources, no_outpdf):
|
||||
with patch(
|
||||
'ocrmypdf._pipeline.img2pdf.convert', side_effect=img2pdf.ImageOpenError()
|
||||
):
|
||||
) as mock:
|
||||
rc = run_ocrmypdf_api(
|
||||
resources / 'baiona_gray.png', no_outpdf, '--image-dpi', '200'
|
||||
)
|
||||
assert rc == ocrmypdf.ExitCode.input_file
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
def test_jpeg_in_jpeg_out(resources, outpdf):
|
||||
|
||||
+2
-2
@@ -92,10 +92,10 @@ def test_skip_ocr(resources, outpdf):
|
||||
|
||||
def test_redo_ocr(resources, outpdf):
|
||||
in_ = resources / 'graph_ocred.pdf'
|
||||
before = PdfInfo(in_)
|
||||
before = PdfInfo(in_, detailed_analysis=True)
|
||||
out = outpdf
|
||||
out = check_ocrmypdf(in_, out, '--redo-ocr')
|
||||
after = PdfInfo(out)
|
||||
after = PdfInfo(out, detailed_analysis=True)
|
||||
assert before[0].has_text and after[0].has_text
|
||||
assert (
|
||||
before[0].get_textareas() != after[0].get_textareas()
|
||||
|
||||
@@ -306,6 +306,9 @@ def test_kodak_toc(resources, outpdf):
|
||||
assert isinstance(p.Root.Outlines.First, pikepdf.Dictionary)
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
pikepdf.__version__ in ('2.2.2', '2.2.3'), reason="Raises wrong warning"
|
||||
)
|
||||
def test_metadata_fixup_warning(resources, outdir, caplog):
|
||||
options = get_parser().parse_args(
|
||||
args=['--output-type', 'pdfa-2', 'graph.pdf', 'out.pdf']
|
||||
@@ -318,7 +321,7 @@ def test_metadata_fixup_warning(resources, outdir, caplog):
|
||||
)
|
||||
metadata_fixup(working_file=outdir / 'graph.pdf', context=context)
|
||||
for record in caplog.records:
|
||||
assert record.levelname != 'WARNING'
|
||||
assert record.levelname != 'WARNING', "Unexpected warning"
|
||||
|
||||
# Now add some metadata that will not be copyable
|
||||
graph = pikepdf.open(outdir / 'graph.pdf')
|
||||
|
||||
+17
-7
@@ -21,7 +21,15 @@ from ocrmypdf.helpers import Resolution
|
||||
|
||||
check_ocrmypdf = pytest.helpers.check_ocrmypdf # pylint: disable=e1101
|
||||
|
||||
needs_pngquant = pytest.mark.skipif(
|
||||
not pngquant.available(), reason="pngquant not installed"
|
||||
)
|
||||
needs_jbig2enc = pytest.mark.skipif(
|
||||
not jbig2enc.available(), reason="jbig2enc not installed"
|
||||
)
|
||||
|
||||
|
||||
@needs_pngquant
|
||||
@pytest.mark.parametrize('pdf', ['multipage.pdf', 'palette.pdf'])
|
||||
def test_basic(resources, pdf, outpdf):
|
||||
infile = resources / pdf
|
||||
@@ -30,6 +38,7 @@ def test_basic(resources, pdf, outpdf):
|
||||
assert 0.98 * Path(outpdf).stat().st_size <= Path(infile).stat().st_size
|
||||
|
||||
|
||||
@needs_pngquant
|
||||
def test_mono_not_inverted(resources, outdir):
|
||||
infile = resources / '2400dpi.pdf'
|
||||
opt.main(infile, outdir / 'out.pdf', level=3)
|
||||
@@ -45,7 +54,7 @@ def test_mono_not_inverted(resources, outdir):
|
||||
assert im.getpixel((0, 0)) == 255, "Expected white background"
|
||||
|
||||
|
||||
@pytest.mark.skipif(not pngquant.available(), reason='need pngquant')
|
||||
@needs_pngquant
|
||||
def test_jpg_png_params(resources, outpdf):
|
||||
check_ocrmypdf(
|
||||
resources / 'crom.png',
|
||||
@@ -63,7 +72,7 @@ def test_jpg_png_params(resources, outpdf):
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.skipif(not jbig2enc.available(), reason='need jbig2enc')
|
||||
@needs_jbig2enc
|
||||
@pytest.mark.parametrize('lossy', [False, True])
|
||||
def test_jbig2_lossy(lossy, resources, outpdf):
|
||||
args = [
|
||||
@@ -95,10 +104,8 @@ def test_jbig2_lossy(lossy, resources, outpdf):
|
||||
assert len(pim.decode_parms) == 0
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
not jbig2enc.available() or not pngquant.available(),
|
||||
reason='need jbig2enc and pngquant',
|
||||
)
|
||||
@needs_pngquant
|
||||
@needs_jbig2enc
|
||||
def test_flate_to_jbig2(resources, outdir):
|
||||
# This test requires an image that pngquant is capable of converting to
|
||||
# to 1bpp - so use an existing 1bpp image, convert up, confirm it can
|
||||
@@ -126,6 +133,7 @@ def test_flate_to_jbig2(resources, outdir):
|
||||
assert pim.filters[0] == '/JBIG2Decode'
|
||||
|
||||
|
||||
@needs_pngquant
|
||||
def test_multiple_pngs(resources, outdir):
|
||||
with Path.open(outdir / 'in.pdf', 'wb') as inpdf:
|
||||
img2pdf.convert(
|
||||
@@ -141,7 +149,8 @@ def test_multiple_pngs(resources, outdir):
|
||||
draw.rectangle((0, 0, im.width, im.height), fill=128)
|
||||
im.save(output_file)
|
||||
|
||||
with patch('ocrmypdf.optimize.pngquant.quantize', new=mockquant):
|
||||
with patch('ocrmypdf.optimize.pngquant.quantize') as mock:
|
||||
mock.side_effect = mockquant
|
||||
check_ocrmypdf(
|
||||
outdir / 'in.pdf',
|
||||
outdir / 'out.pdf',
|
||||
@@ -155,6 +164,7 @@ def test_multiple_pngs(resources, outdir):
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_noop.py',
|
||||
)
|
||||
mock.assert_called()
|
||||
|
||||
with pikepdf.open(outdir / 'in.pdf') as inpdf, pikepdf.open(
|
||||
outdir / 'out.pdf'
|
||||
|
||||
@@ -28,6 +28,12 @@ from ocrmypdf.pdfinfo import PdfInfo
|
||||
['1,3,-11', BadArgsError],
|
||||
['1-,', BadArgsError],
|
||||
['start-end', BadArgsError],
|
||||
['1-0', BadArgsError],
|
||||
['99-98', BadArgsError],
|
||||
['0-0', BadArgsError],
|
||||
['1-0,3-4', BadArgsError],
|
||||
[',', BadArgsError],
|
||||
['', BadArgsError],
|
||||
],
|
||||
)
|
||||
def test_pages(pages, result):
|
||||
|
||||
@@ -28,11 +28,12 @@ def test_no_unpaper(resources, no_outpdf):
|
||||
output = fspath(no_outpdf)
|
||||
|
||||
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
|
||||
with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version:
|
||||
mock_unpaper_version.side_effect = FileNotFoundError("unpaper")
|
||||
with patch("ocrmypdf._exec.unpaper.version") as mock:
|
||||
mock.side_effect = FileNotFoundError("unpaper")
|
||||
|
||||
with pytest.raises(MissingDependencyError):
|
||||
check_options(options, pm)
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
def test_old_unpaper(resources, no_outpdf):
|
||||
@@ -40,11 +41,12 @@ def test_old_unpaper(resources, no_outpdf):
|
||||
output = fspath(no_outpdf)
|
||||
|
||||
_parser, options, pm = get_parser_options_plugins(["--clean", input_, output])
|
||||
with patch("ocrmypdf._exec.unpaper.version") as mock_unpaper_version:
|
||||
mock_unpaper_version.return_value = '0.5'
|
||||
with patch("ocrmypdf._exec.unpaper.version") as mock:
|
||||
mock.return_value = '0.5'
|
||||
|
||||
with pytest.raises(MissingDependencyError):
|
||||
check_options(options, pm)
|
||||
mock.assert_called()
|
||||
|
||||
|
||||
@pytest.mark.skipif(not have_unpaper(), reason="requires unpaper")
|
||||
|
||||
@@ -188,18 +188,20 @@ def test_language_warning(caplog):
|
||||
caplog.set_level(logging.DEBUG)
|
||||
with patch(
|
||||
'ocrmypdf._validation.locale.getlocale', return_value=('en_US', 'UTF-8')
|
||||
):
|
||||
) as mock:
|
||||
vd.check_options_languages(opts, {'eng'})
|
||||
assert opts.languages == {'eng'}
|
||||
assert '' in caplog.text
|
||||
mock.assert_called_once()
|
||||
|
||||
opts = make_opts(language=None)
|
||||
with patch(
|
||||
'ocrmypdf._validation.locale.getlocale', return_value=('fr_FR', 'UTF-8')
|
||||
):
|
||||
) as mock:
|
||||
vd.check_options_languages(opts, {'eng'})
|
||||
assert opts.languages == {'eng'}
|
||||
assert 'assuming --language' in caplog.text
|
||||
mock.assert_called_once()
|
||||
|
||||
|
||||
def test_version_comparison():
|
||||
@@ -265,7 +267,8 @@ def test_pagesegmode_warning(caplog):
|
||||
|
||||
|
||||
def test_two_languages():
|
||||
with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True):
|
||||
with patch('ocrmypdf._exec.tesseract.has_textonly_pdf', return_value=True) as mock:
|
||||
vd._check_options(
|
||||
*make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'}
|
||||
)
|
||||
mock.assert_called()
|
||||
|
||||
Reference in New Issue
Block a user