Compare commits

...
12 Commits
Author SHA1 Message Date
James R. Barlow 28c60c4f82 v13.6.0 release notes 2022-07-03 15:35:22 -07:00
James R. Barlow a5efc4af9b unpaper: replace input pnm with png
Unpaper or its underlying libraries don't seem to accept pnms with an
odd integer width. Although it's not clear if this is the issue at all.

In any case, keeping the image a PNG works around the issue. unpaper
only accepted PNM input in the past, which is why we send it PNM.
Since it now accepts PNG, we might as well use PNG.

Unpaper can write PNG as output too, but this added a few seconds to
the test suite was not committed.

Related issues:

https://github.com/ocrmypdf/OCRmyPDF/issues/887

https://github.com/ocrmypdf/OCRmyPDF/issues/665

https://github.com/unpaper/unpaper/issues/82
2022-07-03 15:32:16 -07:00
James R. Barlow 1141235c42 Merge remote-tracking branch 'origin/master' 2022-06-24 01:05:08 -07:00
2b6b7a4975 Update README to show nix package manager install option (#904)
ocrmypdf now works on M1 Mac (aarch64-darwin)

Co-authored-by: xave <xavieking@gmail.com>
2022-06-19 01:47:10 -07:00
James R. Barlow b062c9e8c0 pluginspec: hint that optimize may need to implement initialize
[ci skip]
2022-06-19 01:35:30 -07:00
James R. Barlow ed632ae366 docs: update batch to avoid suggesting Docker volumes
[ci skip]
2022-06-19 01:01:39 -07:00
James R. Barlow af742229e7 docs: fix sentence fragment in batch page
Closes #980

[ci skip]
2022-06-19 00:41:58 -07:00
Alexander JaustandGitHub e2d998245d Fix type in cookbook.rst (#978)
Add missing dash in warning about `--clean-final` and `--remove-background` commands.
2022-06-19 00:32:35 -07:00
James R. Barlow 8c58e95c3a Add new initialize hookspec to make suppress plugins easier 2022-06-19 00:31:42 -07:00
James R. Barlow 61600111d3 test_pdfinfo: refactor by extracting fixtures 2022-06-18 16:29:57 -07:00
James R. Barlow e4c45e3d3b Plugins should ideally not import from ocrmypdf._* 2022-06-18 16:03:55 -07:00
Julius BullingerandGitHub 7cabbb125f watcher: Add an option to archive processed originals (#951)
* watcher: Add an option to archive processed originals

This adds a feature from existing OCRmyPDF watchdog Docker containers like meyay/ocrmypdf-batch and unze/ocrmypdf-watchdog. With this option, the input directory can be kept clean from already processed files, without losing the originals.

* docs: Improve watcher.py's Docker parameters documentation
2022-06-17 15:17:03 -07:00
15 changed files with 147 additions and 57 deletions
+2 -1
View File
@@ -62,7 +62,8 @@ Linux, Windows, macOS and FreeBSD are supported. Docker images are also availabl
| Debian, Ubuntu | ``apt install ocrmypdf`` | | Debian, Ubuntu | ``apt install ocrmypdf`` |
| Windows Subsystem for Linux | ``apt install ocrmypdf`` | | Windows Subsystem for Linux | ``apt install ocrmypdf`` |
| Fedora | ``dnf install ocrmypdf`` | | Fedora | ``dnf install ocrmypdf`` |
| macOS | ``brew install ocrmypdf`` | | macOS (Homebrew) | ``brew install ocrmypdf`` |
| macOS (nix) | ``nix-env -i ocrmypdf`` |
| LinuxBrew | ``brew install ocrmypdf`` | | LinuxBrew | ``brew install ocrmypdf`` |
| FreeBSD | ``pkg install py37-ocrmypdf`` | | FreeBSD | ``pkg install py37-ocrmypdf`` |
| Conda | ``conda install ocrmypdf`` | | Conda | ``conda install ocrmypdf`` |
+20 -12
View File
@@ -36,18 +36,21 @@ Directory trees
=============== ===============
This will walk through a directory tree and run OCR on all files in This will walk through a directory tree and run OCR on all files in
place, printing the output in a way that makes place, and printing each filename in between runs:
.. code-block:: bash .. code-block:: bash
find . -printf '%p' -name '*.pdf' -exec ocrmypdf '{}' '{}' \; find . -printf '%p\n' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
Alternatively, with a docker container (mounts a volume to the container Alternatively, with a Docker container and streaming the file through
where the PDFs are stored): standard input and output:
.. code-block:: bash .. code-block:: bash
find . -printf '%p' -name '*.pdf' -exec docker run --rm -v <host dir>:<container dir> jbarlow83/ocrmypdf '<container dir>/{}' '<container dir>/{}' \; find . -name '*.pdf' -print0 | xargs -0 | while read pdf; do
pdfout=$(mktemp)
docker run --rm -i jbarlow83/ocrmypdf - - <$pdf >$pdfout && cp $pdfout $pdf
done
This only runs one ``ocrmypdf`` process at a time. This variation uses This only runs one ``ocrmypdf`` process at a time. This variation uses
``find`` to create a directory list and ``parallel`` to parallelize runs ``find`` to create a directory list and ``parallel`` to parallelize runs
@@ -124,7 +127,9 @@ Users may need to customize the script to meet their requirements.
"OCR_INPUT_DIRECTORY", "Set input directory to monitor (recursive)" "OCR_INPUT_DIRECTORY", "Set input directory to monitor (recursive)"
"OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)" "OCR_OUTPUT_DIRECTORY", "Set output directory (should not be under input)"
"OCR_ARCHIVE_DIRECTORY", "Set archive directory for processed originals (should not be under input, requires ``OCR_ON_SUCCESS_ARCHIVE`` to be set)"
"OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)" "OCR_ON_SUCCESS_DELETE", "This will delete the input file if the exit code is 0 (OK)"
"OCR_ON_SUCCESS_ARCHIVE", "This will move the processed orignal file to ``OCR_ARCHIVE_DIRECTORY`` if the exit code is 0 (OK). Note that ``OCR_ON_SUCCESS_DELETE`` takes precedence over this option, i.e. if both options are set, the input file will be deleted."
"OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``" "OCR_OUTPUT_DIRECTORY_YEAR_MONTH", "This will place files in the output in ``{output}/{year}/{month}/{filename}``"
"OCR_DESKEW", "Apply deskew to crooked input PDFs" "OCR_DESKEW", "Apply deskew to crooked input PDFs"
"OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``." "OCR_JSON_SETTINGS", "A JSON string specifying any other arguments for ``ocrmypdf.ocr``, e.g. ``'OCR_JSON_SETTINGS={""rotate_pages"": true}'``."
@@ -144,16 +149,18 @@ The watcher service is included in the OCRmyPDF Docker image. To run it:
docker run \ docker run \
-v <path to files to convert>:/input \ -v <path to files to convert>:/input \
-v <path to store results>:/output \ -v <path to store results>:/output \
-v <path to store processed originals>:/archive \
-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \ -e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1 \
-e OCR_ON_SUCCESS_DELETE=1 \ -e OCR_ON_SUCCESS_ARCHIVE=1 \
-e OCR_DESKEW=1 \ -e OCR_DESKEW=1 \
-e PYTHONUNBUFFERED=1 \ -e PYTHONUNBUFFERED=1 \
-it --entrypoint python3 \ -it --entrypoint python3 \
jbarlow83/ocrmypdf \ jbarlow83/ocrmypdf \
watcher.py watcher.py
This service will watch for a file that matches ``/input/\*.pdf`` and will This service will watch for a file that matches ``/input/\*.pdf``,
convert it to a OCRed PDF in ``/output/``. The parameters to this image are: convert it to a OCRed PDF in ``/output/``, and move the processed
original to ``/archive``. The parameters to this image are:
.. csv-table:: watcher.py parameters for Docker .. csv-table:: watcher.py parameters for Docker
:header: "Parameter", "Description" :header: "Parameter", "Description"
@@ -161,10 +168,11 @@ convert it to a OCRed PDF in ``/output/``. The parameters to this image are:
"``-v <path to files to convert>:/input``", "Files placed in this location will be OCRed" "``-v <path to files to convert>:/input``", "Files placed in this location will be OCRed"
"``-v <path to store results>:/output``", "This is where OCRed files will be stored" "``-v <path to store results>:/output``", "This is where OCRed files will be stored"
"``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1" "``-v <path to store processed originals>:/archive``", "Archive processed originals here"
"``-e OCR_ON_SUCCESS_DELETE=1``", "Define environment variable" "``-e OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1``", "Define environment variable ``OCR_OUTPUT_DIRECTORY_YEAR_MONTH=1`` to place files in the output in ``{output}/{year}/{month}/{filename}``"
"``-e OCR_DESKEW=1``", "Define environment variable" "``-e OCR_ON_SUCCESS_ARCHIVE=1``", "Define environment variable ``OCR_ON_SUCCESS_ARCHIVE`` to move processed originals"
"``-e PYTHONBUFFERED=1``", "This will force STDOUT to be unbuffered and allow you to see messages in docker logs" "``-e OCR_DESKEW=1``", "Define environment variable ``OCR_DESKEW`` to apply deskew to crooked input PDFs"
"``-e PYTHONBUFFERED=1``", "This will force ``STDOUT`` to be unbuffered and allow you to see messages in docker logs"
This service relies on polling to check for changes to the filesystem. It This service relies on polling to check for changes to the filesystem. It
may not be suitable for some environments, such as filesystems shared on a may not be suitable for some environments, such as filesystems shared on a
+1 -1
View File
@@ -200,7 +200,7 @@ might remove desirable content, especially from poor quality scans.
.. warning:: .. warning::
``--clean-final`` and ``-remove-background`` may leave undesirable ``--clean-final`` and ``--remove-background`` may leave undesirable
visual artifacts in some images where their algorithms have visual artifacts in some images where their algorithms have
shortcomings. Files should be visually reviewed after using these shortcomings. Files should be visually reviewed after using these
options. options.
+9 -1
View File
@@ -24,10 +24,18 @@ tagged yet.
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg .. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
v13.6.0
=======
- Added a new ``initialize`` plugin hook, making it possible to suppress built-in
plugins more easily, among other possibilities.
- Fixed an issue where unpaper would exit with a "wrong stream" error, probably
related to images with an odd integer width. :issue:`887, 665`
v13.5.0 v13.5.0
======= =======
- Added a new ``optimize_pdf`` pluginhook, making it possible to create plugins that - Added a new ``optimize_pdf`` plugin hook, making it possible to create plugins that
replace or enhance OCRmyPDF's PDF optimizer. replace or enhance OCRmyPDF's PDF optimizer.
- Removed all max version restrictions. Our new policy is to blacklist known-bad releases - Removed all max version restrictions. Our new policy is to blacklist known-bad releases
and only block known-bad versions of dependencies. and only block known-bad versions of dependencies.
+14 -4
View File
@@ -23,6 +23,7 @@
import json import json
import logging import logging
import os import os
import shutil
import sys import sys
import time import time
from datetime import datetime from datetime import datetime
@@ -44,8 +45,10 @@ def getenv_bool(name: str, default: str = 'False'):
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input') INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output') OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
ARCHIVE_DIRECTORY = os.getenv('OCR_ARCHIVE_DIRECTORY', '/processed')
OUTPUT_DIRECTORY_YEAR_MONTH = getenv_bool('OCR_OUTPUT_DIRECTORY_YEAR_MONTH') OUTPUT_DIRECTORY_YEAR_MONTH = getenv_bool('OCR_OUTPUT_DIRECTORY_YEAR_MONTH')
ON_SUCCESS_DELETE = getenv_bool('OCR_ON_SUCCESS_DELETE') ON_SUCCESS_DELETE = getenv_bool('OCR_ON_SUCCESS_DELETE')
ON_SUCCESS_ARCHIVE = getenv_bool('OCR_ON_SUCCESS_ARCHIVE')
DESKEW = getenv_bool('OCR_DESKEW') DESKEW = getenv_bool('OCR_DESKEW')
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}')) OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1')) POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
@@ -108,9 +111,13 @@ def execute_ocrmypdf(file_path):
deskew=DESKEW, deskew=DESKEW,
**OCR_JSON_SETTINGS, **OCR_JSON_SETTINGS,
) )
if exit_code == 0 and ON_SUCCESS_DELETE: if exit_code == 0:
log.info(f'OCR is done. Deleting: {file_path}') if ON_SUCCESS_DELETE:
file_path.unlink() log.info(f'OCR is done. Deleting: {file_path}')
file_path.unlink()
elif ON_SUCCESS_ARCHIVE:
log.info(f'OCR is done. Archiving {file_path.name} to {ARCHIVE_DIRECTORY}')
shutil.move(file_path, f'{ARCHIVE_DIRECTORY}/{file_path.name}')
else: else:
log.info('OCR is done') log.info('OCR is done')
@@ -135,13 +142,16 @@ def main():
f"Starting OCRmyPDF watcher with config:\n" f"Starting OCRmyPDF watcher with config:\n"
f"Input Directory: {INPUT_DIRECTORY}\n" f"Input Directory: {INPUT_DIRECTORY}\n"
f"Output Directory: {OUTPUT_DIRECTORY}\n" f"Output Directory: {OUTPUT_DIRECTORY}\n"
f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}" f"Output Directory Year & Month: {OUTPUT_DIRECTORY_YEAR_MONTH}\n"
f"Archive Directory: {ARCHIVE_DIRECTORY}"
) )
log.debug( log.debug(
f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n" f"INPUT_DIRECTORY: {INPUT_DIRECTORY}\n"
f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n" f"OUTPUT_DIRECTORY: {OUTPUT_DIRECTORY}\n"
f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n" f"OUTPUT_DIRECTORY_YEAR_MONTH: {OUTPUT_DIRECTORY_YEAR_MONTH}\n"
f"ARCHIVE_DIRECTORY: {ARCHIVE_DIRECTORY}\n"
f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n" f"ON_SUCCESS_DELETE: {ON_SUCCESS_DELETE}\n"
f"ON_SUCCESS_ARCHIVE: {ON_SUCCESS_ARCHIVE}\n"
f"DESKEW: {DESKEW}\n" f"DESKEW: {DESKEW}\n"
f"ARGS: {OCR_JSON_SETTINGS}\n" f"ARGS: {OCR_JSON_SETTINGS}\n"
f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n" f"POLL_NEW_FILE_SECONDS: {POLL_NEW_FILE_SECONDS}\n"
+19 -19
View File
@@ -72,13 +72,13 @@ def version() -> str:
return get_version('unpaper') return get_version('unpaper')
SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'} SUPPORTED_MODES = {'1', 'L', 'RGB'}
def _convert_image(im: Image.Image) -> Tuple[Image.Image, bool, str]: def _convert_image(im: Image.Image) -> Tuple[Image.Image, bool]:
im_modified = False im_modified = False
if im.mode not in SUFFIXES: if im.mode not in SUPPORTED_MODES:
log.info("Converting image to other colorspace") log.info("Converting image to other colorspace")
try: try:
if im.mode == 'P' and len(im.getcolors()) == 2: if im.mode == 'P' and len(im.getcolors()) == 2:
@@ -91,13 +91,11 @@ def _convert_image(im: Image.Image) -> Tuple[Image.Image, bool, str]:
) from e ) from e
else: else:
im_modified = True im_modified = True
try: if im.mode not in SUPPORTED_MODES:
suffix = SUFFIXES[im.mode] raise MissingDependencyError(
except KeyError: "Failed to convert image to a supported format."
raise MissingDependencyError( ) from None
"Failed to convert image to a supported format." return im, im_modified
) from None
return im, im_modified, suffix
@contextmanager @contextmanager
@@ -105,19 +103,21 @@ def _setup_unpaper_io(input_file: Path) -> Iterator[Tuple[Path, Path, Path]]:
with Image.open(input_file) as im: with Image.open(input_file) as im:
if im.width * im.height >= UNPAPER_IMAGE_PIXEL_LIMIT: if im.width * im.height >= UNPAPER_IMAGE_PIXEL_LIMIT:
raise UnpaperImageTooLargeError(w=im.width, h=im.height) raise UnpaperImageTooLargeError(w=im.width, h=im.height)
im, im_modified, suffix = _convert_image(im) im, im_modified = _convert_image(im)
with TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir: with TemporaryDirectory(ignore_cleanup_errors=True) as tmpdir:
tmppath = Path(tmpdir) tmppath = Path(tmpdir)
if im_modified or input_file.suffix != '.pnm': if im_modified or input_file.suffix != '.png':
input_pnm = tmppath / 'input.pnm' input_png = tmppath / 'input.png'
im.save(input_pnm, format='PPM') im.save(input_png, format='PNG')
else: else:
# No changes, PNG input, just use the file we already have # No changes, PNG input, just use the file we already have
input_pnm = input_file input_png = input_file
output_pnm = tmppath / f'output{suffix}' # unpaper can write .png too, but it seems to write them slowly
yield input_pnm, output_pnm, tmppath # adds a few seconds to test suite - so just use pnm
output_pnm = tmppath / 'output.pnm'
yield input_png, output_pnm, tmppath
def run_unpaper( def run_unpaper(
@@ -125,7 +125,7 @@ def run_unpaper(
) -> None: ) -> None:
args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args args_unpaper = ['unpaper', '-v', '--dpi', str(round(dpi, 6))] + mode_args
with _setup_unpaper_io(input_file) as (input_pnm, output_pnm, tmpdir): with _setup_unpaper_io(input_file) as (input_png, output_pnm, tmpdir):
# To prevent any shenanigans from accepting arbitrary parameters in # To prevent any shenanigans from accepting arbitrary parameters in
# --unpaper-args, we: # --unpaper-args, we:
# 1) run with cwd set to a tmpdir with only unpaper's files # 1) run with cwd set to a tmpdir with only unpaper's files
@@ -133,7 +133,7 @@ def run_unpaper(
# 3) append absolute paths for the input and output file # 3) append absolute paths for the input and output file
# This should ensure that a user cannot clobber some other file with # This should ensure that a user cannot clobber some other file with
# their unpaper arguments (whether intentionally or otherwise) # their unpaper arguments (whether intentionally or otherwise)
args_unpaper.extend([os.fspath(input_pnm), os.fspath(output_pnm)]) args_unpaper.extend([os.fspath(input_png), os.fspath(output_pnm)])
run( run(
args_unpaper, args_unpaper,
close_fds=True, close_fds=True,
+3
View File
@@ -116,6 +116,9 @@ def get_parser_options_plugins(
plugin_manager = get_plugin_manager(pre_options.plugins) plugin_manager = get_plugin_manager(pre_options.plugins)
parser = get_parser() parser = get_parser()
plugin_manager.hook.initialize( # pylint: disable=no-member
plugin_manager=plugin_manager
)
plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
options = parser.parse_args(args=args) options = parser.parse_args(args=args)
+1 -2
View File
@@ -12,8 +12,7 @@ import logging
from pathlib import Path from pathlib import Path
from typing import Sequence, Tuple from typing import Sequence, Tuple
from ocrmypdf import PdfContext, hookimpl from ocrmypdf import Executor, PdfContext, hookimpl
from ocrmypdf._concurrent import Executor
from ocrmypdf._exec import jbig2enc, pngquant from ocrmypdf._exec import jbig2enc, pngquant
from ocrmypdf._pipeline import get_pdf_save_settings from ocrmypdf._pipeline import get_pdf_save_settings
from ocrmypdf.cli import numeric from ocrmypdf.cli import numeric
+33 -4
View File
@@ -22,8 +22,7 @@ from typing import (
import pluggy import pluggy
from ocrmypdf import PdfContext from ocrmypdf import Executor, PdfContext
from ocrmypdf._concurrent import Executor
from ocrmypdf.helpers import Resolution from ocrmypdf.helpers import Resolution
if TYPE_CHECKING: if TYPE_CHECKING:
@@ -52,6 +51,33 @@ def get_logging_console() -> Handler:
""" """
@hookspec
def initialize(plugin_manager: pluggy.PluginManager):
"""Called when this plugin is first loaded into OCRmyPDF.
The primary intended use of this is for plugins to check compatibility with other
plugins and possibly block other blocks, a plugin that wishes to block ocrmypdf's
built-in optimize plugin could do:
.. code-block::
plugin_manager.set_blocked('ocrmypdf.builtin_plugins.optimize')
It would also be reasonable for an plugin implementation to check if it is unable
to proceed, for example, because a required dependency is missing. (If the plugin's
ability to proceed depends on options and arguments, use ``validate`` instead.)
Raises:
ocrmypdf.exceptions.ExitCodeException: If options are not acceptable
and the application should terminate gracefully with an informative
message and error code.
Note:
This hook will be called from the main process, and may modify global state
before child worker processes are forked.
"""
@hookspec @hookspec
def add_options(parser: ArgumentParser) -> None: def add_options(parser: ArgumentParser) -> None:
"""Allows the plugin to add its own command line and API arguments. """Allows the plugin to add its own command line and API arguments.
@@ -484,6 +510,9 @@ def optimize_pdf(
If the implementation fails to produce a smaller file than the input file, it If the implementation fails to produce a smaller file than the input file, it
should return input_pdf instead. should return input_pdf instead.
A plugin that implements a new optimizer may need to suppress the built-in
optimizer by implementing an ``initialize`` hook.
Arguments: Arguments:
input_pdf: The input PDF, which has OCR added. input_pdf: The input PDF, which has OCR added.
output_pdf: The requested filename of the output PDF which should be created output_pdf: The requested filename of the output PDF which should be created
@@ -512,8 +541,8 @@ def optimize_pdf(
def is_optimization_enabled(context: PdfContext) -> bool: def is_optimization_enabled(context: PdfContext) -> bool:
"""For a given PdfContext, OCRmyPDF asks the plugin if optimization is enabled. """For a given PdfContext, OCRmyPDF asks the plugin if optimization is enabled.
It is assumed that an optimization plugin might be installed but could be An optimization plugin might be installed and active but could be disabled by
disabled by user settings. user settings.
If this returns False, OCRmyPDF will take certain actions to finalize the PDF. If this returns False, OCRmyPDF will take certain actions to finalize the PDF.
@@ -0,0 +1 @@
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
@@ -0,0 +1,13 @@
Portez ce vieux whisky au juge
blond qui fume sur son Ile
interieure, a cöte de l'alcöve
ovoide, oU les büches se
consument dans l'ätre, ce qui
lui permet de penser & la
caenogenese de |'etre dont il
est question dans la cause
ambigu& entendue a MoY, dans
un capharnaüm qui, pense-t-il,
diminue ca et la la qualite de son
ceuvre.
+1
View File
@@ -82,3 +82,4 @@
{"tesseract_version": "4.1.1", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__--oem__1__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "--oem", "1", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "4.1.1", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__--oem__1__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "--oem", "1", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
{"tesseract_version": "5.0.0", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__thresholding_method=1__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "-c", "thresholding_method=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "5.0.0", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__thresholding_method=1__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "-c", "thresholding_method=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
{"tesseract_version": "5.0.0", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__thresholding_method=2__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "-c", "thresholding_method=2", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "5.0.0", "system": "Linux", "python": "3.9.5", "argv_slug": "__-l__eng__thresholding_method=2__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/trivial.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "-c", "thresholding_method=2", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
{"tesseract_version": "4.1.1", "system": "Linux", "python": "3.10.4", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/francais.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]}
+30 -13
View File
@@ -24,20 +24,24 @@ from ocrmypdf.pdfinfo.layout import PDFPage
# pylint: disable=protected-access # pylint: disable=protected-access
def test_single_page_text(outdir): @pytest.fixture
def single_page_text(outdir):
filename = outdir / 'text.pdf' filename = outdir / 'text.pdf'
pdf = Canvas(str(filename), pagesize=(8 * inch, 6 * inch)) pdf = Canvas(str(filename), pagesize=(8 * inch, 6 * inch))
text = pdf.beginText() text = pdf.beginText()
text.setFont('Helvetica', 12) text.setFont('Helvetica', 12)
text.setTextOrigin(1 * inch, 3 * inch) text.setTextOrigin(1 * inch, 3 * inch)
text.textLine( text.textLine(
"Methink'st thou art a general offence and every" " man should beat thee." "Methink'st thou art a general offence and every man should beat thee."
) )
pdf.drawText(text) pdf.drawText(text)
pdf.showPage() pdf.showPage()
pdf.save() pdf.save()
return filename
info = pdfinfo.PdfInfo(filename)
def test_single_page_text(single_page_text):
info = pdfinfo.PdfInfo(single_page_text)
assert len(info) == 1 assert len(info) == 1
page = info[0] page = info[0]
@@ -54,7 +58,8 @@ def eight_by_eight():
return im return im
def test_single_page_image(eight_by_eight, outpdf): @pytest.fixture
def eight_by_eight_regular_image(eight_by_eight, outpdf):
im = eight_by_eight im = eight_by_eight
bio = BytesIO() bio = BytesIO()
im.save(bio, format='PNG') im.save(bio, format='PNG')
@@ -71,7 +76,11 @@ def test_single_page_image(eight_by_eight, outpdf):
outputstream=f, outputstream=f,
**IMG2PDF_KWARGS, **IMG2PDF_KWARGS,
) )
info = pdfinfo.PdfInfo(outpdf) return outpdf
def test_single_page_image(eight_by_eight_regular_image):
info = pdfinfo.PdfInfo(eight_by_eight_regular_image)
assert len(info) == 1 assert len(info) == 1
page = info[0] page = info[0]
@@ -88,16 +97,18 @@ def test_single_page_image(eight_by_eight, outpdf):
assert isclose(pdfimage.dpi.y, 8) assert isclose(pdfimage.dpi.y, 8)
def test_single_page_inline_image(eight_by_eight, outdir): @pytest.fixture
filename = outdir / 'image-mono-inline.pdf' def eight_by_eight_inline_image(eight_by_eight, outpdf):
pdf = Canvas(str(filename), pagesize=(8 * 72, 6 * 72)) pdf = Canvas(str(outpdf), pagesize=(8 * 72, 6 * 72))
# Draw image in a 72x72 pt or 1"x1" area # Draw image in a 72x72 pt or 1"x1" area
pdf.drawInlineImage(eight_by_eight, 0, 0, width=72, height=72) pdf.drawInlineImage(eight_by_eight, 0, 0, width=72, height=72)
pdf.showPage() pdf.showPage()
pdf.save() pdf.save()
return outpdf
info = pdfinfo.PdfInfo(filename)
def test_single_page_inline_image(eight_by_eight_inline_image):
info = pdfinfo.PdfInfo(eight_by_eight_inline_image)
print(info) print(info)
pdfimage = info[0].images[0] pdfimage = info[0].images[0]
assert isclose(pdfimage.dpi.x, 8) assert isclose(pdfimage.dpi.x, 8)
@@ -177,7 +188,7 @@ def test_stack_abuse():
pdfinfo.info._interpret_contents(stream) pdfinfo.info._interpret_contents(stream)
stream = pikepdf.Stream(p, b'q Q Q Q Q') stream = pikepdf.Stream(p, b'q Q Q Q Q')
with pytest.warns(UserWarning, match="underflowed") as record: with pytest.warns(UserWarning, match="underflowed"):
pdfinfo.info._interpret_contents(stream) pdfinfo.info._interpret_contents(stream)
stream = pikepdf.Stream(p, b'q ' * 135) stream = pikepdf.Stream(p, b'q ' * 135)
@@ -201,7 +212,8 @@ def test_pages_issue700(monkeypatch, resources):
) )
def test_image_scale0(resources, outpdf): @pytest.fixture
def image_scale0(resources, outpdf):
with pikepdf.open(resources / 'cmyk.pdf') as cmyk: with pikepdf.open(resources / 'cmyk.pdf') as cmyk:
xobj = cmyk.pages[0].as_form_xobject() xobj = cmyk.pages[0].as_form_xobject()
@@ -215,7 +227,12 @@ def test_image_scale0(resources, outpdf):
p, b"q 0 0 0 0 0 0 cm %s Do Q" % bytes(objname) p, b"q 0 0 0 0 0 0 cm %s Do Q" % bytes(objname)
) )
p.save(outpdf) p.save(outpdf)
return outpdf
pi = pdfinfo.PdfInfo(outpdf, detailed_analysis=True, progbar=False, max_workers=1)
def test_image_scale0(image_scale0):
pi = pdfinfo.PdfInfo(
image_scale0, detailed_analysis=True, progbar=False, max_workers=1
)
assert not pi.pages[0]._images[0].dpi.is_finite assert not pi.pages[0]._images[0].dpi.is_finite
assert pi.pages[0].dpi == Resolution(0, 0) assert pi.pages[0].dpi == Resolution(0, 0)