Compare commits

...
17 Commits
Author SHA1 Message Date
James R. Barlow f4f0f3c022 v11.7.3 release notes 2021-03-20 23:30:42 -07:00
James R. Barlow 0a42934c08 Exclude Group 3 images from optimization 2021-03-20 23:28:21 -07:00
James R. Barlow d8f47768f9 v11.7.2 release notes 2021-03-19 00:31:38 -07:00
James R. Barlow c9594a4a5f Update pinned versions to avoid Pillow vulnerabilties
See https://github.com/python-pillow/Pillow/blob/master/CHANGES.rst
2021-03-19 00:31:27 -07:00
James R. Barlow 079c162a96 Ensure sidecar is not input or output file 2021-03-05 00:29:42 -08:00
James R. Barlow 25c8c4656f Fix error message change 2021-03-03 01:02:59 -08:00
James R. Barlow ffcae9a1a0 v11.7.1 release notes 2021-03-03 00:46:35 -08:00
James R. Barlow 6e71fe1186 Clarify --unpaper-args errors 2021-03-03 00:44:21 -08:00
James R. Barlow 0885799010 Update docs for conda
Closes #743
2021-03-03 00:43:45 -08:00
James R. Barlow 8ffc99f648 optimize: log errors more loudly 2021-03-03 00:43:40 -08:00
James R. Barlow 2261c51eff Reactivate pngquant on windows 2021-02-26 01:19:03 -08:00
James R. Barlow 5c470778a3 v11.7.0 release notes 2021-02-26 00:29:52 -08:00
James R. Barlow 4124889f36 Don't generate PDF/A-1b with object streams
Acrobat insists that PDF/A-1b should not have object streams.
Other programs like veraPDF disagree with this restriction, but
we can accommodate Acrobat so we will.

Also add more tests around this.
2021-02-26 00:23:57 -08:00
James R. Barlow a23c22b0e8 helpers: tidy check_pdf 2021-02-25 22:51:53 -08:00
James R. Barlow dd1f5f7215 pyproject: black doesn't like py39 yet 2021-02-25 16:10:20 -08:00
Dima KuznetsovandGitHub 5e2206bae7 Allow --sidecar along --pages (#735) 2021-02-19 16:55:35 -08:00
James R. Barlow 079ee86d43 pyproject: also target py39 2021-02-18 01:48:56 -08:00
16 changed files with 306 additions and 97 deletions
+9 -17
View File
@@ -59,23 +59,15 @@ I searched the web for a free command line tool to OCR PDF files: I found many,
Linux, Windows, macOS and FreeBSD are supported. Docker images are also available. Linux, Windows, macOS and FreeBSD are supported. Docker images are also available.
Users of Debian 9 or later or Ubuntu 16.10 or later may simply | Operating system | Install command |
| ----------------------------- | ------------------------------|
```bash | Debian, Ubuntu | ``apt install ocrmypdf`` |
apt-get install ocrmypdf | Windows Subsystem for Linux | ``apt install ocrmypdf`` |
``` | Fedora | ``dnf install ocrmypdf`` |
| macOS | ``brew install ocrmypdf`` |
and users of Fedora 29 or later may simply | LinuxBrew | ``brew install ocrmypdf`` |
| FreeBSD | ``pkg install py37-ocrmypdf`` |
```bash | Conda | ``conda install ocrmypdf`` |
dnf install ocrmypdf
```
and Homebrew users (macOS, Linux, Windows Subsystem for Linux) may simply
```bash
brew install ocrmypdf
```
For everyone else, [see our documentation](https://ocrmypdf.readthedocs.io/en/latest/installation.html) for installation steps. For everyone else, [see our documentation](https://ocrmypdf.readthedocs.io/en/latest/installation.html) for installation steps.
+1 -1
View File
@@ -32,7 +32,7 @@ stages:
choco install --yes --no-progress --pre tesseract choco install --yes --no-progress --pre tesseract
choco install --yes --no-progress python3 choco install --yes --no-progress python3
choco install --yes --no-progress ghostscript choco install --yes --no-progress ghostscript
# choco install --yes --no-progress pngquant choco install --yes --no-progress pngquant
displayName: "Install system packages" displayName: "Install system packages"
- pwsh: | - pwsh: |
refreshenv refreshenv
+27 -21
View File
@@ -12,19 +12,21 @@ system/platform. This version may be out of date, however.
These platforms have one-liner installs: These platforms have one-liner installs:
+-----------------------------+-------------------------------+ +-------------------------------+-------------------------------+
| Debian, Ubuntu | ``apt install ocrmypdf`` | | Debian, Ubuntu | ``apt install ocrmypdf`` |
+-----------------------------+-------------------------------+ +-------------------------------+-------------------------------+
| Windows Subsystem for Linux | ``apt install ocrmypdf`` | | Windows Subsystem for Linux | ``apt install ocrmypdf`` |
+-----------------------------+-------------------------------+ +-------------------------------+-------------------------------+
| Fedora | ``dnf install ocrmypdf`` | | Fedora | ``dnf install ocrmypdf`` |
+-----------------------------+-------------------------------+ +-------------------------------+-------------------------------+
| macOS | ``brew install ocrmypdf`` | | macOS | ``brew install ocrmypdf`` |
+-----------------------------+-------------------------------+ +-------------------------------+-------------------------------+
| LinuxBrew | ``brew install ocrmypdf`` | | LinuxBrew | ``brew install ocrmypdf`` |
+-----------------------------+-------------------------------+ +-------------------------------+-------------------------------+
| FreeBSD | ``pkg install py37-ocrmypdf`` | | FreeBSD | ``pkg install py37-ocrmypdf`` |
+-----------------------------+-------------------------------+ +-------------------------------+-------------------------------+
| Conda (WSL, macOS, Linux) | ``conda install ocrmypdf`` |
+-------------------------------+-------------------------------+
More detailed procedures are outlined below. If you want to do a manual More detailed procedures are outlined below. If you want to do a manual
install, or install a more recent version than your platform provides, read on. install, or install a more recent version than your platform provides, read on.
@@ -54,6 +56,9 @@ Debian and Ubuntu 18.04 or newer
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg .. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
:alt: Ubuntu 20.04 LTS :alt: Ubuntu 20.04 LTS
.. |ubu-2010| image:: https://repology.org/badge/version-for-repo/ubuntu_20_10/ocrmypdf.svg
:alt: Ubuntu 20.10
+-----------------------------------------------+ +-----------------------------------------------+
| **OCRmyPDF versions in Debian & Ubuntu** | | **OCRmyPDF versions in Debian & Ubuntu** |
+-----------------------------------------------+ +-----------------------------------------------+
@@ -61,7 +66,7 @@ Debian and Ubuntu 18.04 or newer
+-----------------------------------------------+ +-----------------------------------------------+
| |deb-stable| |deb-testing| |deb-unstable| | | |deb-stable| |deb-testing| |deb-unstable| |
+-----------------------------------------------+ +-----------------------------------------------+
| |ubu-1804| |ubu-2004| | | |ubu-1804| |ubu-2004| |ubu-2010| |
+-----------------------------------------------+ +-----------------------------------------------+
Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users Users of Debian 9 ("stretch") or later, or Ubuntu 18.04 or later, including users
@@ -90,15 +95,15 @@ For full details on version availability for your platform, check the
automatically detect it (specifically the ``jbig2`` binary) on the automatically detect it (specifically the ``jbig2`` binary) on the
``PATH``. To add JBIG2 encoding, see :ref:`jbig2`. ``PATH``. To add JBIG2 encoding, see :ref:`jbig2`.
Fedora 29 or newer Fedora
------------------ ------
.. |fedora-31| image:: https://repology.org/badge/version-for-repo/fedora_31/ocrmypdf.svg
:alt: Fedora 31
.. |fedora-32| image:: https://repology.org/badge/version-for-repo/fedora_32/ocrmypdf.svg .. |fedora-32| image:: https://repology.org/badge/version-for-repo/fedora_32/ocrmypdf.svg
:alt: Fedora 32 :alt: Fedora 32
.. |fedora-33| image:: https://repology.org/badge/version-for-repo/fedora_33/ocrmypdf.svg
:alt: Fedora 33
.. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg .. |fedora-rawhide| image:: https://repology.org/badge/version-for-repo/fedora_rawhide/ocrmypdf.svg
:alt: Fedore Rawhide :alt: Fedore Rawhide
@@ -107,7 +112,7 @@ Fedora 29 or newer
+-----------------------------------------------+ +-----------------------------------------------+
| |latest| | | |latest| |
+-----------------------------------------------+ +-----------------------------------------------+
| |fedora-31| |fedora-32| |fedora-rawhide| | | |fedora-32| |fedora-33| |fedora-rawhide| |
+-----------------------------------------------+ +-----------------------------------------------+
Users of Fedora 29 or later may simply Users of Fedora 29 or later may simply
@@ -355,7 +360,8 @@ To install OCRmyPDF for Alpine Linux:
Mageia 7 Mageia 7
-------- --------
Install the following dependencies: There is no OS-level packaging available for Mageia, so you must install the
dependencies:
.. code-block:: bash .. code-block:: bash
+33
View File
@@ -12,6 +12,39 @@ may be unreliable. Use the API to depend on precise behavior.
The public API may be useful in scripts that launch OCRmyPDF processes or that The public API may be useful in scripts that launch OCRmyPDF processes or that
wish to use some of its features for working with PDFs. wish to use some of its features for working with PDFs.
v11.7.3
=======
- Exclude CCITT Group 3 images from being optimized. Some libraries
OCRmyPDF uses do not seem to handle this obscure compression format properly.
You may get errors or possible corrupted output images without this fix.
v11.7.2
=======
- Updated pinned versions in main.txt, primarily to upgrade Pillow to 8.1.2, due
to recently disclosed security vulnerabilities in that software.
- The ``--sidecar`` parameter now causes an exception if set to the same file as
the input or output PDF.
v11.7.1
=======
- Some exceptions while attempting image optimization were only logged at the debug
level, causing them to be suppressed. These errors are now logged appropriately.
- Improved the error message related to ``--unpaper-args``.
- Updated documentation to mention the new conda distribution.
v11.7.0
=======
- We now support using ``--sidecar`` in conjunction with ``--pages``; these arguments
used to be mutually exclusive. (#735)
- Fixed a possible issue with PDF/A-1b generation. Acrobat complained that our PDFs use
object streams. More robust PDF/A validators like veraPDF don't consider this a
problem, but we'll honor Acrobat's objection from here on. This may increase file
size of PDF/A-1b files. PDF/A-2b files will not be affected.
v11.6.2 v11.6.2
======= =======
+6 -6
View File
@@ -1,12 +1,12 @@
# requirements.txt can be used to replicate the developer's build environment # requirements.txt can be used to replicate the developer's build environment
# setup.py lists a separate set of requirements that are looser to simplify # setup.py lists a separate set of requirements that are looser to simplify
# installation # installation
cffi == 1.14.3 cffi == 1.14.5
coloredlogs == 14.0 # technically optional coloredlogs == 15.0 # technically optional
img2pdf == 0.4.0 img2pdf == 0.4.0
pdfminer.six == 20201018 pdfminer.six == 20201018
pikepdf == 2.0.0 pikepdf == 2.9.0
pluggy == 0.13.1 pluggy == 0.13.1
Pillow == 8.0.1 Pillow == 8.1.2
reportlab == 3.5.55 reportlab == 3.5.65
tqdm == 4.51.0 tqdm == 4.59.0
+3 -3
View File
@@ -1,7 +1,7 @@
pytest >= 5.0.0 pytest >= 6.0.0
pytest-helpers-namespace >= 2019.1.8 pytest-helpers-namespace >= 2019.1.8
pytest-xdist >= 1.31.0 pytest-xdist >= 2.2.0
pytest-cov >= 2.10.0 pytest-cov >= 2.11.1
python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3 python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3
# or brew install exempi # or brew install exempi
#PyMuPDF == 1.13.4 # optional #PyMuPDF == 1.13.4 # optional
+1 -1
View File
@@ -1 +1 @@
watchdog == 0.10.2 watchdog == 1.0.2
+43 -9
View File
@@ -739,6 +739,24 @@ def should_linearize(working_file: Path, context: PdfContext):
return False return False
def get_pdf_save_settings(output_type: str):
if output_type == 'pdfa-1':
# Trigger recompression to ensure object streams are removed, because
# Acrobat complains about them in PDF/A-1b validation.
return dict(
preserve_pdfa=True,
compress_streams=True,
stream_decode_level=pikepdf.StreamDecodeLevel.generalized,
object_stream_mode=pikepdf.ObjectStreamMode.disable,
)
else:
return dict(
preserve_pdfa=True,
compress_streams=True,
object_stream_mode=(pikepdf.ObjectStreamMode.generate),
)
def metadata_fixup(working_file: Path, context: PdfContext): def metadata_fixup(working_file: Path, context: PdfContext):
output_file = context.get_path('metafix.pdf') output_file = context.get_path('metafix.pdf')
options = context.options options = context.options
@@ -783,9 +801,7 @@ def metadata_fixup(working_file: Path, context: PdfContext):
pdf.save( pdf.save(
output_file, output_file,
compress_streams=True, **get_pdf_save_settings(options.output_type),
preserve_pdfa=True,
object_stream_mode=pikepdf.ObjectStreamMode.generate,
linearize=( # Don't linearize if optimize() will be linearizing too linearize=( # Don't linearize if optimize() will be linearizing too
should_linearize(working_file, context) should_linearize(working_file, context)
if options.optimize == 0 if options.optimize == 0
@@ -799,20 +815,34 @@ def metadata_fixup(working_file: Path, context: PdfContext):
def optimize_pdf(input_file: Path, context: PdfContext): def optimize_pdf(input_file: Path, context: PdfContext):
output_file = context.get_path('optimize.pdf') output_file = context.get_path('optimize.pdf')
save_settings = dict( save_settings = dict(
compress_streams=True,
preserve_pdfa=True,
object_stream_mode=pikepdf.ObjectStreamMode.generate,
linearize=should_linearize(input_file, context), linearize=should_linearize(input_file, context),
**get_pdf_save_settings(context.options.output_type),
) )
optimize(input_file, output_file, context, save_settings) optimize(input_file, output_file, context, save_settings)
return output_file return output_file
def enumerate_compress_ranges(iterable):
skipped_from = None
for index, txt_file in enumerate(iterable):
index += 1
if txt_file:
if skipped_from is not None:
yield (skipped_from, index - 1), None
skipped_from = None
yield (index, index), txt_file
else:
if skipped_from is None:
skipped_from = index
if skipped_from is not None:
yield (skipped_from, index), None
def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext): def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
output_file = context.get_path('sidecar.txt') output_file = context.get_path('sidecar.txt')
with open(output_file, 'w', encoding="utf-8") as stream: with open(output_file, 'w', encoding="utf-8") as stream:
for page_num, txt_file in enumerate(txt_files): for (frm, to), txt_file in enumerate_compress_ranges(txt_files):
if page_num != 0: if frm != 1:
stream.write('\f') # Form feed between pages stream.write('\f') # Form feed between pages
if txt_file: if txt_file:
with open(txt_file, 'r', encoding="utf-8") as in_: with open(txt_file, 'r', encoding="utf-8") as in_:
@@ -825,7 +855,11 @@ def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
else: else:
stream.write(txt) stream.write(txt)
else: else:
stream.write(f'[OCR skipped on page {(page_num + 1)}]') if frm != to:
pages = f'{frm}-{to}'
else:
pages = f'{frm}'
stream.write(f'[OCR skipped on page(s) {pages}]')
return output_file return output_file
+5 -3
View File
@@ -112,6 +112,10 @@ def check_options_sidecar(options):
"--sidecar filename must be specified when output file is stdout." "--sidecar filename must be specified when output file is stdout."
) )
options.sidecar = options.output_file + '.txt' options.sidecar = options.output_file + '.txt'
if options.sidecar == options.input_file or options.sidecar == options.output_file:
raise BadArgsError(
"--sidecar file must be different from the input and output files"
)
def check_options_preprocessing(options): def check_options_preprocessing(options):
@@ -133,7 +137,7 @@ def check_options_preprocessing(options):
options.unpaper_args options.unpaper_args
) )
except Exception as e: except Exception as e:
raise BadArgsError(str(e)) raise BadArgsError("--unpaper-args: " + str(e)) from e
def _pages_from_ranges(ranges: str) -> Set[int]: def _pages_from_ranges(ranges: str) -> Set[int]:
@@ -184,8 +188,6 @@ def check_options_ocr_behavior(options):
) )
if exclusive_options >= 2: if exclusive_options >= 2:
raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.") raise BadArgsError("Choose only one of --force-ocr, --skip-text, --redo-ocr.")
if options.pages and options.sidecar:
raise BadArgsError("--pages and --sidecar are mutually exclusive")
if options.pages: if options.pages:
options.pages = _pages_from_ranges(options.pages) options.pages = _pages_from_ranges(options.pages)
+2 -1
View File
@@ -167,7 +167,8 @@ Online documentation is located at:
metavar='FILE', metavar='FILE',
help="Generate sidecar text files that contain the same text recognized " help="Generate sidecar text files that contain the same text recognized "
"by Tesseract. This may be useful for building a OCR text database. " "by Tesseract. This may be useful for building a OCR text database. "
"If FILE is omitted, the sidecar file be named {output_file}.txt " "If FILE is omitted, the sidecar file be named {output_file}.txt; the next "
"argument must NOT be the name of the input PDF. "
"If FILE is set to '-', the sidecar is written to stdout (a " "If FILE is set to '-', the sidecar is written to stdout (a "
"convenient way to preview OCR quality). The output file and sidecar " "convenient way to preview OCR quality). The output file and sidecar "
"may not both use stdout at the same time.", "may not both use stdout at the same time.",
+28 -31
View File
@@ -181,45 +181,42 @@ def check_pdf(input_file: Path) -> bool:
Checks for proper formatting and proper linearization. Uses pikepdf (which in Checks for proper formatting and proper linearization. Uses pikepdf (which in
turn, uses QPDF) to perform the checks. turn, uses QPDF) to perform the checks.
""" """
pdf = None
try: try:
pdf = pikepdf.open(input_file) pdf = pikepdf.open(input_file)
except pikepdf.PdfError as e: except pikepdf.PdfError as e:
log.error(e) log.error(e)
return False return False
else: else:
messages = pdf.check() with pdf:
for msg in messages: messages = pdf.check()
if 'error' in msg.lower(): for msg in messages:
log.error(msg) if 'error' in msg.lower():
log.error(msg)
else:
log.warning(msg)
sio = StringIO()
linearize_msgs = ''
try:
# If linearization is missing entirely, we do not complain. We do
# complain if linearization is present but incorrect.
pdf.check_linearization(sio)
except RuntimeError:
pass
except ( # Workaround for a problematic pikepdf version
getattr(pikepdf, 'ForeignObjectError')
if pikepdf.__version__ == '2.1.0'
else NeverRaise
):
pass
else: else:
log.warning(msg) linearize_msgs = sio.getvalue()
if linearize_msgs:
log.warning(linearize_msgs)
sio = StringIO() if not messages and not linearize_msgs:
linearize_msgs = '' return True
try: return False
# If linearization is missing entirely, we do not complain. We do
# complain if linearization is present but incorrect.
pdf.check_linearization(sio)
except RuntimeError:
pass
except (
getattr(pikepdf, 'ForeignObjectError')
if pikepdf.__version__ == '2.1.0' # This version may throw wrong exception
else NeverRaise
):
pass
else:
linearize_msgs = sio.getvalue()
if linearize_msgs:
log.warning(linearize_msgs)
if not messages and not linearize_msgs:
return True
return False
finally:
if pdf:
pdf.close()
def clamp(n, smallest, largest): # mypy doesn't understand types for this def clamp(n, smallest, largest): # mypy doesn't understand types for this
+6 -2
View File
@@ -95,6 +95,10 @@ def extract_image_filter(
log.debug(f"Skipping JPEG2000 iamge, xref {xref}") log.debug(f"Skipping JPEG2000 iamge, xref {xref}")
return None # Don't do JPEG2000 return None # Don't do JPEG2000
if filtdp[0] == Name.CCITTFaxDecode and filtdp[1].get('/K', 0) >= 0:
log.debug(f"Skipping CCITT Group 3 image, xref {xref}")
return None # pikepdf doesn't support Group 3 yet
if Name.Decode in image: if Name.Decode in image:
log.debug(f"Skipping image with Decode table, xref {xref}") log.debug(f"Skipping image with Decode table, xref {xref}")
return None # Don't mess with custom Decode tables return None # Don't mess with custom Decode tables
@@ -258,8 +262,8 @@ def extract_images(
result = extract_fn( result = extract_fn(
pike=pike, root=root, image=image, xref=xref, options=options pike=pike, root=root, image=image, xref=xref, options=options
) )
except Exception as e: # pylint: disable=broad-except except Exception: # pylint: disable=broad-except
log.debug("Image xref %s, error %s", xref, repr(e)) log.exception(f"While extracting image xref {xref}, an error occurred")
errors += 1 errors += 1
else: else:
if result: if result:
+13
View File
@@ -185,3 +185,16 @@ def test_optimize_off(resources, outpdf):
'--plugin', '--plugin',
'tests/plugins/tesseract_noop.py', 'tests/plugins/tesseract_noop.py',
) )
def test_group3(resources, outdir):
with pikepdf.open(resources / 'ccitt.pdf') as pdf:
im = pdf.pages[0].Resources.XObject['/Im1']
assert (
opt.extract_image_filter(pdf, outdir, im, im.objgen[0]) is not None
), "Group 4 should be allowed"
im.DecodeParms['/K'] = 0
assert (
opt.extract_image_filter(pdf, outdir, im, im.objgen[0]) is None
), "Group 3 should be disallowed"
+34
View File
@@ -0,0 +1,34 @@
# © 2021 James R. Barlow: github.com/jbarlow83
#
# This Source Code Form is subject to the terms of the Mozilla Public
# License, v. 2.0. If a copy of the MPL was not distributed with this
# file, You can obtain one at http://mozilla.org/MPL/2.0/.
import pikepdf
import pytest
check_ocrmypdf = pytest.helpers.check_ocrmypdf
@pytest.mark.parametrize('optimize', (0, 3))
@pytest.mark.parametrize('pdfa_level', (1, 2, 3))
def test_pdfa(resources, outpdf, optimize, pdfa_level):
check_ocrmypdf(
resources / 'francais.pdf',
outpdf,
'--plugin',
'tests/plugins/tesseract_noop.py',
f'--output-type=pdfa-{pdfa_level}',
f'--optimize={optimize}',
)
if pdfa_level in (2, 3):
# PDF/A-2 allows ObjStm
assert b'/ObjStm' in outpdf.read_bytes()
elif pdfa_level == 1:
# PDF/A-1 might allow ObjStm, but Acrobat does not approve it, so
# we don't use it
assert b'/ObjStm' not in outpdf.read_bytes()
with pikepdf.open(outpdf) as pdf:
with pdf.open_metadata() as m:
assert m.pdfa_status == f'{pdfa_level}B'
+87
View File
@@ -61,3 +61,90 @@ def test_dpi_needed(image, text, vector, result, rgb_image, outdir):
assert _pipeline.get_canvas_square_dpi(pi[0], mock) == result assert _pipeline.get_canvas_square_dpi(pi[0], mock) == result
assert _pipeline.get_page_square_dpi(pi[0], mock) == result assert _pipeline.get_page_square_dpi(pi[0], mock) == result
@pytest.mark.parametrize(
# Name for nicer -v output
'name,input,output',
(
(
'empty_input',
# Input:
(),
# Output:
(),
),
(
'no_values',
# Input:
('', '', '', '', ''),
# Output:
(
((1, 5), None),
),
),
(
'no_empty_values',
# Input:
('v', 'w', 'x', 'y', 'z'),
# Output:
(
((1, 1), 'v'),
((2, 2), 'w'),
((3, 3), 'x'),
((4, 4), 'y'),
((5, 5), 'z'),
),
),
(
'skip_head',
# Input:
('', '', 'x', 'y', 'z'),
# Output:
(
((1, 2), None),
((3, 3), 'x'),
((4, 4), 'y'),
((5, 5), 'z'),
),
),
(
'skip_tail',
# Input:
('x', 'y', 'z', '', ''),
# Output:
(
((1, 1), 'x'),
((2, 2), 'y'),
((3, 3), 'z'),
((4, 5), None),
),
),
(
'range_in_middle',
# Input:
('x', '', '', '', 'y'),
# Output:
(
((1, 1), 'x'),
((2, 4), None),
((5, 5), 'y'),
),
),
(
'range_in_middle_2',
# Input:
('x', '', '', 'y', '', '', '', 'z'),
# Output:
(
((1, 1), 'x'),
((2, 3), None),
((4, 4), 'y'),
((5, 7), None),
((8, 8), 'z'),
),
),
),
)
def test_enumerate_compress_ranges(name, input, output):
assert output == tuple(_pipeline.enumerate_compress_ranges(input))
+8 -2
View File
@@ -18,6 +18,8 @@ from ocrmypdf.cli import get_parser
from ocrmypdf.exceptions import BadArgsError, MissingDependencyError from ocrmypdf.exceptions import BadArgsError, MissingDependencyError
from ocrmypdf.pdfinfo import PdfInfo from ocrmypdf.pdfinfo import PdfInfo
run_ocrmypdf_api = pytest.helpers.run_ocrmypdf_api
def make_opts_pm(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs): def make_opts_pm(input_file='a.pdf', output_file='b.pdf', language='eng', **kwargs):
if language is not None: if language is not None:
@@ -90,8 +92,6 @@ def test_mutex_options():
vd.check_options_ocr_behavior(make_opts(redo_ocr=True, skip_text=True)) vd.check_options_ocr_behavior(make_opts(redo_ocr=True, skip_text=True))
with pytest.raises(BadArgsError): with pytest.raises(BadArgsError):
vd.check_options_ocr_behavior(make_opts(redo_ocr=True, force_ocr=True)) vd.check_options_ocr_behavior(make_opts(redo_ocr=True, force_ocr=True))
with pytest.raises(BadArgsError):
vd.check_options_ocr_behavior(make_opts(pages='1-3', sidecar='file.txt'))
def test_optimizing(caplog): def test_optimizing(caplog):
@@ -272,3 +272,9 @@ def test_two_languages():
*make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'} *make_opts_pm(language='fakelang1+fakelang2'), {'fakelang1', 'fakelang2'}
) )
mock.assert_called() mock.assert_called()
def test_sidecar_equals_output(resources, no_outpdf):
op = no_outpdf
with pytest.raises(BadArgsError, match=r'--sidecar'):
run_ocrmypdf_api(resources / 'trivial.pdf', op, '--sidecar', op)