Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
250615561d | ||
|
|
a659f83d67 | ||
|
|
08f95c0b13 | ||
|
|
dbd3c93757 | ||
|
|
5d128a91d2 | ||
|
|
a1b8113d56 | ||
|
|
f052e910c9 | ||
|
|
116e2692d0 | ||
|
|
b2669c7d71 | ||
|
|
c8c53d38a3 |
+6
-2
@@ -1,7 +1,7 @@
|
|||||||
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
# SPDX-License-Identifier: MPL-2.0
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
FROM ubuntu:24.04 AS base
|
FROM ubuntu:22.04 AS base
|
||||||
|
|
||||||
ENV LANG=C.UTF-8
|
ENV LANG=C.UTF-8
|
||||||
ENV TZ=UTC
|
ENV TZ=UTC
|
||||||
@@ -51,7 +51,7 @@ FROM base
|
|||||||
|
|
||||||
RUN apt-get update && apt-get install -y software-properties-common
|
RUN apt-get update && apt-get install -y software-properties-common
|
||||||
|
|
||||||
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr-devel
|
RUN add-apt-repository -y ppa:alex-p/tesseract-ocr5
|
||||||
|
|
||||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||||
ghostscript \
|
ghostscript \
|
||||||
@@ -75,6 +75,10 @@ COPY --from=builder /usr/local/bin/ /usr/local/bin/
|
|||||||
|
|
||||||
COPY --from=builder --chown=app:app /app /app
|
COPY --from=builder --chown=app:app /app /app
|
||||||
|
|
||||||
|
RUN rm -rf /app/.git && \
|
||||||
|
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||||
|
ln -s /app/misc/watcher.py /app/watcher.py
|
||||||
|
|
||||||
ENV PATH="/app/.venv/bin:${PATH}"
|
ENV PATH="/app/.venv/bin:${PATH}"
|
||||||
|
|
||||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||||
|
|||||||
@@ -59,6 +59,10 @@ WORKDIR /app
|
|||||||
|
|
||||||
COPY --from=builder --chown=app:app /app /app
|
COPY --from=builder --chown=app:app /app /app
|
||||||
|
|
||||||
|
RUN rm -rf /app/.git && \
|
||||||
|
ln -s /app/misc/webservice.py /app/webservice.py && \
|
||||||
|
ln -s /app/misc/watcher.py /app/watcher.py
|
||||||
|
|
||||||
ENV PATH="/app/.venv/bin:${PATH}"
|
ENV PATH="/app/.venv/bin:${PATH}"
|
||||||
|
|
||||||
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
ENTRYPOINT ["/app/.venv/bin/ocrmypdf"]
|
||||||
|
|||||||
@@ -50,9 +50,9 @@ jobs:
|
|||||||
python-version: ${{ matrix.python }}
|
python-version: ${{ matrix.python }}
|
||||||
|
|
||||||
- name: Install Tesseract from PPA
|
- name: Install Tesseract from PPA
|
||||||
if: matrix.tesseract_ppa
|
if: matrix.tesseract_ppa == 'ppa'
|
||||||
run: |
|
run: |
|
||||||
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr-devel
|
sudo add-apt-repository -y ppa:alex-p/tesseract-ocr5.3
|
||||||
|
|
||||||
- name: Install common packages
|
- name: Install common packages
|
||||||
run: |
|
run: |
|
||||||
@@ -106,7 +106,6 @@ jobs:
|
|||||||
files: ./coverage.xml
|
files: ./coverage.xml
|
||||||
env_vars: OS,PYTHON
|
env_vars: OS,PYTHON
|
||||||
|
|
||||||
|
|
||||||
test_macos:
|
test_macos:
|
||||||
name: Test macOS
|
name: Test macOS
|
||||||
runs-on: ${{ matrix.os }}
|
runs-on: ${{ matrix.os }}
|
||||||
|
|||||||
+1
-1
@@ -44,7 +44,7 @@ place, and printing each filename in between runs:
|
|||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
find . -printf '%p\n' -name '*.pdf' -exec ocrmypdf '{}' '{}' \;
|
find . -name '*.pdf' -printf '%p\n' -exec ocrmypdf '{}' '{}' \;
|
||||||
|
|
||||||
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
This only runs one ``ocrmypdf`` process at a time. This variation uses
|
||||||
``find`` to create a directory list and ``parallel`` to parallelize runs
|
``find`` to create a directory list and ``parallel`` to parallelize runs
|
||||||
|
|||||||
@@ -30,6 +30,22 @@ OCRmyPDF typically supports the three most recent Python versions.
|
|||||||
|
|
||||||
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||||
|
|
||||||
|
v16.6.2
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Remove invalid hyperlink annotations to satisfy Ghostscript 10.x during PDF/A
|
||||||
|
conversion. :issue:`1425`
|
||||||
|
|
||||||
|
v16.6.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed some issues with Docker build, such as removing unnecessary content and using
|
||||||
|
a stable Tesseract version.
|
||||||
|
- Reverted Docker image to Ubuntu 22.04 to access older/more stable Ghostscript
|
||||||
|
for now.
|
||||||
|
- Clarified batch commands in documentation.
|
||||||
|
- Fixed an issue with JSON serialization and pickling of HOCRResult. :issue:`1427`
|
||||||
|
|
||||||
v16.6.0
|
v16.6.0
|
||||||
=======
|
=======
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,42 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
"""Helper script for bisecting PDFs to find a page with an issue."""
|
||||||
|
|
||||||
|
import sys
|
||||||
|
|
||||||
|
import pikepdf
|
||||||
|
|
||||||
|
if len(sys.argv) != 2:
|
||||||
|
print(f"Usage: {sys.argv[0]} <input.pdf>")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
with pikepdf.open(sys.argv[1]) as pdf:
|
||||||
|
num_pages = len(pdf.pages)
|
||||||
|
low = 0
|
||||||
|
high = num_pages - 1
|
||||||
|
while low <= high:
|
||||||
|
mid = (low + high) // 2
|
||||||
|
with pikepdf.new() as new_pdf:
|
||||||
|
new_pdf.pages.extend(pdf.pages[low : mid + 1])
|
||||||
|
new_pdf.save(f"bisect-issue-{low + 1}-{mid + 1}.pdf")
|
||||||
|
print(f"Is bisect-issue-{low + 1}-{mid + 1}.pdf good or bad?", end=" ")
|
||||||
|
while True:
|
||||||
|
response = input().lower()
|
||||||
|
if response == "good":
|
||||||
|
low = mid + 1
|
||||||
|
break
|
||||||
|
elif response == "bad":
|
||||||
|
high = mid - 1
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
print("Please respond with 'good' or 'bad'.")
|
||||||
|
print(f"The issue is on page {low + 1} of the original PDF.")
|
||||||
|
with pikepdf.new() as new_pdf:
|
||||||
|
new_pdf.pages.extend(pdf.pages[low])
|
||||||
|
new_pdf.save(f"bisect-issue-bad-{low + 1}.pdf")
|
||||||
|
with pikepdf.new() as new_pdf:
|
||||||
|
new_pdf.pages.extend(pdf.pages[:low])
|
||||||
|
new_pdf.pages.extend(pdf.pages[low + 1 :])
|
||||||
|
new_pdf.save(f"bisect-issue-good-{low + 1}.pdf")
|
||||||
@@ -155,3 +155,6 @@ convention = "google"
|
|||||||
|
|
||||||
[tool.ruff.format]
|
[tool.ruff.format]
|
||||||
quote-style = "preserve"
|
quote-style = "preserve"
|
||||||
|
|
||||||
|
[dependency-groups]
|
||||||
|
dev = ["mypy>=1.13.0"]
|
||||||
|
|||||||
@@ -0,0 +1,66 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
"""OCRmyPDF PDF annotation cleanup."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from pikepdf import Dictionary, Name, NameTree, Pdf
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
def remove_broken_goto_annotations(pdf: Pdf) -> bool:
|
||||||
|
"""Remove broken goto annotations from a PDF.
|
||||||
|
|
||||||
|
If a PDF contains a GoTo Action that points to a named destination that does not
|
||||||
|
exist, Ghostscript PDF/A conversion will fail. In any event, a named destination
|
||||||
|
that is not defined is not useful.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
pdf: Opened PDF file.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: True if the file was modified, False if not.
|
||||||
|
"""
|
||||||
|
modified = False
|
||||||
|
|
||||||
|
# Check if there are any named destinations
|
||||||
|
if Name.Names not in pdf.Root:
|
||||||
|
return modified
|
||||||
|
if Name.Dests not in pdf.Root[Name.Names]:
|
||||||
|
return modified
|
||||||
|
|
||||||
|
dests = pdf.Root[Name.Names][Name.Dests]
|
||||||
|
if not isinstance(dests, Dictionary):
|
||||||
|
return modified
|
||||||
|
nametree = NameTree(dests)
|
||||||
|
|
||||||
|
# Create a set of all named destinations
|
||||||
|
names = set(k for k in nametree.keys())
|
||||||
|
|
||||||
|
for n, page in enumerate(pdf.pages):
|
||||||
|
if Name.Annots not in page:
|
||||||
|
continue
|
||||||
|
for annot in page[Name.Annots]:
|
||||||
|
if not isinstance(annot, Dictionary):
|
||||||
|
continue
|
||||||
|
if Name.A not in annot or Name.D not in annot[Name.A]:
|
||||||
|
continue
|
||||||
|
# We found an annotation that points to a named destination
|
||||||
|
named_destination = str(annot[Name.A][Name.D])
|
||||||
|
if named_destination not in names:
|
||||||
|
# If there is no corresponding named destination, remove the
|
||||||
|
# annotation. Having no destination set is still valid and just
|
||||||
|
# makes the link non-functional.
|
||||||
|
log.warning(
|
||||||
|
f"Disabling a hyperlink annotation on page {n + 1} to a "
|
||||||
|
"non-existent named destination "
|
||||||
|
f"{named_destination}."
|
||||||
|
)
|
||||||
|
del annot[Name.A][Name.D]
|
||||||
|
modified = True
|
||||||
|
|
||||||
|
return modified
|
||||||
@@ -15,6 +15,7 @@ from pikepdf import Dictionary, Name, Pdf
|
|||||||
from pikepdf import __version__ as PIKEPDF_VERSION
|
from pikepdf import __version__ as PIKEPDF_VERSION
|
||||||
from pikepdf.models.metadata import PdfMetadata, encode_pdf_date
|
from pikepdf.models.metadata import PdfMetadata, encode_pdf_date
|
||||||
|
|
||||||
|
from ocrmypdf._annots import remove_broken_goto_annotations
|
||||||
from ocrmypdf._defaults import PROGRAM_NAME
|
from ocrmypdf._defaults import PROGRAM_NAME
|
||||||
from ocrmypdf._jobcontext import PdfContext
|
from ocrmypdf._jobcontext import PdfContext
|
||||||
from ocrmypdf._version import __version__ as OCRMYPF_VERSION
|
from ocrmypdf._version import __version__ as OCRMYPF_VERSION
|
||||||
|
|||||||
@@ -20,7 +20,9 @@ from pathlib import Path
|
|||||||
from typing import NamedTuple, cast
|
from typing import NamedTuple, cast
|
||||||
|
|
||||||
import PIL
|
import PIL
|
||||||
|
from pikepdf import Pdf
|
||||||
|
|
||||||
|
from ocrmypdf._annots import remove_broken_goto_annotations
|
||||||
from ocrmypdf._concurrent import Executor, setup_executor
|
from ocrmypdf._concurrent import Executor, setup_executor
|
||||||
from ocrmypdf._jobcontext import PageContext, PdfContext
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from ocrmypdf._logging import PageNumberFilter
|
from ocrmypdf._logging import PageNumberFilter
|
||||||
@@ -104,6 +106,23 @@ class PageResult(NamedTuple):
|
|||||||
"""Orientation correction in degrees."""
|
"""Orientation correction in degrees."""
|
||||||
|
|
||||||
|
|
||||||
|
class HOCRResultEncoder(json.JSONEncoder):
|
||||||
|
def default(self, obj):
|
||||||
|
if isinstance(obj, Path):
|
||||||
|
return {'Path': str(obj)}
|
||||||
|
return super().default(obj)
|
||||||
|
|
||||||
|
|
||||||
|
class HOCRResultDecoder(json.JSONDecoder):
|
||||||
|
def __init__(self, *args, **kwargs):
|
||||||
|
super().__init__(object_hook=self.dict_to_object, *args, **kwargs)
|
||||||
|
|
||||||
|
def dict_to_object(self, d):
|
||||||
|
if 'Path' in d:
|
||||||
|
return Path(d['Path'])
|
||||||
|
return d
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
class HOCRResult:
|
class HOCRResult:
|
||||||
"""Result when hOCR is finished processing."""
|
"""Result when hOCR is finished processing."""
|
||||||
@@ -123,38 +142,14 @@ class HOCRResult:
|
|||||||
orientation_correction: int = 0
|
orientation_correction: int = 0
|
||||||
"""Orientation correction in degrees."""
|
"""Orientation correction in degrees."""
|
||||||
|
|
||||||
def __getstate__(self):
|
|
||||||
"""Return state values to be pickled."""
|
|
||||||
return {
|
|
||||||
k: (
|
|
||||||
('Path://' + str(v))
|
|
||||||
if k in ('pdf_page_from_image', 'hocr', 'textpdf') and v is not None
|
|
||||||
else v
|
|
||||||
)
|
|
||||||
for k, v in self.__dict__.items()
|
|
||||||
}
|
|
||||||
|
|
||||||
def __setstate__(self, state):
|
|
||||||
"""Restore state from the unpickled state values."""
|
|
||||||
self.__dict__.update(
|
|
||||||
{
|
|
||||||
k: (
|
|
||||||
Path(v.removeprefix('Path://'))
|
|
||||||
if k in ('pdf_page_from_image', 'hocr', 'textpdf') and v is not None
|
|
||||||
else v
|
|
||||||
)
|
|
||||||
for k, v in state.items()
|
|
||||||
}
|
|
||||||
)
|
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def from_json(cls, json_str: str) -> HOCRResult:
|
def from_json(cls, json_str: str) -> HOCRResult:
|
||||||
"""Create an instance from a dict."""
|
"""Create an instance from a dict."""
|
||||||
return cls(**json.loads(json_str))
|
return cls(**json.loads(json_str, cls=HOCRResultDecoder))
|
||||||
|
|
||||||
def to_json(self) -> str:
|
def to_json(self) -> str:
|
||||||
"""Serialize to a JSON string."""
|
"""Serialize to a JSON string."""
|
||||||
return json.dumps(self.__getstate__())
|
return json.dumps(self.__dict__, cls=HOCRResultEncoder)
|
||||||
|
|
||||||
|
|
||||||
def configure_debug_logging(
|
def configure_debug_logging(
|
||||||
@@ -445,7 +440,14 @@ def postprocess(
|
|||||||
pdf_file: Path, context: PdfContext, executor: Executor
|
pdf_file: Path, context: PdfContext, executor: Executor
|
||||||
) -> tuple[Path, Sequence[str]]:
|
) -> tuple[Path, Sequence[str]]:
|
||||||
"""Postprocess the PDF file."""
|
"""Postprocess the PDF file."""
|
||||||
pdf_out = pdf_file
|
# pdf_out = pdf_file
|
||||||
|
with Pdf.open(pdf_file) as pdf:
|
||||||
|
fix_annots = context.get_path('fix_annots.pdf')
|
||||||
|
if remove_broken_goto_annotations(pdf):
|
||||||
|
pdf.save(fix_annots)
|
||||||
|
pdf_out = fix_annots
|
||||||
|
else:
|
||||||
|
pdf_out = pdf_file
|
||||||
if context.options.output_type.startswith('pdfa'):
|
if context.options.output_type.startswith('pdfa'):
|
||||||
ps_stub_out = generate_postscript_stub(context)
|
ps_stub_out = generate_postscript_stub(context)
|
||||||
pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context)
|
pdf_out = convert_to_pdfa(pdf_out, ps_stub_out, context)
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
# SPDX-FileCopyrightText: 2024 James R. Barlow
|
||||||
|
# SPDX-License-Identifier: MPL-2.0
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from pikepdf import Array, Dictionary, Name, NameTree, Pdf
|
||||||
|
|
||||||
|
from ocrmypdf._annots import remove_broken_goto_annotations
|
||||||
|
|
||||||
|
|
||||||
|
def test_remove_broken_goto_annotations(resources):
|
||||||
|
with Pdf.open(resources / 'link.pdf') as pdf:
|
||||||
|
assert not remove_broken_goto_annotations(pdf), "File should not be modified"
|
||||||
|
|
||||||
|
# Construct Dests nametree
|
||||||
|
nt = NameTree.new(pdf)
|
||||||
|
names = pdf.Root[Name.Names] = pdf.make_indirect(Dictionary())
|
||||||
|
names[Name.Dests] = nt.obj
|
||||||
|
# Create a broken named destination
|
||||||
|
nt['Invalid'] = pdf.make_indirect(Dictionary())
|
||||||
|
# Create a valid named destination
|
||||||
|
nt['Valid'] = Array([pdf.pages[0].obj, Name.XYZ, 0, 0, 0])
|
||||||
|
|
||||||
|
pdf.pages[0].Annots[0].A.D = 'Missing'
|
||||||
|
pdf.pages[1].Annots[0].A.D = 'Valid'
|
||||||
|
|
||||||
|
assert remove_broken_goto_annotations(pdf), "File should be modified"
|
||||||
|
|
||||||
|
assert Name.D not in pdf.pages[0].Annots[0].A
|
||||||
|
assert Name.D in pdf.pages[1].Annots[0].A
|
||||||
+29
-1
@@ -3,6 +3,7 @@
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import pickle
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -10,6 +11,7 @@ import pytest
|
|||||||
from pdfminer.high_level import extract_text
|
from pdfminer.high_level import extract_text
|
||||||
|
|
||||||
import ocrmypdf
|
import ocrmypdf
|
||||||
|
import ocrmypdf._pipelines
|
||||||
import ocrmypdf.api
|
import ocrmypdf.api
|
||||||
|
|
||||||
|
|
||||||
@@ -35,7 +37,7 @@ def test_sidecar_stringio(resources: Path, outdir: Path, outpdf: Path):
|
|||||||
resources / 'ccitt.pdf',
|
resources / 'ccitt.pdf',
|
||||||
outpdf,
|
outpdf,
|
||||||
plugins=['tests/plugins/tesseract_cache.py'],
|
plugins=['tests/plugins/tesseract_cache.py'],
|
||||||
sidecar=s
|
sidecar=s,
|
||||||
)
|
)
|
||||||
s.seek(0)
|
s.seek(0)
|
||||||
assert b'the' in s.getvalue()
|
assert b'the' in s.getvalue()
|
||||||
@@ -75,3 +77,29 @@ def test_hocr_to_pdf_api(resources: Path, outdir: Path, outpdf: Path):
|
|||||||
text = extract_text(outpdf)
|
text = extract_text(outpdf)
|
||||||
assert 'hocr' in text and 'the' not in text
|
assert 'hocr' in text and 'the' not in text
|
||||||
|
|
||||||
|
|
||||||
|
def test_hocr_result_json():
|
||||||
|
result = ocrmypdf._pipelines._common.HOCRResult(
|
||||||
|
pageno=1,
|
||||||
|
pdf_page_from_image=Path('a'),
|
||||||
|
hocr=Path('b'),
|
||||||
|
textpdf=Path('c'),
|
||||||
|
orientation_correction=180,
|
||||||
|
)
|
||||||
|
assert (
|
||||||
|
result.to_json()
|
||||||
|
== '{"pageno": 1, "pdf_page_from_image": {"Path": "a"}, "hocr": {"Path": "b"}, '
|
||||||
|
'"textpdf": {"Path": "c"}, "orientation_correction": 180}'
|
||||||
|
)
|
||||||
|
assert ocrmypdf._pipelines._common.HOCRResult.from_json(result.to_json()) == result
|
||||||
|
|
||||||
|
|
||||||
|
def test_hocr_result_pickle():
|
||||||
|
result = ocrmypdf._pipelines._common.HOCRResult(
|
||||||
|
pageno=1,
|
||||||
|
pdf_page_from_image=Path('a'),
|
||||||
|
hocr=Path('b'),
|
||||||
|
textpdf=Path('c'),
|
||||||
|
orientation_correction=180,
|
||||||
|
)
|
||||||
|
assert result == pickle.loads(pickle.dumps(result))
|
||||||
|
|||||||
@@ -0,0 +1,2 @@
|
|||||||
|
import hypothesis
|
||||||
|
import pytest
|
||||||
Reference in New Issue
Block a user