Refine non-dict resource guard: prefer isinstance, cover FontDescriptor

Follow-on to the non-dictionary /Font and /XObject guard. Replace the
_dict_entries() helper with an isinstance(..., pikepdf.Dictionary) check
at each resource lookup, which pikepdf's metaclass supports directly and
which reads as exactly the invariant being enforced. Iterate via
as_dict().values() so the values are typed and mypy stays clean once the
Any from the untyped resources argument is narrowed away.

Also guard _cid_font_is_embedded against a non-dictionary /FontDescriptor:
`key in descriptor` raises ValueError on a non-dict, which the caller's
except (AttributeError, TypeError, KeyError) does not catch. Such a font
now counts as non-embedded and is reported, so PDF/A conversion is refused
rather than risking Ghostscript corrupting a pre-existing CID text layer.

Adds a regression test for the FontDescriptor case (issue #1713).
This commit is contained in:
James R. Barlow
2026-07-16 23:42:50 -07:00
parent ef903db360
commit 640b3062b2
2 changed files with 52 additions and 31 deletions
+13 -22
View File
@@ -12,7 +12,7 @@ from importlib.resources import files as package_files
from pathlib import Path
import pikepdf
from pikepdf import Array, Dictionary, Name, Pdf, Stream
from pikepdf import Array, Dictionary, Name, Object, Pdf, Stream
log = logging.getLogger(__name__)
@@ -137,34 +137,19 @@ def file_claims_pdfa(filename: Path):
return pdfa_dict
def _cid_font_is_embedded(type0_font: Dictionary) -> bool:
def _cid_font_is_embedded(type0_font: Object) -> bool:
"""Return True if a Type0 font's CID descendant carries embedded glyphs."""
for descendant in type0_font.get(Name.DescendantFonts, []):
descriptor = descendant.get(Name.FontDescriptor, None)
if descriptor is not None and any(
# A malformed PDF may store a non-dictionary here; `key in descriptor`
# raises on those, so require a real dictionary before probing it.
if isinstance(descriptor, Dictionary) and any(
key in descriptor for key in (Name.FontFile, Name.FontFile2, Name.FontFile3)
):
return True
return False
def _dict_entries(resource):
"""Return the values of a PDF resource sub-dictionary, tolerating garbage.
A ``/Font`` or ``/XObject`` resource is expected to be a dictionary, but a
malformed PDF -- common in OCR workloads -- may store a non-dictionary
object (an array, a name, an empty value) there. Calling ``.values()`` on a
non-dictionary raises, so return an empty list in that case rather than
letting the scan crash (issue #1713).
"""
if resource is None:
return []
try:
return list(resource.values())
except (AttributeError, TypeError):
return []
def find_nonembedded_cid_fonts(pdf: Pdf) -> set[str]:
"""Find CID-keyed (Type0) fonts that lack embedded glyph data.
@@ -191,8 +176,13 @@ def find_nonembedded_cid_fonts(pdf: Pdf) -> set[str]:
def scan_resources(resources, depth: int = 0) -> None:
if resources is None or depth > 10:
return
# A well-formed PDF stores dictionaries under /Font and /XObject, but a
# malformed one (common in OCR workloads) may store an array, a name, or
# another non-dictionary object. Only such dictionaries have .values(),
# so guard with isinstance rather than let the scan crash (issue #1713).
fonts = resources.get(Name.Font, None)
for font in _dict_entries(fonts):
if isinstance(fonts, Dictionary):
for font in fonts.as_dict().values():
try:
if font.get(Name.Subtype) != Name.Type0:
continue
@@ -202,7 +192,8 @@ def find_nonembedded_cid_fonts(pdf: Pdf) -> set[str]:
except (AttributeError, TypeError, KeyError):
continue
xobjects = resources.get(Name.XObject, None)
for xobj in _dict_entries(xobjects):
if isinstance(xobjects, Dictionary):
for xobj in xobjects.as_dict().values():
if xobj.get(Name.Subtype) == Name.Form and Name.Resources in xobj:
scan_resources(xobj[Name.Resources], depth + 1)
+30
View File
@@ -110,6 +110,36 @@ class TestFindNonembeddedCidFonts:
with pikepdf.open(path) as pdf:
assert find_nonembedded_cid_fonts(pdf) == set()
def test_non_dictionary_font_descriptor_is_reported(self, tmp_path):
# A Type0 font whose descendant carries a non-dictionary /FontDescriptor
# has no embedded glyph data, so it must be reported -- not crash. This
# is the same malformed-resource bug class as #1713, one level deeper:
# `key in descriptor` raises ValueError on a non-dictionary.
path = tmp_path / 'bad_descriptor.pdf'
with pikepdf.new() as pdf:
page = pdf.add_blank_page()
cidfont = pdf.make_indirect(
pikepdf.Dictionary(
Type=Name.Font,
Subtype=Name.CIDFontType2,
BaseFont=Name('/BOGUS+CID'),
FontDescriptor=Name.NotADictionary,
)
)
type0 = pdf.make_indirect(
pikepdf.Dictionary(
Type=Name.Font,
Subtype=Name.Type0,
BaseFont=Name('/BOGUS+CID'),
Encoding=Name.Identity_H,
DescendantFonts=pikepdf.Array([cidfont]),
)
)
page.Resources = pikepdf.Dictionary(Font=pikepdf.Dictionary(F0=type0))
pdf.save(path)
with pikepdf.open(path) as pdf:
assert find_nonembedded_cid_fonts(pdf) == {'BOGUS+CID'}
@pytest.fixture
def nonembedded_cid_pdf(tmp_path):