Merge pull request #1714 from mvanhorn/fix-nonembedded-cid-fonts-nondict-resource
fix: guard find_nonembedded_cid_fonts against non-dictionary resources
This commit is contained in:
+13
-7
@@ -12,7 +12,7 @@ from importlib.resources import files as package_files
|
||||
from pathlib import Path
|
||||
|
||||
import pikepdf
|
||||
from pikepdf import Array, Dictionary, Name, Pdf, Stream
|
||||
from pikepdf import Array, Dictionary, Name, Object, Pdf, Stream
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
@@ -137,11 +137,13 @@ def file_claims_pdfa(filename: Path):
|
||||
return pdfa_dict
|
||||
|
||||
|
||||
def _cid_font_is_embedded(type0_font: Dictionary) -> bool:
|
||||
def _cid_font_is_embedded(type0_font: Object) -> bool:
|
||||
"""Return True if a Type0 font's CID descendant carries embedded glyphs."""
|
||||
for descendant in type0_font.get(Name.DescendantFonts, []):
|
||||
descriptor = descendant.get(Name.FontDescriptor, None)
|
||||
if descriptor is not None and any(
|
||||
# A malformed PDF may store a non-dictionary here; `key in descriptor`
|
||||
# raises on those, so require a real dictionary before probing it.
|
||||
if isinstance(descriptor, Dictionary) and any(
|
||||
key in descriptor for key in (Name.FontFile, Name.FontFile2, Name.FontFile3)
|
||||
):
|
||||
return True
|
||||
@@ -174,9 +176,13 @@ def find_nonembedded_cid_fonts(pdf: Pdf) -> set[str]:
|
||||
def scan_resources(resources, depth: int = 0) -> None:
|
||||
if resources is None or depth > 10:
|
||||
return
|
||||
# A well-formed PDF stores dictionaries under /Font and /XObject, but a
|
||||
# malformed one (common in OCR workloads) may store an array, a name, or
|
||||
# another non-dictionary object. Only such dictionaries have .values(),
|
||||
# so guard with isinstance rather than let the scan crash (issue #1713).
|
||||
fonts = resources.get(Name.Font, None)
|
||||
if fonts is not None:
|
||||
for font in fonts.values():
|
||||
if isinstance(fonts, Dictionary):
|
||||
for font in fonts.as_dict().values():
|
||||
try:
|
||||
if font.get(Name.Subtype) != Name.Type0:
|
||||
continue
|
||||
@@ -186,8 +192,8 @@ def find_nonembedded_cid_fonts(pdf: Pdf) -> set[str]:
|
||||
except (AttributeError, TypeError, KeyError):
|
||||
continue
|
||||
xobjects = resources.get(Name.XObject, None)
|
||||
if xobjects is not None:
|
||||
for xobj in xobjects.values():
|
||||
if isinstance(xobjects, Dictionary):
|
||||
for xobj in xobjects.as_dict().values():
|
||||
if xobj.get(Name.Subtype) == Name.Form and Name.Resources in xobj:
|
||||
scan_resources(xobj[Name.Resources], depth + 1)
|
||||
|
||||
|
||||
@@ -94,6 +94,52 @@ class TestFindNonembeddedCidFonts:
|
||||
with pikepdf.open(path) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf) == {'ZZZ+Hidden'}
|
||||
|
||||
def test_non_dictionary_font_and_xobject_resources_are_ignored(self, tmp_path):
|
||||
# A malformed PDF may carry a /Font or /XObject resource that is not a
|
||||
# dictionary (an array, a name, an empty value). Scanning must skip it
|
||||
# rather than raise when iterating its values (regression test for the
|
||||
# crash reported in issue #1713).
|
||||
path = tmp_path / 'malformed_resources.pdf'
|
||||
with pikepdf.new() as pdf:
|
||||
page = pdf.add_blank_page()
|
||||
page.Resources = pikepdf.Dictionary(
|
||||
Font=pikepdf.Array([]),
|
||||
XObject=pikepdf.Array([]),
|
||||
)
|
||||
pdf.save(path)
|
||||
with pikepdf.open(path) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf) == set()
|
||||
|
||||
def test_non_dictionary_font_descriptor_is_reported(self, tmp_path):
|
||||
# A Type0 font whose descendant carries a non-dictionary /FontDescriptor
|
||||
# has no embedded glyph data, so it must be reported -- not crash. This
|
||||
# is the same malformed-resource bug class as #1713, one level deeper:
|
||||
# `key in descriptor` raises ValueError on a non-dictionary.
|
||||
path = tmp_path / 'bad_descriptor.pdf'
|
||||
with pikepdf.new() as pdf:
|
||||
page = pdf.add_blank_page()
|
||||
cidfont = pdf.make_indirect(
|
||||
pikepdf.Dictionary(
|
||||
Type=Name.Font,
|
||||
Subtype=Name.CIDFontType2,
|
||||
BaseFont=Name('/BOGUS+CID'),
|
||||
FontDescriptor=Name.NotADictionary,
|
||||
)
|
||||
)
|
||||
type0 = pdf.make_indirect(
|
||||
pikepdf.Dictionary(
|
||||
Type=Name.Font,
|
||||
Subtype=Name.Type0,
|
||||
BaseFont=Name('/BOGUS+CID'),
|
||||
Encoding=Name.Identity_H,
|
||||
DescendantFonts=pikepdf.Array([cidfont]),
|
||||
)
|
||||
)
|
||||
page.Resources = pikepdf.Dictionary(Font=pikepdf.Dictionary(F0=type0))
|
||||
pdf.save(path)
|
||||
with pikepdf.open(path) as pdf:
|
||||
assert find_nonembedded_cid_fonts(pdf) == {'BOGUS+CID'}
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def nonembedded_cid_pdf(tmp_path):
|
||||
|
||||
Reference in New Issue
Block a user