Main changeset for pikepdf-based refactor pdfinfo

This commit is contained in:
James R. Barlow
2018-05-24 22:22:01 -07:00
parent c00aeafff0
commit 16f70ff054
4 changed files with 80 additions and 148 deletions
+68 -138
View File
@@ -28,13 +28,11 @@ from pathlib import Path
from enum import Enum
from contextlib import contextmanager
import PyPDF2 as pypdf
from .lib import fitz
from .helpers import universal_open, fspath
matrix_mult = pypdf.pdf.utils.matrixMultiply
from pikepdf import PdfMatrix
import pikepdf
Colorspace = Enum('Colorspace',
'gray rgb cmyk lab icc index sep devn pattern jpeg2000')
@@ -118,28 +116,12 @@ XobjectSettings = namedtuple('XobjectSettings',
['name', 'shorthand', 'stack_depth'])
InlineSettings = namedtuple('InlineSettings',
['settings', 'shorthand', 'stack_depth'])
['iimage', 'shorthand', 'stack_depth'])
ContentsInfo = namedtuple('ContentsInfo',
['xobject_settings', 'inline_images', 'found_text'])
def _normalize_stack(operations):
"""Fix runs of qQ's in the stack
For some reason PyPDF2 converts runs of qqq, QQ, QQQq, etc. into single
operations. Break this silliness up and issue each stack operation
individually so we don't lose count.
"""
for operands, command in operations:
if re.match(br'Q*q+$', command): # Zero or more Q, one or more q
for char in command: # Split into individual bytes
yield ([], bytes([char])) # Yield individual bytes
else:
yield (operands, command)
def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
"""Interpret the PDF content stream
@@ -159,46 +141,44 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
PDF units suit our needs so we initialize ctm to the identity matrix.
PyPDF2 replaces inline images with a fake "INLINE IMAGE" operator.
"""
operations = contentstream.operations
stack = []
ctm = _matrix_from_shorthand(initial_shorthand)
ctm = PdfMatrix(initial_shorthand)
xobject_settings = []
inline_images = []
found_text = False
text_operators = tuple(
pikepdf.Operator(op) for op in ('Tj', 'TJ', '"', "'"))
for n, op in enumerate(_normalize_stack(operations)):
for n, op in enumerate(pikepdf.parse_content_stream(contentstream)):
operands, command = op
if command == b'q':
if command == pikepdf.Operator('q'):
stack.append(ctm)
if len(stack) > 32:
raise RuntimeError(
"PDF graphics stack overflow, command %i" % n)
elif command == b'Q':
elif command == pikepdf.Operator('Q'):
try:
ctm = stack.pop()
except IndexError:
raise RuntimeError(
"PDF graphics stack underflow, command %i" % n)
elif command == b'cm':
ctm = matrix_mult(
_matrix_from_shorthand(operands), ctm)
elif command == b'Do':
elif command == pikepdf.Operator('cm'):
ctm = PdfMatrix(operands) @ ctm
elif command == pikepdf.Operator('Do'):
image_name = operands[0]
settings = XobjectSettings(
name=image_name, shorthand=_shorthand_from_matrix(ctm),
name=image_name, shorthand=ctm.shorthand,
stack_depth=len(stack))
xobject_settings.append(settings)
elif command == b'INLINE IMAGE':
settings = operands['settings']
elif command == pikepdf.Operator('INLINE IMAGE'):
iimage = operands[0]
inline = InlineSettings(
settings=settings, shorthand=_shorthand_from_matrix(ctm),
iimage=iimage, shorthand=ctm.shorthand,
stack_depth=len(stack))
inline_images.append(inline)
elif command in (b'Tj', b'TJ', b'"', b"'"):
elif command in text_operators:
found_text = True
@@ -278,73 +258,41 @@ class ImageInfo:
def __init__(self, *, name='', pdfimage=None, inline=None,
shorthand=None):
self._name = name
self._name = str(name)
self._shorthand = shorthand
if inline:
# Fixme does not work for inline images with non abbreviated
# fields
if inline is not None:
self._origin = 'inline'
self._width = inline.settings['/W']
self._height = inline.settings['/H']
self._type = 'stencil' if inline.settings.get('/IM') else 'image'
default_bpc = 8 if self._type == 'image' else 1
self._bpc = inline.settings.get('/BPC', default_bpc)
try:
self._color = FRIENDLY_COLORSPACE[inline.settings['/CS']]
except Exception:
self._color = '-'
self._comp = FRIENDLY_COMP.get(self._color, '?')
if '/F' in inline.settings:
filter_ = inline.settings['/F']
if isinstance(filter_, pypdf.generic.ArrayObject):
filter_ = filter_[0]
self._enc = FRIENDLY_ENCODING.get(filter_, 'image')
else:
self._enc = 'image'
elif pdfimage:
pim = inline.iimage
elif pdfimage is not None:
self._origin = 'xobject'
self._width = pdfimage['/Width']
self._height = pdfimage['/Height']
pim = pikepdf.PdfImage(pdfimage)
self._width = pim.width
self._height = pim.height
# If /ImageMask is true, then this image is a stencil mask
# (Images that draw with this stencil mask will have a reference to
# it in their /Mask, but we don't actually need that information)
if '/ImageMask' in pdfimage:
self._type = 'stencil' if pdfimage['/ImageMask'].value \
else 'image'
else:
self._type = 'image'
# If /ImageMask is true, then this image is a stencil mask
# (Images that draw with this stencil mask will have a reference to
# it in their /Mask, but we don't actually need that information)
if pim.image_mask:
self._type = 'stencil'
else:
self._type = 'image'
default_bpc = 8 if self._type == 'image' else 1
if '/BitsPerComponent' in pdfimage:
self._bpc = pdfimage['/BitsPerComponent']
else:
self._bpc = default_bpc
self._bpc = int(pim.bits_per_component)
self._enc = FRIENDLY_ENCODING.get(pim.filters[0], 'image')
if '/Filter' in pdfimage:
filter_ = pdfimage['/Filter']
if isinstance(filter_, pypdf.generic.ArrayObject):
filter_ = filter_[0]
self._enc = FRIENDLY_ENCODING.get(filter_, 'image')
else:
self._enc = 'image'
if '/ColorSpace' in pdfimage:
cs = pdfimage['/ColorSpace']
if isinstance(cs, pypdf.generic.ArrayObject):
cs = cs[0]
self._color = FRIENDLY_COLORSPACE.get(cs, '-')
else:
self._color = FRIENDLY_COLORSPACE[Colorspace.jpeg2000] \
if self._enc == Encoding.jpeg2000 else '?'
self._color = FRIENDLY_COLORSPACE.get(pim.colorspace, '?')
if self._enc == Encoding.jpeg2000:
self._color = Colorspace.jpeg2000
self._comp = FRIENDLY_COMP.get(self._color, '?')
self._comp = FRIENDLY_COMP.get(self._color, '?')
# Bit of a hack... infer grayscale if component count is uncertain
# but encoding must be monochrome. This happens if a monochrome image
# has an ICC profile attached. Better solution would be to examine
# the ICC profile.
if self._comp == '?' and self._enc in (Encoding.ccitt, 'jbig2'):
self._comp = FRIENDLY_COMP[Colorspace.gray]
# Bit of a hack... infer grayscale if component count is uncertain
# but encoding must be monochrome. This happens if a monochrome image
# has an ICC profile attached. Better solution would be to examine
# the ICC profile.
if self._comp == '?' and self._enc in (Encoding.ccitt, 'jbig2'):
self._comp = FRIENDLY_COMP[Colorspace.gray]
@property
def name(self):
@@ -386,20 +334,6 @@ class ImageInfo:
def yres(self):
return _get_dpi(self._shorthand, (self._width, self._height))[1]
def __getitem__(self, item):
warnings.warn("ImageInfo.__getitem__", DeprecationWarning)
if item in ('name', 'width', 'height', 'bpc', 'color', 'comp', 'enc'):
return getattr(self, item)
elif item == 'dpi_w':
return Decimal(self.xres).quantize(self.DPI_PREC)
elif item == 'dpi_h':
return Decimal(self.yres).quantize(self.DPI_PREC)
elif item == 'dpi':
return Decimal(self.xres * self.yres).sqrt().quantize(
self.DPI_PREC)
else:
raise KeyError(item)
def __repr__(self):
class_locals = {attr: getattr(self, attr, None) for attr in dir(self)
if not attr.startswith('_')}
@@ -434,8 +368,9 @@ def _image_xobjects(container):
resources = container['/Resources']
if '/XObject' not in resources:
return
for xobj in resources['/XObject']:
candidate = resources['/XObject'][xobj]
xobjs = resources['/XObject'].as_dict()
for xobj in xobjs:
candidate = xobjs[xobj]
if candidate['/Subtype'] == '/Image':
pdfimage = candidate
yield (pdfimage, xobj)
@@ -482,8 +417,9 @@ def _find_form_xobject_images(pdf, container, contentsinfo):
resources = container['/Resources']
if '/XObject' not in resources:
return
for xobj in resources['/XObject']:
candidate = resources['/XObject'][xobj]
xobjs = resources['/XObject'].as_dict()
for xobj in xobjs:
candidate = xobjs[xobj]
if candidate['/Subtype'] != '/Form':
continue
@@ -521,32 +457,26 @@ def _find_images(*, pdf, container, shorthand=None):
"""
if container.get('/Type') == '/Page' and '/Contents' in container:
# For a /Page the content stream is attached to the page's /Contents
page = container
contentstream = pypdf.pdf.ContentStream(page.getContents(), pdf)
initial_shorthand = shorthand or UNIT_SQUARE
elif container.get('/Type') == '/XObject' and \
container['/Subtype'] == '/Form':
# For a Form XObject that content stream is attached to the XObject
contentstream = pypdf.pdf.ContentStream(container, pdf)
# Set the CTM to the state it was when the "Do" operator was
# encountered that is drawing this instance of the Form XObject
ctm = _matrix_from_shorthand(shorthand or UNIT_SQUARE)
ctm = PdfMatrix(shorthand) if shorthand else PdfMatrix.identity()
# A Form XObject may provide its own matrix to map form space into
# user space. Get this if one exists
form_matrix = _matrix_from_shorthand(
container.get('/Matrix', UNIT_SQUARE))
form_shorthand = container.get('/Matrix', PdfMatrix.identity())
form_matrix = PdfMatrix(form_shorthand)
# Concatenate form matrix with CTM to ensure CTM is correct for
# drawing this instance of the XObject
ctm = matrix_mult(form_matrix, ctm)
initial_shorthand = _shorthand_from_matrix(ctm)
ctm = form_matrix @ ctm
initial_shorthand = ctm.shorthand
else:
return
contentsinfo = _interpret_contents(contentstream, initial_shorthand)
contentsinfo = _interpret_contents(container, initial_shorthand)
yield from _find_inline_images(contentsinfo)
yield from _find_regular_images(container, contentsinfo)
@@ -559,8 +489,7 @@ def _naive_find_text(*, pdf, page):
return False
# First we check the main content stream
contentstream = pypdf.pdf.ContentStream(page.getContents(), pdf)
contentsinfo = _interpret_contents(contentstream, UNIT_SQUARE)
contentsinfo = _interpret_contents(page, UNIT_SQUARE)
if contentsinfo.found_text:
return True
@@ -571,15 +500,15 @@ def _naive_find_text(*, pdf, page):
if '/Resources' in page:
resources = page['/Resources']
if '/XObject' in resources:
for xobj in resources['/XObject']:
candidate = resources['/XObject'][xobj]
xobjs = resources['/XObject'].as_dict()
for xobj in xobjs:
candidate = xobjs[xobj]
if candidate['/Subtype'] != '/Form':
continue
form_xobject = candidate
# Content stream is attached to Form XObject dictionary
contentstream = pypdf.pdf.ContentStream(form_xobject, pdf)
sub_contentsinfo = _interpret_contents(
contentstream, UNIT_SQUARE)
form_xobject, UNIT_SQUARE)
if sub_contentsinfo.found_text:
return True
return False
@@ -637,11 +566,13 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile):
else:
pageinfo['has_text'] = _naive_find_text(pdf=pdf, page=page)
mediabox = page.MediaBox.as_list()
mediabox = [Decimal(d.decode()) for d in page.MediaBox.as_list()]
width_pt = mediabox[2] - mediabox[0]
height_pt = mediabox[3] - mediabox[1]
userunit = page.get('/UserUnit', Decimal(1.0))
if not isinstance(userunit, Decimal):
userunit = userunit.decode()
pageinfo['userunit'] = userunit
pageinfo['width_inches'] = width_pt * userunit / Decimal(72.0)
pageinfo['height_inches'] = height_pt * userunit / Decimal(72.0)
@@ -656,8 +587,8 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile):
_find_images(pdf=pdf, container=page,
shorthand=userunit_shorthand)]
if pageinfo['images']:
xres = max(image['dpi_w'] for image in pageinfo['images'])
yres = max(image['dpi_h'] for image in pageinfo['images'])
xres = Decimal(max(image.xres for image in pageinfo['images']))
yres = Decimal(max(image.yres for image in pageinfo['images']))
pageinfo['xres'], pageinfo['yres'] = xres, yres
pageinfo['width_pixels'] = \
int(round(xres * pageinfo['width_inches']))
@@ -668,9 +599,8 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile):
def _pdf_get_all_pageinfo(infile):
with universal_open(infile, 'rb') as f:
pdf = pypdf.PdfFileReader(f)
return [PageInfo(pdf, n, infile) for n in range(pdf.numPages)]
pdf = pikepdf.open(infile)
return [PageInfo(pdf, n, infile) for n in range(len(pdf.pages))]
class PageInfo:
+1 -1
View File
@@ -603,7 +603,7 @@ def select_visible_page_image(
pageinfo = get_pageinfo(image, context)
if pageinfo.images and \
all(im['enc'] == 'jpeg' for im in pageinfo.images):
all(im.enc == 'jpeg' for im in pageinfo.images):
log.debug('{:4d}: JPEG input -> JPEG output'.format(
page_number(image)))
# If all images were JPEGs originally, produce a JPEG as output
+7 -7
View File
@@ -109,7 +109,7 @@ def test_deskew(spoof_tesseract_noop, resources, outdir):
raster_device='pngmono',
log=log,
pageno=1)
pix = Pix.read(str(deskewed_png))
skew_angle, skew_confidence = pix.find_skew()
@@ -383,7 +383,7 @@ def test_tesseract_image_too_big(renderer, spoof_tesseract_big_image_error,
resources, outpdf):
check_ocrmypdf(
resources / 'hugemono.pdf', outpdf, '-r',
'--pdf-renderer', renderer,
'--pdf-renderer', renderer,
'--max-image-mpixels', '0',
env=spoof_tesseract_big_image_error)
@@ -439,8 +439,8 @@ def test_convert_to_square_resolution(renderer, spoof_tesseract_cache,
# Because we rasterized the page to produce a new image, it should occupy
# the entire page
out_im_w = out_p0.images[0]['width'] / out_p0.images[0]['dpi_w']
out_im_h = out_p0.images[0]['height'] / out_p0.images[0]['dpi_h']
out_im_w = out_p0.images[0].width / out_p0.images[0].xres
out_im_h = out_p0.images[0].height / out_p0.images[0].yres
assert isclose(out_p0.width_inches, out_im_w)
assert isclose(out_p0.height_inches, out_im_h)
@@ -651,7 +651,7 @@ def test_form_xobject(spoof_tesseract_noop, resources, outpdf):
@pytest.mark.parametrize('renderer', RENDERERS)
def test_pagesize_consistency(renderer, resources, outpdf):
first_page_dimensions = pytest.helpers.first_page_dimensions
@@ -893,10 +893,10 @@ def test_text_curves(spoof_tesseract_noop, resources, outpdf):
check_ocrmypdf(
resources / 'vector.pdf', outpdf, '--force-ocr',
env=spoof_tesseract_noop)
info = PdfInfo(outpdf)
assert len(info.pages[0].images) != 0, "force did not rasterize"
def test_dev_null(spoof_tesseract_noop, resources):
p, out, err = run_ocrmypdf(
+4 -2
View File
@@ -28,6 +28,8 @@ import pytest
import img2pdf
import sys
import PyPDF2 as pypdf
import pikepdf
import pickle
def test_single_page_text(outdir):
@@ -127,8 +129,8 @@ def test_form_xobject(resources):
def test_naive_find_text(resources):
filename = resources / 'formxobject.pdf'
reader = pypdf.PdfFileReader(str(filename))
page = reader.getPage(0)
reader = pikepdf.open(filename)
page = reader.pages[0]
assert pdfinfo._naive_find_text(pdf=reader, page=page)