pdfinfo: improve the regex

This commit is contained in:
James R. Barlow
2018-07-04 00:59:32 -07:00
parent 8b0496d35e
commit 216d60ea2c
2 changed files with 31 additions and 1 deletions
@@ -148,3 +148,24 @@ def test_pickle(resources):
filename = resources / 'formxobject.pdf'
pdf = pdfinfo.PdfInfo(filename)
pickle.dumps(pdf)
def test_regex():
rx = pdfinfo.regex_remove_char_tags
must_match = [
b'<char bbox="0 108 0 108" c="/"/>',
b'<char bbox="0 108 0 108" c=">"/>',
b'<char bbox="0 108 0 108" c="X"/>',
]
must_not_match = [
b'<span stuff="c">',
b'<span>',
b'</span>',
b'</page>'
]
for s in must_match:
assert rx.match(s)
for s in must_not_match:
assert not rx.match(s)