pdfinfo: improve the regex

This commit is contained in:
James R. Barlow
2018-07-04 00:59:32 -07:00
parent 8b0496d35e
commit 216d60ea2c
2 changed files with 31 additions and 1 deletions
+10 -1
View File
@@ -36,7 +36,16 @@ Encoding = Enum('Encoding',
'ccitt jpeg jpeg2000 jbig2 asciihex ascii85 lzw flate ' + \
'runlength')
regex_remove_char_tags = re.compile(br"<char[^\/]+\/>")
# Forgive me for I have sinned
# I am using regular expressions to parse XML. However the XML in this case,
# generated by Ghostscript, is self-consistent enough to be parseable.
regex_remove_char_tags = re.compile(br"""
<char\b
(?: [^>] # anything single character but >
| \">\" # special case: trap ">"
)*
/> # terminate with '/>'
""", re.VERBOSE)
FRIENDLY_COLORSPACE = {
'/DeviceGray': Colorspace.gray,