@@ -33,6 +33,7 @@ from reportlab.lib.units import inch
from xml . etree import ElementTree
from PIL import Image
from collections import namedtuple
from math import atan , sin , cos
import re
import argparse
@@ -49,13 +50,24 @@ class HocrTransform():
"""
A class for converting documents from the hOCR format.
For details of the hOCR format, see:
http://docs.google.com/View?docid=dfxcv4vc_67g844kf
http://kba.cloud/hocr-spec/
"""
box_pattern = re . compile ( r ' bbox(( \ s+ \ d+) {4} ) ' )
baseline_pattern = re . compile ( r '''
baseline \ s+
([ \ - \ +]? \ d* \ .? \ d*) \ s+ # +/- decimal float
([ \ - \ +]? \ d+) # +/- int ''' , re . VERBOSE )
ligatures = str . maketrans ( {
' ff ' : ' ff ' ,
' ffi ' : ' f f i ' ,
' ffl ' : ' f f l ' ,
' fi ' : ' fi ' ,
' fl ' : ' fl ' ,
} )
def __init__ ( self , hocrFileName , dpi ) :
self . dpi = dpi
self . boxPattern = re . compile ( r ' bbox(( \ s+ \ d+) {4} ) ' )
self . hocr = ElementTree . parse ( hocrFileName )
# if the hOCR file has a namespace, ElementTree requires its use to
@@ -104,19 +116,31 @@ class HocrTransform():
text + = element . tail
return text
def element_coordinates ( self , element ) :
@classmethod
def element_coordinates ( cls , element ) :
"""
Returns a tuple containing the coordinates of the bounding box around
an element
"""
out = ( 0 , 0 , 0 , 0 )
if ' title ' in element . attrib :
matches = self . boxPattern . search ( element . attrib [ ' title ' ] )
matches = cls . box_pattern . search ( element . attrib [ ' title ' ] )
if matches :
coords = matches . group ( 1 ) . split ( )
out = Rect . _make ( int ( coords [ n ] ) for n in range ( 4 ) )
return out
@classmethod
def baseline ( cls , element ) :
"""
Returns a tuple containing the baseline slope and intercept.
"""
if ' title ' in element . attrib :
matches = cls . baseline_pattern . search ( element . attrib [ ' title ' ] )
if matches :
return float ( matches . group ( 1 ) ) , int ( matches . group ( 2 ) )
return ( 0 , 0 )
def pt_from_pixel ( self , pxl ) :
"""
Returns the quantity in PDF units (pt) given quantity in pixels
@@ -124,20 +148,17 @@ class HocrTransform():
return Rect . _make (
( c / self . dpi * inch ) for c in pxl )
def replace_unsupported_chars ( self , s ) :
@classmethod
def replace_unsupported_chars ( cls , s ) :
"""
Given an input string, returns the corresponding string that:
- is available in the helvetica facetype
- does not contain any ligature (to allow easy search in the PDF file)
"""
# The 'u' before the character to replace indicates that it is a
# unicode character
s = s . replace ( u " fl " , " fl " )
s = s . replace ( u " fi " , " fi " )
return s
return s . translate ( cls . ligatures )
def to_pdf ( self , outFileName , imageFileName = None , showBoundingboxes = False ,
fontname = " Helvetica " , invisibleText = False ) :
fontname = " Helvetica " , invisibleText = False , interwordSpaces = False ) :
"""
Creates a PDF file with an image superimposed on top of the text.
Text is positioned according to the bounding box of the lines in
@@ -172,57 +193,19 @@ class HocrTransform():
pdf . rect (
pt . x1 , self . height - pt . y2 , pt . x2 - pt . x1 , pt . y2 - pt . y1 ,
fill = 1 )
found_lines = False
for line in self . hocr . findall (
" .// %s span[@class= ' %s ' ] " % ( self . xmlns , " ocr_line " ) ) :
found_lines = True
self . _do_line ( pdf , line , " ocrx_word " , fontname , invisibleText ,
interwordSpaces , showBoundingboxes )
# check if element with class 'ocrx_word' are available
# otherwise use 'ocr_line' as fallback
elemclass = " ocr_line "
if self . hocr . find (
" .// %s span[@class= ' ocrx_word ' ] " % ( self . xmlns ) ) is not None :
elemclass = " ocrx_word "
# itterate all text elements
# light green for bounding box of word/line
pdf . setStrokeColorRGB ( 1 , 0 , 0 )
pdf . setLineWidth ( 0.5 ) # bounding box line width
pdf . setDash ( 6 , 3 ) # bounding box is dashed
pdf . setFillColorRGB ( 0 , 0 , 0 ) # text in black
for elem in self . hocr . findall (
" .// %s span[@class= ' %s ' ] " % ( self . xmlns , elemclass ) ) :
elemtxt = self . _get_element_text ( elem ) . rstrip ( )
elemtxt = self . replace_unsupported_chars ( elemtxt )
if len ( elemtxt ) == 0 :
continue
pxl_coords = self . element_coordinates ( elem )
pt = self . pt_from_pixel ( pxl_coords )
# draw the bbox border
if showBoundingboxes :
pdf . rect (
pt . x1 , self . height - pt . y2 , pt . x2 - pt . x1 , pt . y2 - pt . y1 ,
fill = 0 )
text = pdf . beginText ( )
fontsize = pt . y2 - pt . y1
text . setFont ( fontname , fontsize )
if invisibleText :
text . setTextRenderMode ( 3 ) # Invisible (indicates OCR text)
# set cursor to bottom left corner of bbox (adjust for dpi)
text . setTextOrigin ( pt . x1 , self . height - pt . y2 )
# scale the width of the text to fill the width of the bbox
text . setHorizScale (
100 * ( pt . x2 - pt . x1 ) / pdf . stringWidth (
elemtxt , fontname , fontsize ) )
# write the text to the page
text . textLine ( elemtxt )
pdf . drawText ( text )
if not found_lines :
# Tesseract did not report any lines (just words)
root = self . hocr . find ( " .// %s div[@class= ' %s ' ] " % ( self . xmlns , " ocr_page " ) )
self . _do_line ( pdf , root , " ocrx_word " , fontname , invisibleText ,
interwordSpaces , showBoundingboxes )
# put the image on the page, scaled to fill the page
if imageFileName is not None :
pdf . drawImage ( imageFileName , 0 , 0 ,
@@ -233,6 +216,117 @@ class HocrTransform():
pdf . save ( )
@classmethod
def polyval ( cls , poly , x ) :
return x * poly [ 0 ] + poly [ 1 ]
def _do_line ( self , pdf , line , elemclass , fontname , invisibleText ,
interwordSpaces , showBoundingboxes ) :
pxl_line_coords = self . element_coordinates ( line )
line_box = self . pt_from_pixel ( pxl_line_coords )
line_height = line_box . y2 - line_box . y1
slope , pxl_intercept = self . baseline ( line )
if abs ( slope ) < 0.005 :
slope = 0.0
angle = atan ( slope )
cos_a , sin_a = cos ( angle ) , sin ( angle )
text = pdf . beginText ( )
intercept = pxl_intercept / self . dpi * inch
# Don't allow the font to break out of the bounding box. Division by
# cos_a accounts for extra clearance between the glyph's vertical axis
# on a sloped baseline and the edge of the bounding box.
fontsize = ( line_height - abs ( intercept ) ) / cos_a
text . setFont ( fontname , fontsize )
if invisibleText :
text . setTextRenderMode ( 3 ) # Invisible (indicates OCR text)
# Intercept is normally negative, so this places it above the bottom
# of the line box
baseline_y2 = self . height - ( line_box . y2 + intercept )
if showBoundingboxes :
# draw the baseline in magenta, dashed
pdf . setDash ( )
pdf . setStrokeColorRGB ( 0.95 , 0.65 , 0.95 )
pdf . setLineWidth ( 0.5 )
# negate slope because it is defined as a rise/run in pixel
# coordinates and page coordinates have the y axis flipped
pdf . line ( line_box . x1 ,
baseline_y2 ,
line_box . x2 ,
self . polyval ( ( - slope , baseline_y2 ) ,
line_box . x2 - line_box . x1 ) )
# light green for bounding box of word/line
pdf . setDash ( 6 , 3 )
pdf . setStrokeColorRGB ( 1 , 0 , 0 )
text . setTextTransform (
cos_a , - sin_a , sin_a , cos_a ,
line_box . x1 , baseline_y2
)
pdf . setFillColorRGB ( 0 , 0 , 0 ) # text in black
elements = line . findall (
" .// %s span[@class= ' %s ' ] " % ( self . xmlns , elemclass ) )
for elem in elements :
elemtxt = self . _get_element_text ( elem ) . strip ( )
elemtxt = self . replace_unsupported_chars ( elemtxt )
if elemtxt == ' ' :
continue
pxl_coords = self . element_coordinates ( elem )
box = self . pt_from_pixel ( pxl_coords )
if interwordSpaces :
# if `--interword-spaces` is true, append a space
# to the end of each text element to allow simpler PDF viewers
# such as PDF.js to better recognize words in search and copy
# and paste. Do not remove space from last word in line, even
# though it would look better, because it will interfere with
# naive text extraction. \n does not work either.
elemtxt + = ' '
box = Rect . _make ( (
box . x1 ,
line_box . y1 ,
box . x2 + pdf . stringWidth ( ' ' , fontname , line_height ) ,
line_box . y2 ) )
box_width = box . x2 - box . x1
font_width = pdf . stringWidth ( elemtxt , fontname , fontsize )
# draw the bbox border
if showBoundingboxes :
pdf . rect (
box . x1 ,
self . height - line_box . y2 ,
box_width ,
line_height ,
fill = 0 )
# Adjust relative position of cursor
# This is equivalent to:
# text.setTextOrigin(pt.x1, self.height - line_box.y2)
# but the former generates a full text reposition matrix (Tm) in the
# content stream while this issues a "offset" (Td) command.
# .moveCursor() is relative to start of the text line, where the
# "text line" means whatever reportlab defines it as. Do not use
# use .getCursor(), since moveCursor() rather unintuitively plans
# its moves relative to .getStartOfLine().
# For skewed lines, in the text transform we set up a rotated
# coordinate system, so we don't have to account for the
# incremental offset. Surprisingly most PDF viewers can handle this.
cursor = text . getStartOfLine ( )
dx = box . x1 - cursor [ 0 ]
dy = baseline_y2 - cursor [ 1 ]
text . moveCursor ( dx , dy )
text . setHorizScale ( 100 * box_width / font_width )
text . textOut ( elemtxt )
pdf . drawText ( text )
if __name__ == " __main__ " :
parser = argparse . ArgumentParser ( description = ' Convert hocr file to PDF ' )
parser . add_argument ( ' -b ' , ' --boundingboxes ' , action = " store_true " ,
@@ -242,10 +336,12 @@ if __name__ == "__main__":
help = ' Resolution of the image that was OCRed ' )
parser . add_argument ( ' -i ' , ' --image ' , default = None ,
help = ' Path to the image to be placed above the text ' )
parser . add_argument ( ' --interword-spaces ' , action = ' store_true ' ,
default = False , help = ' Add spaces between words ' )
parser . add_argument ( ' hocrfile ' , help = ' Path to the hocr file to be parsed ' )
parser . add_argument (
' outputfile ' , help = ' Path to the PDF file to be generated ' )
args = parser . parse_args ( )
hocr = HocrTransform ( args . hocrfile , args . resolution )
hocr . to_pdf ( args . outputfile , args . image , args . boundingboxes )
hocr . to_pdf ( args . outputfile , args . image , args . boundingboxes , interwordSpaces = args . interword_spaces )