Add tests for fpdf2 renderer and font infrastructure
- Add hOCR test fixtures for Latin, Arabic, CJK, Devanagari scripts - Add tests for fpdf2 renderer, multi-font manager, system font provider - Add multilingual rendering tests - Update existing tests to use fpdf2 renderer
This commit is contained in:
@@ -134,7 +134,7 @@ def cached_run(options, run_args, **run_kwargs):
|
||||
args.configfiles.append('txt')
|
||||
|
||||
for configfile in args.configfiles:
|
||||
if configfile not in ('hocr', 'pdf', 'txt'):
|
||||
if configfile not in ('fpdf2', 'pdf', 'txt'):
|
||||
continue
|
||||
# cp pwd/{outputbase}.{configfile} -> {cache}/{configfile}
|
||||
tessfile = args.outputbase + '.' + configfile
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
To quickly run tests where getting OCR output is not necessary and we want to test
|
||||
the rotation pipeline.
|
||||
|
||||
In 'hocr' mode, create a .hocr file that specifies no text found.
|
||||
In generate_hocr mode, create a .hocr file that specifies no text found.
|
||||
|
||||
In 'pdf' mode, convert the image to PDF using another program.
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@
|
||||
|
||||
To quickly run tests where getting OCR output is not necessary.
|
||||
|
||||
In 'hocr' mode, create a .hocr file that specifies no text found.
|
||||
In generate_hocr mode, create a .hocr file that specifies no text found.
|
||||
|
||||
In 'pdf' mode, convert the image to PDF using another program.
|
||||
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="ar" lang="ar">
|
||||
<head>
|
||||
<title></title>
|
||||
<meta http-equiv="content-type" content="text/html; charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 5.0.0' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word'/>
|
||||
</head>
|
||||
<body>
|
||||
<div class='ocr_page' id='page_1' title='image "test.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
||||
<div class='ocr_carea' id='carea_1_1' title="bbox 200 200 2350 1200">
|
||||
<p class='ocr_par' id='par_1_1' lang='ara' dir='rtl' title="bbox 200 200 2350 400">
|
||||
<span class='ocr_line' id='line_1_1' title="bbox 200 200 2350 400; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_1' title='bbox 200 200 600 400; x_wconf 95'>مرحبا</span>
|
||||
<span class='ocrx_word' id='word_1_2' title='bbox 650 200 1050 400; x_wconf 95'>بالعالم</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_2' lang='ara' dir='rtl' title="bbox 200 500 2350 700">
|
||||
<span class='ocr_line' id='line_1_2' title="bbox 200 500 2350 700; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_3' title='bbox 200 500 600 700; x_wconf 95'>هذا</span>
|
||||
<span class='ocrx_word' id='word_1_4' title='bbox 650 500 1050 700; x_wconf 95'>نص</span>
|
||||
<span class='ocrx_word' id='word_1_5' title='bbox 1100 500 1500 700; x_wconf 95'>عربي</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_3' lang='per' dir='rtl' title="bbox 200 800 2350 1000">
|
||||
<span class='ocr_line' id='line_1_3' title="bbox 200 800 2350 1000; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_6' title='bbox 200 800 600 1000; x_wconf 95'>سلام</span>
|
||||
<span class='ocrx_word' id='word_1_7' title='bbox 650 800 1050 1000; x_wconf 95'>فارسی</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,41 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="zh" lang="zh">
|
||||
<head>
|
||||
<title></title>
|
||||
<meta http-equiv="content-type" content="text/html; charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 5.0.0' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word'/>
|
||||
</head>
|
||||
<body>
|
||||
<div class='ocr_page' id='page_1' title='image "test.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
||||
<div class='ocr_carea' id='carea_1_1' title="bbox 200 200 2350 1500">
|
||||
<p class='ocr_par' id='par_1_1' lang='chi_sim' title="bbox 200 200 2350 400">
|
||||
<span class='ocr_line' id='line_1_1' title="bbox 200 200 2350 400; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_1' title='bbox 200 200 600 400; x_wconf 95'>你好</span>
|
||||
<span class='ocrx_word' id='word_1_2' title='bbox 650 200 1050 400; x_wconf 95'>世界</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_2' lang='chi_tra' title="bbox 200 500 2350 700">
|
||||
<span class='ocr_line' id='line_1_2' title="bbox 200 500 2350 700; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_3' title='bbox 200 500 600 700; x_wconf 95'>繁體</span>
|
||||
<span class='ocrx_word' id='word_1_4' title='bbox 650 500 1050 700; x_wconf 95'>中文</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_3' lang='jpn' title="bbox 200 800 2350 1000">
|
||||
<span class='ocr_line' id='line_1_3' title="bbox 200 800 2350 1000; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_5' title='bbox 200 800 600 1000; x_wconf 95'>こんにちは</span>
|
||||
<span class='ocrx_word' id='word_1_6' title='bbox 650 800 1050 1000; x_wconf 95'>世界</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_4' lang='kor' title="bbox 200 1100 2350 1300">
|
||||
<span class='ocr_line' id='line_1_4' title="bbox 200 1100 2350 1300; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_7' title='bbox 200 1100 600 1300; x_wconf 95'>안녕하세요</span>
|
||||
<span class='ocrx_word' id='word_1_8' title='bbox 650 1100 1050 1300; x_wconf 95'>세계</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,37 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="hi" lang="hi">
|
||||
<head>
|
||||
<title></title>
|
||||
<meta http-equiv="content-type" content="text/html; charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 5.0.0' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word'/>
|
||||
</head>
|
||||
<body>
|
||||
<div class='ocr_page' id='page_1' title='image "test.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
||||
<div class='ocr_carea' id='carea_1_1' title="bbox 200 200 2350 1200">
|
||||
<p class='ocr_par' id='par_1_1' lang='hin' title="bbox 200 200 2350 400">
|
||||
<span class='ocr_line' id='line_1_1' title="bbox 200 200 2350 400; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_1' title='bbox 200 200 600 400; x_wconf 95'>नमस्ते</span>
|
||||
<span class='ocrx_word' id='word_1_2' title='bbox 650 200 1050 400; x_wconf 95'>दुनिया</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_2' lang='hin' title="bbox 200 500 2350 700">
|
||||
<span class='ocr_line' id='line_1_2' title="bbox 200 500 2350 700; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_3' title='bbox 200 500 600 700; x_wconf 95'>यह</span>
|
||||
<span class='ocrx_word' id='word_1_4' title='bbox 650 500 1050 700; x_wconf 95'>हिंदी</span>
|
||||
<span class='ocrx_word' id='word_1_5' title='bbox 1100 500 1500 700; x_wconf 95'>पाठ</span>
|
||||
<span class='ocrx_word' id='word_1_6' title='bbox 1550 500 1950 700; x_wconf 95'>है</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_3' lang='san' title="bbox 200 800 2350 1000">
|
||||
<span class='ocr_line' id='line_1_3' title="bbox 200 800 2350 1000; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_7' title='bbox 200 800 700 1000; x_wconf 95'>संस्कृत</span>
|
||||
<span class='ocrx_word' id='word_1_8' title='bbox 750 800 1250 1000; x_wconf 95'>भाषा</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,192 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||
<head>
|
||||
<title>Multilingual Hello World Script Test</title>
|
||||
<meta http-equiv="content-type" content="text/html; charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 5.0.0' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word'/>
|
||||
</head>
|
||||
<body>
|
||||
<!-- Page: 8.5x11 inches at 300 DPI = 2550x3300 pixels -->
|
||||
<div class='ocr_page' id='page_1' title='image "hello_scripts.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
||||
|
||||
<!-- Row 1: English and Spanish (Latin script with accents/punctuation) -->
|
||||
<div class='ocr_carea' id='carea_1_1' title="bbox 150 150 1200 400">
|
||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 150 150 600 350">
|
||||
<span class='ocr_line' id='line_1_1' title="bbox 150 150 600 350; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_1_1' title='bbox 150 150 600 350; x_wconf 98'>Hello!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class='ocr_carea' id='carea_1_2' title="bbox 1400 150 2400 400">
|
||||
<p class='ocr_par' id='par_1_2' lang='spa' title="bbox 1400 150 2400 350">
|
||||
<span class='ocr_line' id='line_1_2' title="bbox 1400 150 2000 350; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_1_2' title='bbox 1400 150 2000 350; x_wconf 97'>¡Hola!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Row 2: French (accents) and German (umlauts, eszett) -->
|
||||
<div class='ocr_carea' id='carea_2_1' title="bbox 150 450 1200 700">
|
||||
<p class='ocr_par' id='par_2_1' lang='fra' title="bbox 150 450 800 650">
|
||||
<span class='ocr_line' id='line_2_1' title="bbox 150 450 800 650; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_2_1' title='bbox 150 450 800 650; x_wconf 96'>Bonjour!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class='ocr_carea' id='carea_2_2' title="bbox 1400 450 2400 700">
|
||||
<p class='ocr_par' id='par_2_2' lang='deu' title="bbox 1400 450 2100 650">
|
||||
<span class='ocr_line' id='line_2_2' title="bbox 1400 450 2100 650; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_2_2' title='bbox 1400 450 2100 650; x_wconf 95'>Grüß Gott!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Row 3: Russian (Cyrillic) and Greek -->
|
||||
<div class='ocr_carea' id='carea_3_1' title="bbox 150 750 1200 1000">
|
||||
<p class='ocr_par' id='par_3_1' lang='rus' title="bbox 150 750 900 950">
|
||||
<span class='ocr_line' id='line_3_1' title="bbox 150 750 900 950; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_3_1' title='bbox 150 750 900 950; x_wconf 94'>Привет!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class='ocr_carea' id='carea_3_2' title="bbox 1400 750 2400 1000">
|
||||
<p class='ocr_par' id='par_3_2' lang='ell' title="bbox 1400 750 2200 950">
|
||||
<span class='ocr_line' id='line_3_2' title="bbox 1400 750 2200 950; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_3_2' title='bbox 1400 750 2200 950; x_wconf 93'>Γειά σου!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Row 4: Chinese (Simplified) and Japanese -->
|
||||
<div class='ocr_carea' id='carea_4_1' title="bbox 150 1050 1200 1300">
|
||||
<p class='ocr_par' id='par_4_1' lang='chi_sim' title="bbox 150 1050 700 1250">
|
||||
<span class='ocr_line' id='line_4_1' title="bbox 150 1050 700 1250; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_4_1' title='bbox 150 1050 700 1250; x_wconf 92'>你好!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class='ocr_carea' id='carea_4_2' title="bbox 1400 1050 2400 1300">
|
||||
<p class='ocr_par' id='par_4_2' lang='jpn' title="bbox 1400 1050 2300 1250">
|
||||
<span class='ocr_line' id='line_4_2' title="bbox 1400 1050 2300 1250; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_4_2' title='bbox 1400 1050 2300 1250; x_wconf 91'>こんにちは!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Row 5: Korean and Turkish (Latin with special chars) -->
|
||||
<div class='ocr_carea' id='carea_5_1' title="bbox 150 1350 1200 1600">
|
||||
<p class='ocr_par' id='par_5_1' lang='kor' title="bbox 150 1350 900 1550">
|
||||
<span class='ocr_line' id='line_5_1' title="bbox 150 1350 900 1550; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_5_1' title='bbox 150 1350 900 1550; x_wconf 90'>안녕하세요!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class='ocr_carea' id='carea_5_2' title="bbox 1400 1350 2400 1600">
|
||||
<p class='ocr_par' id='par_5_2' lang='tur' title="bbox 1400 1350 2300 1550">
|
||||
<span class='ocr_line' id='line_5_2' title="bbox 1400 1350 2300 1550; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_5_2' title='bbox 1400 1350 2300 1550; x_wconf 89'>Merhaba!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Row 6: Hindi (Devanagari) and Arabic (RTL) -->
|
||||
<div class='ocr_carea' id='carea_6_1' title="bbox 150 1650 1200 1900">
|
||||
<p class='ocr_par' id='par_6_1' lang='hin' title="bbox 150 1650 900 1850">
|
||||
<span class='ocr_line' id='line_6_1' title="bbox 150 1650 900 1850; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_6_1' title='bbox 150 1650 900 1850; x_wconf 88'>नमस्ते!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class='ocr_carea' id='carea_6_2' title="bbox 1400 1650 2400 1900">
|
||||
<p class='ocr_par' id='par_6_2' lang='ara' dir='rtl' title="bbox 1400 1650 2300 1850">
|
||||
<span class='ocr_line' id='line_6_2' title="bbox 1400 1650 2300 1850; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_6_2' title='bbox 1400 1650 2300 1850; x_wconf 87'>!مرحبا</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Row 7: Hebrew (RTL) and Portuguese (accents) -->
|
||||
<div class='ocr_carea' id='carea_7_1' title="bbox 150 1950 1200 2200">
|
||||
<p class='ocr_par' id='par_7_1' lang='heb' dir='rtl' title="bbox 150 1950 800 2150">
|
||||
<span class='ocr_line' id='line_7_1' title="bbox 150 1950 800 2150; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_7_1' title='bbox 150 1950 800 2150; x_wconf 86'>שלום</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div class='ocr_carea' id='carea_7_2' title="bbox 1400 1950 2000 2200">
|
||||
<p class='ocr_par' id='par_7_2' lang='por' title="bbox 1400 1950 1900 2150">
|
||||
<span class='ocr_line' id='line_7_2' title="bbox 1400 1950 1900 2150; baseline 0 -40; x_size 140; x_descenders 28; x_ascenders 35">
|
||||
<span class='ocrx_word' id='word_7_2' title='bbox 1400 1950 1900 2150; x_wconf 85'>Olá!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Rotated text section: Various scripts at angles -->
|
||||
<!-- Rotated baseline: 15 degrees clockwise (baseline slope ~0.27) -->
|
||||
<div class='ocr_carea' id='carea_8_1' title="bbox 200 2150 900 2700">
|
||||
<p class='ocr_par' id='par_8_1' lang='ita' title="bbox 200 2150 900 2650">
|
||||
<span class='ocr_line' id='line_8_1' title="bbox 200 2150 900 2450; baseline 0.27 -30; x_size 130; x_descenders 26; x_ascenders 32">
|
||||
<span class='ocrx_word' id='word_8_1' title='bbox 200 2150 900 2450; x_wconf 84'>Ciao!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Rotated baseline: -10 degrees (baseline slope ~-0.18) -->
|
||||
<div class='ocr_carea' id='carea_8_2' title="bbox 1000 2350 1700 2700">
|
||||
<p class='ocr_par' id='par_8_2' lang='pol' title="bbox 1000 2400 1700 2650">
|
||||
<span class='ocr_line' id='line_8_2' title="bbox 1000 2400 1700 2650; baseline -0.18 -25; x_size 130; x_descenders 26; x_ascenders 32">
|
||||
<span class='ocrx_word' id='word_8_2' title='bbox 1000 2400 1700 2650; x_wconf 83'>Cześć!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Rotated baseline: 8 degrees clockwise (baseline slope ~0.14) - Chinese -->
|
||||
<div class='ocr_carea' id='carea_8_3' title="bbox 1800 2350 2450 2700">
|
||||
<p class='ocr_par' id='par_8_3' lang='chi_tra' title="bbox 1800 2400 2450 2650">
|
||||
<span class='ocr_line' id='line_8_3' title="bbox 1800 2400 2450 2650; baseline 0.14 -35; x_size 130; x_descenders 26; x_ascenders 32">
|
||||
<span class='ocrx_word' id='word_8_3' title='bbox 1800 2400 2450 2650; x_wconf 82'>您好!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Bottom row: More rotated examples -->
|
||||
<!-- Rotated baseline: -20 degrees (baseline slope ~-0.36) - Russian -->
|
||||
<div class='ocr_carea' id='carea_9_1' title="bbox 200 2750 900 3100">
|
||||
<p class='ocr_par' id='par_9_1' lang='rus' title="bbox 200 2800 900 3050">
|
||||
<span class='ocr_line' id='line_9_1' title="bbox 200 2800 900 3050; baseline -0.36 -20; x_size 120; x_descenders 24; x_ascenders 30">
|
||||
<span class='ocrx_word' id='word_9_1' title='bbox 200 2800 900 3050; x_wconf 81'>Здравствуй!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Rotated baseline: 12 degrees clockwise (baseline slope ~0.21) - Greek -->
|
||||
<div class='ocr_carea' id='carea_9_2' title="bbox 1000 2750 1700 3100">
|
||||
<p class='ocr_par' id='par_9_2' lang='ell' title="bbox 1000 2780 1700 3050">
|
||||
<span class='ocr_line' id='line_9_2' title="bbox 1000 2780 1700 3050; baseline 0.21 -30; x_size 120; x_descenders 24; x_ascenders 30">
|
||||
<span class='ocrx_word' id='word_9_2' title='bbox 1000 2780 1700 3050; x_wconf 80'>Χαίρετε!</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<!-- Rotated baseline: -5 degrees (baseline slope ~-0.09) - Arabic RTL rotated -->
|
||||
<div class='ocr_carea' id='carea_9_3' title="bbox 1800 2750 2450 3100">
|
||||
<p class='ocr_par' id='par_9_3' lang='ara' dir='rtl' title="bbox 1800 2800 2450 3050">
|
||||
<span class='ocr_line' id='line_9_3' title="bbox 1800 2800 2450 3050; baseline -0.09 -25; x_size 120; x_descenders 24; x_ascenders 30">
|
||||
<span class='ocrx_word' id='word_9_3' title='bbox 1800 2800 2450 3050; x_wconf 79'>!أهلاً</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,40 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||
<head>
|
||||
<title></title>
|
||||
<meta http-equiv="content-type" content="text/html; charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 5.0.0' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word'/>
|
||||
</head>
|
||||
<body>
|
||||
<div class='ocr_page' id='page_1' title='image "test.png"; bbox 0 0 2550 3300; ppageno 0; scan_res 300 300'>
|
||||
<div class='ocr_carea' id='carea_1_1' title="bbox 200 200 2350 1200">
|
||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 200 200 2350 400">
|
||||
<span class='ocr_line' id='line_1_1' title="bbox 200 200 2350 400; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_1' title='bbox 200 200 600 400; x_wconf 95'>The</span>
|
||||
<span class='ocrx_word' id='word_1_2' title='bbox 650 200 1050 400; x_wconf 95'>quick</span>
|
||||
<span class='ocrx_word' id='word_1_3' title='bbox 1100 200 1500 400; x_wconf 95'>brown</span>
|
||||
<span class='ocrx_word' id='word_1_4' title='bbox 1550 200 1850 400; x_wconf 95'>fox</span>
|
||||
<span class='ocrx_word' id='word_1_5' title='bbox 1900 200 2350 400; x_wconf 95'>jumps</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_2' lang='fra' title="bbox 200 500 2350 700">
|
||||
<span class='ocr_line' id='line_1_2' title="bbox 200 500 2350 700; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_6' title='bbox 200 500 500 700; x_wconf 95'>Café</span>
|
||||
<span class='ocrx_word' id='word_1_7' title='bbox 550 500 950 700; x_wconf 95'>résumé</span>
|
||||
<span class='ocrx_word' id='word_1_8' title='bbox 1000 500 1400 700; x_wconf 95'>naïve</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_3' lang='deu' title="bbox 200 800 2350 1000">
|
||||
<span class='ocr_line' id='line_1_3' title="bbox 200 800 2350 1000; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_9' title='bbox 200 800 700 1000; x_wconf 95'>Größe</span>
|
||||
<span class='ocrx_word' id='word_1_10' title='bbox 750 800 1250 1000; x_wconf 95'>Zürich</span>
|
||||
<span class='ocrx_word' id='word_1_11' title='bbox 1300 800 1800 1000; x_wconf 95'>Ärger</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,30 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||
<head>
|
||||
<title></title>
|
||||
<meta http-equiv="content-type" content="text/html; charset=utf-8" />
|
||||
<meta name='ocr-system' content='tesseract 5.0.0' />
|
||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word'/>
|
||||
</head>
|
||||
<body>
|
||||
<div class='ocr_page' id='page_1' title='image "test.png"; bbox 0 0 2550 3300; ppageno 0'>
|
||||
<div class='ocr_carea' id='carea_1_1' title="bbox 200 200 2350 800">
|
||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 200 200 2350 400">
|
||||
<span class='ocr_line' id='line_1_1' title="bbox 200 200 2350 400; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_1' title='bbox 200 200 500 400; x_wconf 95'>English</span>
|
||||
<span class='ocrx_word' id='word_1_2' title='bbox 550 200 750 400; x_wconf 95'>Text</span>
|
||||
<span class='ocrx_word' id='word_1_3' title='bbox 800 200 1000 400; x_wconf 95'>Here</span>
|
||||
</span>
|
||||
</p>
|
||||
<p class='ocr_par' id='par_1_2' lang='ara' dir='rtl' title="bbox 200 500 2350 800">
|
||||
<span class='ocr_line' id='line_1_2' title="bbox 200 500 2350 800; baseline 0 -50; x_size 150; x_descenders 30; x_ascenders 40">
|
||||
<span class='ocrx_word' id='word_1_4' title='bbox 200 500 600 800; x_wconf 95'>مرحبا</span>
|
||||
<span class='ocrx_word' id='word_1_5' title='bbox 650 500 950 800; x_wconf 95'>بك</span>
|
||||
</span>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,366 @@
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Tests for fpdf2-based PDF renderer."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from ocrmypdf.font import MultiFontManager
|
||||
from ocrmypdf.fpdf_renderer import DebugRenderOptions, Fpdf2MultiPageRenderer, Fpdf2PdfRenderer
|
||||
from ocrmypdf.hocrtransform.hocr_parser import HocrParser
|
||||
from ocrmypdf.hocrtransform.ocr_element import OcrClass
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def font_dir():
|
||||
"""Return path to font directory."""
|
||||
return Path(__file__).parent.parent / "src" / "ocrmypdf" / "data"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def multi_font_manager(font_dir):
|
||||
"""Create MultiFontManager instance for testing."""
|
||||
return MultiFontManager(font_dir)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def resources():
|
||||
"""Return path to test resources directory."""
|
||||
return Path(__file__).parent / "resources"
|
||||
|
||||
|
||||
class TestFpdf2RendererImports:
|
||||
"""Test that all fpdf2 renderer modules can be imported."""
|
||||
|
||||
def test_imports(self):
|
||||
"""Test that all fpdf_renderer modules can be imported."""
|
||||
from ocrmypdf.fpdf_renderer import (
|
||||
DebugRenderOptions,
|
||||
Fpdf2MultiPageRenderer,
|
||||
Fpdf2PdfRenderer,
|
||||
)
|
||||
assert DebugRenderOptions is not None
|
||||
assert Fpdf2PdfRenderer is not None
|
||||
assert Fpdf2MultiPageRenderer is not None
|
||||
|
||||
|
||||
class TestDebugRenderOptions:
|
||||
"""Test DebugRenderOptions dataclass."""
|
||||
|
||||
def test_defaults(self):
|
||||
"""Test default values."""
|
||||
opts = DebugRenderOptions()
|
||||
assert opts.render_baseline is False
|
||||
assert opts.render_line_bbox is False
|
||||
assert opts.render_word_bbox is False
|
||||
|
||||
def test_custom_values(self):
|
||||
"""Test custom values."""
|
||||
opts = DebugRenderOptions(
|
||||
render_baseline=True,
|
||||
render_line_bbox=True,
|
||||
render_word_bbox=True,
|
||||
)
|
||||
assert opts.render_baseline is True
|
||||
assert opts.render_line_bbox is True
|
||||
assert opts.render_word_bbox is True
|
||||
|
||||
|
||||
class TestFpdf2PdfRenderer:
|
||||
"""Test Fpdf2PdfRenderer."""
|
||||
|
||||
def test_requires_page_element(self, multi_font_manager):
|
||||
"""Test that renderer requires ocr_page element."""
|
||||
from ocrmypdf.hocrtransform.ocr_element import BoundingBox, OcrElement
|
||||
|
||||
# Create a non-page element
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
text="test",
|
||||
bbox=BoundingBox(left=0, top=0, right=100, bottom=20),
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="Root element must be ocr_page"):
|
||||
Fpdf2PdfRenderer(
|
||||
page=word,
|
||||
dpi=300,
|
||||
multi_font_manager=multi_font_manager,
|
||||
)
|
||||
|
||||
def test_requires_bbox(self, multi_font_manager):
|
||||
"""Test that renderer requires page with bounding box."""
|
||||
from ocrmypdf.hocrtransform.ocr_element import OcrElement
|
||||
|
||||
page = OcrElement(ocr_class=OcrClass.PAGE)
|
||||
|
||||
with pytest.raises(ValueError, match="Page must have bounding box"):
|
||||
Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300,
|
||||
multi_font_manager=multi_font_manager,
|
||||
)
|
||||
|
||||
def test_render_simple_page(self, multi_font_manager, tmp_path):
|
||||
"""Test rendering a simple page with one word."""
|
||||
from ocrmypdf.hocrtransform.ocr_element import BoundingBox, OcrElement
|
||||
|
||||
# Create a simple page with one word
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
text="Hello",
|
||||
bbox=BoundingBox(left=100, top=100, right=200, bottom=130),
|
||||
)
|
||||
line = OcrElement(
|
||||
ocr_class=OcrClass.LINE,
|
||||
bbox=BoundingBox(left=100, top=100, right=200, bottom=130),
|
||||
children=[word],
|
||||
)
|
||||
page = OcrElement(
|
||||
ocr_class=OcrClass.PAGE,
|
||||
bbox=BoundingBox(left=0, top=0, right=612, bottom=792),
|
||||
children=[line],
|
||||
)
|
||||
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=72, # 1:1 mapping to PDF points
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
|
||||
output_path = tmp_path / "test_simple.pdf"
|
||||
renderer.render(output_path)
|
||||
|
||||
assert output_path.exists()
|
||||
assert output_path.stat().st_size > 0
|
||||
|
||||
def test_render_invisible_text(self, multi_font_manager, tmp_path):
|
||||
"""Test rendering invisible text (OCR layer)."""
|
||||
from ocrmypdf.hocrtransform.ocr_element import BoundingBox, OcrElement
|
||||
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
text="Invisible",
|
||||
bbox=BoundingBox(left=100, top=100, right=250, bottom=130),
|
||||
)
|
||||
line = OcrElement(
|
||||
ocr_class=OcrClass.LINE,
|
||||
bbox=BoundingBox(left=100, top=100, right=250, bottom=130),
|
||||
children=[word],
|
||||
)
|
||||
page = OcrElement(
|
||||
ocr_class=OcrClass.PAGE,
|
||||
bbox=BoundingBox(left=0, top=0, right=612, bottom=792),
|
||||
children=[line],
|
||||
)
|
||||
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=72,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=True, # This is the default
|
||||
)
|
||||
|
||||
output_path = tmp_path / "test_invisible.pdf"
|
||||
renderer.render(output_path)
|
||||
|
||||
assert output_path.exists()
|
||||
assert output_path.stat().st_size > 0
|
||||
|
||||
|
||||
class TestFpdf2MultiPageRenderer:
|
||||
"""Test Fpdf2MultiPageRenderer."""
|
||||
|
||||
def test_requires_pages(self, multi_font_manager):
|
||||
"""Test that renderer requires at least one page."""
|
||||
with pytest.raises(ValueError, match="No pages to render"):
|
||||
renderer = Fpdf2MultiPageRenderer(
|
||||
pages_data=[],
|
||||
multi_font_manager=multi_font_manager,
|
||||
)
|
||||
renderer.render(Path("/tmp/test.pdf"))
|
||||
|
||||
def test_render_multiple_pages(self, multi_font_manager, tmp_path):
|
||||
"""Test rendering multiple pages."""
|
||||
from ocrmypdf.hocrtransform.ocr_element import BoundingBox, OcrElement
|
||||
|
||||
pages_data = []
|
||||
for i in range(3):
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
text=f"Page{i+1}",
|
||||
bbox=BoundingBox(left=100, top=100, right=200, bottom=130),
|
||||
)
|
||||
line = OcrElement(
|
||||
ocr_class=OcrClass.LINE,
|
||||
bbox=BoundingBox(left=100, top=100, right=200, bottom=130),
|
||||
children=[word],
|
||||
)
|
||||
page = OcrElement(
|
||||
ocr_class=OcrClass.PAGE,
|
||||
bbox=BoundingBox(left=0, top=0, right=612, bottom=792),
|
||||
children=[line],
|
||||
)
|
||||
pages_data.append((i + 1, page, 72))
|
||||
|
||||
renderer = Fpdf2MultiPageRenderer(
|
||||
pages_data=pages_data,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
|
||||
output_path = tmp_path / "test_multipage.pdf"
|
||||
renderer.render(output_path)
|
||||
|
||||
assert output_path.exists()
|
||||
assert output_path.stat().st_size > 0
|
||||
|
||||
|
||||
class TestFpdf2RendererWithHocr:
|
||||
"""Test fpdf2 renderer with actual hOCR files."""
|
||||
|
||||
def test_render_latin_hocr(self, resources, multi_font_manager, tmp_path):
|
||||
"""Test rendering Latin text from hOCR."""
|
||||
hocr_path = resources / "latin.hocr"
|
||||
if not hocr_path.exists():
|
||||
pytest.skip("latin.hocr not found")
|
||||
|
||||
parser = HocrParser(hocr_path)
|
||||
page = parser.parse()
|
||||
|
||||
# Ensure we got a page
|
||||
assert page.ocr_class == OcrClass.PAGE
|
||||
assert page.bbox is not None
|
||||
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
|
||||
output_path = tmp_path / "latin_fpdf2.pdf"
|
||||
renderer.render(output_path)
|
||||
|
||||
assert output_path.exists()
|
||||
assert output_path.stat().st_size > 0
|
||||
|
||||
def test_render_cjk_hocr(self, resources, multi_font_manager, tmp_path):
|
||||
"""Test rendering CJK text from hOCR."""
|
||||
hocr_path = resources / "cjk.hocr"
|
||||
if not hocr_path.exists():
|
||||
pytest.skip("cjk.hocr not found")
|
||||
|
||||
parser = HocrParser(hocr_path)
|
||||
page = parser.parse()
|
||||
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
|
||||
output_path = tmp_path / "cjk_fpdf2.pdf"
|
||||
renderer.render(output_path)
|
||||
|
||||
assert output_path.exists()
|
||||
assert output_path.stat().st_size > 0
|
||||
|
||||
def test_render_arabic_hocr(self, resources, multi_font_manager, tmp_path):
|
||||
"""Test rendering Arabic text from hOCR."""
|
||||
hocr_path = resources / "arabic.hocr"
|
||||
if not hocr_path.exists():
|
||||
pytest.skip("arabic.hocr not found")
|
||||
|
||||
parser = HocrParser(hocr_path)
|
||||
page = parser.parse()
|
||||
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
|
||||
output_path = tmp_path / "arabic_fpdf2.pdf"
|
||||
renderer.render(output_path)
|
||||
|
||||
assert output_path.exists()
|
||||
assert output_path.stat().st_size > 0
|
||||
|
||||
def test_render_hello_world_scripts_hocr(self, resources, multi_font_manager, tmp_path):
|
||||
"""Test rendering comprehensive multilingual 'Hello!' hOCR file.
|
||||
|
||||
This tests all major scripts including:
|
||||
- Latin (English, Spanish, French, German, Italian, Polish, Portuguese, Turkish)
|
||||
- Cyrillic (Russian)
|
||||
- Greek
|
||||
- CJK (Chinese Simplified, Chinese Traditional, Japanese, Korean)
|
||||
- Devanagari (Hindi)
|
||||
- Arabic (RTL)
|
||||
- Hebrew (RTL)
|
||||
|
||||
Also includes rotated baselines to exercise skew handling.
|
||||
"""
|
||||
hocr_path = resources / "hello_world_scripts.hocr"
|
||||
if not hocr_path.exists():
|
||||
pytest.skip("hello_world_scripts.hocr not found")
|
||||
|
||||
parser = HocrParser(hocr_path)
|
||||
page = parser.parse()
|
||||
|
||||
# Verify we parsed the page correctly
|
||||
assert page.ocr_class == OcrClass.PAGE
|
||||
assert page.bbox is not None
|
||||
# Should have 2550x3300 at 300 DPI
|
||||
assert page.bbox.right == 2550
|
||||
assert page.bbox.bottom == 3300
|
||||
|
||||
# Test with visible text for visual inspection
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
|
||||
output_path = tmp_path / "hello_world_scripts_fpdf2.pdf"
|
||||
renderer.render(output_path)
|
||||
|
||||
assert output_path.exists()
|
||||
assert output_path.stat().st_size > 0
|
||||
|
||||
def test_render_hello_world_scripts_multipage(
|
||||
self, resources, multi_font_manager, tmp_path
|
||||
):
|
||||
"""Test rendering hello_world_scripts.hocr using MultiPageRenderer.
|
||||
|
||||
Uses Fpdf2MultiPageRenderer to render the multilingual test file,
|
||||
demonstrating font handling across all major writing systems.
|
||||
"""
|
||||
hocr_path = resources / "hello_world_scripts.hocr"
|
||||
if not hocr_path.exists():
|
||||
pytest.skip("hello_world_scripts.hocr not found")
|
||||
|
||||
parser = HocrParser(hocr_path)
|
||||
page = parser.parse()
|
||||
|
||||
# Build pages_data list as expected by MultiPageRenderer
|
||||
pages_data = [(1, page, 300)] # (page_number, page_element, dpi)
|
||||
|
||||
renderer = Fpdf2MultiPageRenderer(
|
||||
pages_data=pages_data,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
|
||||
output_path = tmp_path / "hello_world_scripts_multipage.pdf"
|
||||
renderer.render(output_path)
|
||||
|
||||
assert output_path.exists()
|
||||
assert output_path.stat().st_size > 0
|
||||
+41
-19
@@ -5,6 +5,7 @@ from __future__ import annotations
|
||||
|
||||
import re
|
||||
from io import StringIO
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from pdfminer.converter import TextConverter
|
||||
@@ -15,9 +16,11 @@ from pdfminer.pdfpage import PDFPage
|
||||
from pdfminer.pdfparser import PDFParser
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf import hocrtransform
|
||||
from ocrmypdf._exec.tesseract import generate_hocr
|
||||
from ocrmypdf.font import MultiFontManager
|
||||
from ocrmypdf.fpdf_renderer import Fpdf2PdfRenderer
|
||||
from ocrmypdf.helpers import check_pdf
|
||||
from ocrmypdf.hocrtransform import HocrParser
|
||||
|
||||
from .conftest import check_ocrmypdf
|
||||
|
||||
@@ -38,6 +41,18 @@ def text_from_pdf(filename):
|
||||
# pylint: disable=redefined-outer-name
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def font_dir():
|
||||
"""Get the font directory."""
|
||||
return Path(__file__).parent.parent / "src" / "ocrmypdf" / "data"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def multi_font_manager(font_dir):
|
||||
"""Create a MultiFontManager for tests."""
|
||||
return MultiFontManager(font_dir)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def blank_hocr(tmp_path):
|
||||
im = Image.new('1', (8, 8), 0)
|
||||
@@ -58,23 +73,38 @@ def blank_hocr(tmp_path):
|
||||
return tmp_path / 'blank.hocr'
|
||||
|
||||
|
||||
def test_mono_image(blank_hocr, outdir):
|
||||
def test_mono_image(blank_hocr, outdir, multi_font_manager):
|
||||
im = Image.new('1', (8, 8), 0)
|
||||
for n in range(8):
|
||||
im.putpixel((n, n), 1)
|
||||
im.save(outdir / 'mono.tif', format='TIFF')
|
||||
|
||||
hocr = hocrtransform.HocrTransform(hocr_filename=str(blank_hocr), dpi=8)
|
||||
hocr.to_pdf(
|
||||
out_filename=str(outdir / 'mono.pdf'), image_filename=str(outdir / 'mono.tif')
|
||||
# Parse hOCR file
|
||||
parser = HocrParser(str(blank_hocr))
|
||||
ocr_page = parser.parse()
|
||||
|
||||
# Use DPI from hOCR or default
|
||||
dpi = ocr_page.dpi or 8
|
||||
|
||||
# Render to PDF using fpdf2
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=ocr_page,
|
||||
dpi=dpi,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=True,
|
||||
)
|
||||
# shutil.copy(outdir / 'mono.pdf', 'mono.pdf')
|
||||
check_pdf(str(outdir / 'mono.pdf'))
|
||||
renderer.render(outdir / 'mono.pdf')
|
||||
|
||||
check_pdf(outdir / 'mono.pdf')
|
||||
|
||||
|
||||
@pytest.mark.slow
|
||||
def test_hocrtransform_matches_sandwich(resources, outdir):
|
||||
check_ocrmypdf(resources / 'ccitt.pdf', outdir / 'hocr.pdf', '--pdf-renderer=hocr')
|
||||
def test_fpdf2_matches_sandwich(resources, outdir):
|
||||
"""Test that fpdf2 renderer produces similar output to sandwich renderer."""
|
||||
# Note: hocr renderer now redirects to fpdf2
|
||||
check_ocrmypdf(
|
||||
resources / 'ccitt.pdf', outdir / 'fpdf2.pdf', '--pdf-renderer=fpdf2'
|
||||
)
|
||||
check_ocrmypdf(
|
||||
resources / 'ccitt.pdf', outdir / 'tess.pdf', '--pdf-renderer=sandwich'
|
||||
)
|
||||
@@ -86,17 +116,9 @@ def test_hocrtransform_matches_sandwich(resources, outdir):
|
||||
words = s.split(' ')
|
||||
return set(words)
|
||||
|
||||
hocr_words = clean(text_from_pdf(outdir / 'hocr.pdf'))
|
||||
fpdf2_words = clean(text_from_pdf(outdir / 'fpdf2.pdf'))
|
||||
tess_words = clean(text_from_pdf(outdir / 'tess.pdf'))
|
||||
|
||||
similarity = len(hocr_words & tess_words) / len(hocr_words | tess_words)
|
||||
|
||||
# from pathlib import Path
|
||||
|
||||
# Path('hocr.txt').write_text(sorted('\n'.join(hocr_words)))
|
||||
# Path('tess.txt').write_text(sorted('\n'.join(tess_words)))
|
||||
# Path('mismatch.txt').write_text(
|
||||
# '\n'.join(sorted(hocr_words ^ tess_words)), encoding='utf8'
|
||||
# )
|
||||
similarity = len(fpdf2_words & tess_words) / len(fpdf2_words | tess_words)
|
||||
|
||||
assert similarity > 0.99
|
||||
|
||||
+2
-2
@@ -35,7 +35,7 @@ from .conftest import (
|
||||
# pylint: disable=redefined-outer-name
|
||||
|
||||
|
||||
RENDERERS = ['hocr', 'sandwich']
|
||||
RENDERERS = ['fpdf2', 'sandwich']
|
||||
|
||||
|
||||
def test_quick(resources, outpdf):
|
||||
@@ -435,7 +435,7 @@ def test_jbig2_passthrough(resources, outpdf):
|
||||
'--output-type',
|
||||
'pdf',
|
||||
'--pdf-renderer',
|
||||
'hocr',
|
||||
'fpdf2',
|
||||
'--plugin',
|
||||
'tests/plugins/tesseract_cache.py',
|
||||
)
|
||||
|
||||
@@ -0,0 +1,446 @@
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Unit tests for MultiFontManager and FontProvider."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from ocrmypdf.font import BuiltinFontProvider, FontManager, MultiFontManager
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def font_dir():
|
||||
"""Return path to font directory."""
|
||||
return Path(__file__).parent.parent / "src" / "ocrmypdf" / "data"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def multi_font_manager(font_dir):
|
||||
"""Create MultiFontManager instance for testing."""
|
||||
return MultiFontManager(font_dir)
|
||||
|
||||
|
||||
def has_cjk_font(manager: MultiFontManager) -> bool:
|
||||
"""Check if CJK font is available (from system)."""
|
||||
return 'NotoSansCJK-Regular' in manager.fonts
|
||||
|
||||
|
||||
def has_arabic_font(manager: MultiFontManager) -> bool:
|
||||
"""Check if Arabic font is available (from system)."""
|
||||
return 'NotoSansArabic-Regular' in manager.fonts
|
||||
|
||||
|
||||
def has_devanagari_font(manager: MultiFontManager) -> bool:
|
||||
"""Check if Devanagari font is available (from system)."""
|
||||
return 'NotoSansDevanagari-Regular' in manager.fonts
|
||||
|
||||
|
||||
# Marker for tests that require CJK fonts
|
||||
requires_cjk = pytest.mark.skipif(
|
||||
"not has_cjk_font(MultiFontManager())",
|
||||
reason="CJK font not available (not installed on system)"
|
||||
)
|
||||
|
||||
|
||||
# --- MultiFontManager Initialization Tests ---
|
||||
|
||||
|
||||
def test_init_loads_builtin_fonts(multi_font_manager):
|
||||
"""Test that initialization loads all expected builtin fonts."""
|
||||
# Only NotoSans-Regular and Occulta are bundled
|
||||
assert 'NotoSans-Regular' in multi_font_manager.fonts
|
||||
assert 'Occulta' in multi_font_manager.fonts
|
||||
|
||||
# At least 2 builtin fonts should be loaded
|
||||
assert len(multi_font_manager.fonts) >= 2
|
||||
|
||||
# Arabic, Devanagari, CJK are optional (system fonts)
|
||||
|
||||
|
||||
def test_missing_font_directory():
|
||||
"""Test that missing font directory raises error for fallback font."""
|
||||
with pytest.raises(FileNotFoundError):
|
||||
MultiFontManager(Path("/nonexistent/path"))
|
||||
|
||||
|
||||
# --- Arabic Script Language Tests ---
|
||||
# These tests require Arabic fonts to be installed on the system
|
||||
|
||||
|
||||
def test_select_font_for_arabic_language(multi_font_manager):
|
||||
"""Test font selection with Arabic language hint."""
|
||||
if not has_arabic_font(multi_font_manager):
|
||||
pytest.skip("Arabic font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("مرحبا", "ara")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansArabic-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_persian_language(multi_font_manager):
|
||||
"""Test font selection with Persian language hint."""
|
||||
if not has_arabic_font(multi_font_manager):
|
||||
pytest.skip("Arabic font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("سلام", "per")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansArabic-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_urdu_language(multi_font_manager):
|
||||
"""Test font selection with Urdu language hint."""
|
||||
if not has_arabic_font(multi_font_manager):
|
||||
pytest.skip("Arabic font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("ہیلو", "urd")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansArabic-Regular']
|
||||
|
||||
|
||||
def test_farsi_language_code(multi_font_manager):
|
||||
"""Test that 'fas' (Farsi alternative code) maps to Arabic font."""
|
||||
if not has_arabic_font(multi_font_manager):
|
||||
pytest.skip("Arabic font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("سلام", "fas")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansArabic-Regular']
|
||||
|
||||
|
||||
# --- Devanagari Script Language Tests ---
|
||||
# These tests require Devanagari fonts to be installed on the system
|
||||
|
||||
|
||||
def test_select_font_for_hindi_language(multi_font_manager):
|
||||
"""Test font selection with Hindi language hint."""
|
||||
if not has_devanagari_font(multi_font_manager):
|
||||
pytest.skip("Devanagari font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("नमस्ते", "hin")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansDevanagari-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_sanskrit_language(multi_font_manager):
|
||||
"""Test font selection with Sanskrit language hint."""
|
||||
if not has_devanagari_font(multi_font_manager):
|
||||
pytest.skip("Devanagari font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("संस्कृतम्", "san")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansDevanagari-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_marathi_language(multi_font_manager):
|
||||
"""Test font selection with Marathi language hint."""
|
||||
if not has_devanagari_font(multi_font_manager):
|
||||
pytest.skip("Devanagari font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("मराठी", "mar")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansDevanagari-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_nepali_language(multi_font_manager):
|
||||
"""Test font selection with Nepali language hint."""
|
||||
if not has_devanagari_font(multi_font_manager):
|
||||
pytest.skip("Devanagari font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("नेपाली", "nep")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansDevanagari-Regular']
|
||||
|
||||
|
||||
# --- CJK Language Tests ---
|
||||
# These tests require CJK fonts to be installed on the system
|
||||
|
||||
|
||||
def test_select_font_for_chinese_language(multi_font_manager):
|
||||
"""Test font selection with Chinese language hint (ISO 639-3)."""
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("你好", "zho")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_chinese_generic(multi_font_manager):
|
||||
"""Test font selection with generic Chinese language code."""
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("中文", "chi")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_chinese_simplified(multi_font_manager):
|
||||
"""Test font selection with Tesseract's chi_sim language code."""
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("简体字", "chi_sim")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_chinese_traditional(multi_font_manager):
|
||||
"""Test font selection with Tesseract's chi_tra language code."""
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("漢字", "chi_tra")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_japanese_language(multi_font_manager):
|
||||
"""Test font selection with Japanese language hint."""
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("こんにちは", "jpn")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
|
||||
|
||||
def test_select_font_for_korean_language(multi_font_manager):
|
||||
"""Test font selection with Korean language hint."""
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("안녕하세요", "kor")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
|
||||
|
||||
# --- Latin/English Tests ---
|
||||
|
||||
|
||||
def test_select_font_for_english_text(multi_font_manager):
|
||||
"""Test font selection for English text."""
|
||||
font_manager = multi_font_manager.select_font_for_word("Hello World", "eng")
|
||||
assert font_manager == multi_font_manager.fonts['NotoSans-Regular']
|
||||
|
||||
|
||||
def test_select_font_without_language_hint(multi_font_manager):
|
||||
"""Test font selection without language hint falls back to glyph checking."""
|
||||
font_manager = multi_font_manager.select_font_for_word("Hello", None)
|
||||
assert font_manager == multi_font_manager.fonts['NotoSans-Regular']
|
||||
|
||||
|
||||
# --- Fallback Behavior Tests ---
|
||||
|
||||
|
||||
def test_select_font_arabic_text_without_language_hint(multi_font_manager):
|
||||
"""Test that Arabic text is handled via fallback without language hint."""
|
||||
if not has_arabic_font(multi_font_manager):
|
||||
pytest.skip("Arabic font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("مرحبا", None)
|
||||
# Should get NotoSansArabic-Regular via fallback chain glyph checking
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansArabic-Regular']
|
||||
|
||||
|
||||
def test_devanagari_text_without_language_hint(multi_font_manager):
|
||||
"""Test that Devanagari text is handled via fallback without language hint."""
|
||||
# NotoSans-Regular includes Devanagari glyphs, so it's selected first in fallback
|
||||
font_manager = multi_font_manager.select_font_for_word("नमस्ते", None)
|
||||
# Could be NotoSans-Regular or NotoSansDevanagari-Regular depending on availability
|
||||
assert font_manager is not None
|
||||
|
||||
|
||||
def test_cjk_text_without_language_hint(multi_font_manager):
|
||||
"""Test that CJK text is handled via fallback without language hint."""
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
font_manager = multi_font_manager.select_font_for_word("你好", None)
|
||||
assert font_manager == multi_font_manager.fonts['NotoSansCJK-Regular']
|
||||
|
||||
|
||||
def test_fallback_to_occulta_font(multi_font_manager):
|
||||
"""Test that unsupported characters fall back to Occulta.ttf."""
|
||||
# Use a character unlikely to be in any standard font
|
||||
font_manager = multi_font_manager.select_font_for_word("test", "xyz")
|
||||
# Should return some valid font
|
||||
assert font_manager in multi_font_manager.fonts.values()
|
||||
|
||||
|
||||
def test_fallback_fonts_constant(multi_font_manager):
|
||||
"""Test that FALLBACK_FONTS contains expected fonts."""
|
||||
# Check that core fonts are in fallback list
|
||||
assert 'NotoSans-Regular' in MultiFontManager.FALLBACK_FONTS
|
||||
assert 'NotoSansArabic-Regular' in MultiFontManager.FALLBACK_FONTS
|
||||
assert 'NotoSansDevanagari-Regular' in MultiFontManager.FALLBACK_FONTS
|
||||
assert 'NotoSansCJK-Regular' in MultiFontManager.FALLBACK_FONTS
|
||||
|
||||
# Only NotoSans-Regular is bundled; other scripts are system fonts
|
||||
assert 'NotoSans-Regular' in multi_font_manager.fonts
|
||||
|
||||
|
||||
# --- Glyph Coverage Tests ---
|
||||
|
||||
|
||||
def test_has_all_glyphs_for_english(multi_font_manager):
|
||||
"""Test glyph coverage checking for English text."""
|
||||
assert multi_font_manager.has_all_glyphs('NotoSans-Regular', "Hello World")
|
||||
assert multi_font_manager.has_all_glyphs('NotoSans-Regular', "café")
|
||||
|
||||
|
||||
def test_has_all_glyphs_for_arabic(multi_font_manager):
|
||||
"""Test glyph coverage checking for Arabic text."""
|
||||
if not has_arabic_font(multi_font_manager):
|
||||
pytest.skip("Arabic font not available")
|
||||
assert multi_font_manager.has_all_glyphs('NotoSansArabic-Regular', "مرحبا")
|
||||
|
||||
|
||||
def test_has_all_glyphs_for_devanagari(multi_font_manager):
|
||||
"""Test glyph coverage checking for Devanagari text."""
|
||||
if not has_devanagari_font(multi_font_manager):
|
||||
pytest.skip("Devanagari font not available")
|
||||
assert multi_font_manager.has_all_glyphs('NotoSansDevanagari-Regular', "नमस्ते")
|
||||
|
||||
|
||||
def test_has_all_glyphs_for_cjk(multi_font_manager):
|
||||
"""Test glyph coverage checking for CJK text."""
|
||||
if not has_cjk_font(multi_font_manager):
|
||||
pytest.skip("CJK font not available")
|
||||
assert multi_font_manager.has_all_glyphs('NotoSansCJK-Regular', "你好")
|
||||
|
||||
|
||||
def test_empty_text_has_all_glyphs(multi_font_manager):
|
||||
"""Test that empty text returns True for glyph coverage."""
|
||||
assert multi_font_manager.has_all_glyphs('NotoSans-Regular', "")
|
||||
|
||||
|
||||
def test_has_all_glyphs_missing_font(multi_font_manager):
|
||||
"""Test that has_all_glyphs returns False for non-existent font."""
|
||||
assert not multi_font_manager.has_all_glyphs('NonExistentFont', "test")
|
||||
|
||||
|
||||
# --- Caching Tests ---
|
||||
|
||||
|
||||
def test_font_selection_caching(multi_font_manager):
|
||||
"""Test that font selection results are cached."""
|
||||
font1 = multi_font_manager.select_font_for_word("Hello", "eng")
|
||||
|
||||
cache_key = ("Hello", "eng")
|
||||
assert cache_key in multi_font_manager._selection_cache
|
||||
|
||||
font2 = multi_font_manager.select_font_for_word("Hello", "eng")
|
||||
assert font1 == font2
|
||||
|
||||
|
||||
# --- Language Font Map Tests ---
|
||||
|
||||
|
||||
def test_language_font_map_coverage():
|
||||
"""Test that LANGUAGE_FONT_MAP has valid structure."""
|
||||
# Only NotoSans-Regular is bundled now
|
||||
# This test just verifies the structure is valid
|
||||
for font_name in MultiFontManager.LANGUAGE_FONT_MAP.values():
|
||||
# All font names should be valid strings
|
||||
assert isinstance(font_name, str)
|
||||
assert font_name.startswith('NotoSans')
|
||||
|
||||
|
||||
# --- get_all_fonts Tests ---
|
||||
|
||||
|
||||
def test_get_all_fonts(multi_font_manager):
|
||||
"""Test get_all_fonts returns all loaded fonts."""
|
||||
all_fonts = multi_font_manager.get_all_fonts()
|
||||
assert isinstance(all_fonts, dict)
|
||||
# At least 2 builtin fonts should be loaded (NotoSans-Regular and Occulta)
|
||||
assert len(all_fonts) >= 2
|
||||
assert 'NotoSans-Regular' in all_fonts
|
||||
assert 'Occulta' in all_fonts
|
||||
# Arabic, Devanagari, CJK are optional (system fonts)
|
||||
|
||||
|
||||
# --- FontProvider Tests ---
|
||||
|
||||
|
||||
class MockFontProvider:
|
||||
"""Mock FontProvider for testing missing fonts."""
|
||||
|
||||
def __init__(
|
||||
self, available_fonts: dict[str, FontManager], fallback: FontManager
|
||||
):
|
||||
"""Initialize mock font provider with given fonts."""
|
||||
self._fonts = available_fonts
|
||||
self._fallback = fallback
|
||||
|
||||
def get_font(self, font_name: str) -> FontManager | None:
|
||||
return self._fonts.get(font_name)
|
||||
|
||||
def get_available_fonts(self) -> list[str]:
|
||||
return list(self._fonts.keys())
|
||||
|
||||
def get_fallback_font(self) -> FontManager:
|
||||
return self._fallback
|
||||
|
||||
|
||||
def test_custom_font_provider(font_dir):
|
||||
"""Test that custom FontProvider can be injected."""
|
||||
fonts = {
|
||||
'NotoSans-Regular': FontManager(font_dir / 'NotoSans-Regular.ttf'),
|
||||
'Occulta': FontManager(font_dir / 'Occulta.ttf'),
|
||||
}
|
||||
provider = MockFontProvider(fonts, fonts['Occulta'])
|
||||
|
||||
manager = MultiFontManager(font_provider=provider)
|
||||
|
||||
# Should only have the fonts we provided
|
||||
assert len(manager.fonts) == 2
|
||||
assert 'NotoSans-Regular' in manager.fonts
|
||||
assert 'Occulta' in manager.fonts
|
||||
|
||||
|
||||
def test_missing_font_uses_fallback(font_dir):
|
||||
"""Test that missing fonts gracefully fall back."""
|
||||
fonts = {
|
||||
'NotoSans-Regular': FontManager(font_dir / 'NotoSans-Regular.ttf'),
|
||||
'Occulta': FontManager(font_dir / 'Occulta.ttf'),
|
||||
}
|
||||
provider = MockFontProvider(fonts, fonts['Occulta'])
|
||||
|
||||
manager = MultiFontManager(font_provider=provider)
|
||||
|
||||
# Arabic text should fall back to Occulta since NotoSansArabic is missing
|
||||
font = manager.select_font_for_word("مرحبا", "ara")
|
||||
assert font == fonts['Occulta']
|
||||
|
||||
|
||||
def test_builtin_font_provider_loads_expected_fonts(font_dir):
|
||||
"""Test BuiltinFontProvider loads all expected builtin fonts."""
|
||||
provider = BuiltinFontProvider(font_dir)
|
||||
|
||||
available = provider.get_available_fonts()
|
||||
assert 'NotoSans-Regular' in available
|
||||
assert 'Occulta' in available
|
||||
# Only Latin (NotoSans) and glyphless fallback (Occulta) are bundled.
|
||||
# All other scripts (Arabic, Devanagari, CJK, etc.) are discovered
|
||||
# from system fonts by SystemFontProvider to reduce package size.
|
||||
assert len(available) == 2
|
||||
|
||||
|
||||
def test_builtin_font_provider_get_font(font_dir):
|
||||
"""Test BuiltinFontProvider.get_font returns correct fonts."""
|
||||
provider = BuiltinFontProvider(font_dir)
|
||||
|
||||
font = provider.get_font('NotoSans-Regular')
|
||||
assert font is not None
|
||||
assert isinstance(font, FontManager)
|
||||
|
||||
missing = provider.get_font('NonExistent')
|
||||
assert missing is None
|
||||
|
||||
|
||||
def test_builtin_font_provider_get_fallback(font_dir):
|
||||
"""Test BuiltinFontProvider.get_fallback_font returns Occulta font."""
|
||||
provider = BuiltinFontProvider(font_dir)
|
||||
|
||||
fallback = provider.get_fallback_font()
|
||||
assert fallback is not None
|
||||
assert fallback == provider.get_font('Occulta')
|
||||
|
||||
|
||||
def test_builtin_font_provider_missing_font_logs_warning(tmp_path, font_dir, caplog):
|
||||
"""Test that missing expected fonts log a warning."""
|
||||
# Create minimal font directory with only Occulta.ttf
|
||||
(tmp_path / 'Occulta.ttf').write_bytes((font_dir / 'Occulta.ttf').read_bytes())
|
||||
|
||||
with caplog.at_level(logging.WARNING):
|
||||
provider = BuiltinFontProvider(tmp_path)
|
||||
|
||||
# Should have logged warnings for missing fonts
|
||||
assert 'NotoSans-Regular' in caplog.text
|
||||
assert 'not found' in caplog.text
|
||||
|
||||
# But Occulta should be loaded
|
||||
assert provider.get_fallback_font() is not None
|
||||
|
||||
|
||||
def test_builtin_font_provider_missing_occulta_raises(tmp_path):
|
||||
"""Test that missing Occulta.ttf raises FileNotFoundError."""
|
||||
with pytest.raises(FileNotFoundError, match="Required fallback font"):
|
||||
BuiltinFontProvider(tmp_path)
|
||||
@@ -0,0 +1,550 @@
|
||||
#!/usr/bin/env python3
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Direct tests for multilingual text rendering with fpdf2 renderer.
|
||||
|
||||
This tests the fpdf2 renderer with various language groups:
|
||||
- Latin (English, French, German with diacritics)
|
||||
- Arabic (Arabic, Persian - RTL scripts)
|
||||
- CJK (Chinese Simplified/Traditional, Japanese, Korean)
|
||||
- Devanagari (Hindi, Sanskrit)
|
||||
"""
|
||||
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from ocrmypdf.font import MultiFontManager
|
||||
from ocrmypdf.fpdf_renderer import DebugRenderOptions, Fpdf2PdfRenderer
|
||||
from ocrmypdf.hocrtransform.hocr_parser import HocrParser
|
||||
|
||||
RESOURCES = Path(__file__).parent / "resources"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def font_dir():
|
||||
"""Return path to font directory."""
|
||||
return Path(__file__).parent.parent / "src" / "ocrmypdf" / "data"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def multi_font_manager(font_dir):
|
||||
"""Create MultiFontManager instance for testing."""
|
||||
return MultiFontManager(font_dir)
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Latin Script Tests
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class TestLatinScript:
|
||||
"""Tests for Latin script (English, French, German, etc.)."""
|
||||
|
||||
@pytest.fixture
|
||||
def latin_hocr(self):
|
||||
"""Return path to Latin HOCR test file."""
|
||||
return RESOURCES / "latin.hocr"
|
||||
|
||||
def test_render_latin_basic(self, latin_hocr, multi_font_manager, tmp_path):
|
||||
"""Test rendering Latin script with various diacritics."""
|
||||
parser = HocrParser(latin_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
assert page is not None
|
||||
paras = list(page.paragraphs)
|
||||
assert len(paras) == 3 # English, French, German
|
||||
|
||||
# Check languages
|
||||
assert paras[0].language == 'eng'
|
||||
assert paras[1].language == 'fra'
|
||||
assert paras[2].language == 'deu'
|
||||
|
||||
# Render to PDF
|
||||
output_pdf = tmp_path / "latin_output.pdf"
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
assert output_pdf.exists()
|
||||
assert output_pdf.stat().st_size > 0
|
||||
|
||||
# Extract text and verify
|
||||
text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True)
|
||||
|
||||
# English words
|
||||
assert 'quick' in text or 'brown' in text or 'fox' in text
|
||||
|
||||
# French with diacritics
|
||||
assert 'Café' in text or 'résumé' in text or 'naïve' in text
|
||||
|
||||
# German with umlauts and eszett
|
||||
assert 'Größe' in text or 'Zürich' in text or 'Ärger' in text
|
||||
|
||||
def test_latin_font_selection(self, latin_hocr, multi_font_manager):
|
||||
"""Test that NotoSans is selected for Latin text."""
|
||||
parser = HocrParser(latin_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
for line in page.lines:
|
||||
for word in line.children:
|
||||
if word.text:
|
||||
font = multi_font_manager.select_font_for_word(
|
||||
word.text, line.language
|
||||
)
|
||||
assert font is not None
|
||||
# Latin text should use NotoSans-Regular
|
||||
assert multi_font_manager.has_all_glyphs(
|
||||
'NotoSans-Regular', word.text
|
||||
)
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Arabic Script Tests
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class TestArabicScript:
|
||||
"""Tests for Arabic script (Arabic, Persian, etc.)."""
|
||||
|
||||
@pytest.fixture
|
||||
def arabic_hocr(self):
|
||||
"""Return path to Arabic HOCR test file."""
|
||||
return RESOURCES / "arabic.hocr"
|
||||
|
||||
def test_render_arabic_basic(self, arabic_hocr, multi_font_manager, tmp_path):
|
||||
"""Test rendering Arabic script text."""
|
||||
parser = HocrParser(arabic_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
assert page is not None
|
||||
paras = list(page.paragraphs)
|
||||
assert len(paras) == 3 # Arabic paragraphs and Persian
|
||||
|
||||
# Render to PDF
|
||||
output_pdf = tmp_path / "arabic_output.pdf"
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
assert output_pdf.exists()
|
||||
assert output_pdf.stat().st_size > 0
|
||||
|
||||
# Extract text and verify Arabic content
|
||||
text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True)
|
||||
|
||||
# Arabic words: مرحبا بالعالم (Hello world)
|
||||
assert 'مرحبا' in text or 'بالعالم' in text
|
||||
# هذا نص عربي (This is Arabic text)
|
||||
assert 'عربي' in text or 'نص' in text
|
||||
|
||||
def test_arabic_font_selection(self, arabic_hocr, multi_font_manager):
|
||||
"""Test that NotoSansArabic is selected for Arabic text."""
|
||||
parser = HocrParser(arabic_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
for line in page.lines:
|
||||
for word in line.children:
|
||||
if word.text and line.language in ('ara', 'per'):
|
||||
font = multi_font_manager.select_font_for_word(
|
||||
word.text, line.language
|
||||
)
|
||||
assert font is not None
|
||||
# Arabic text should use NotoSansArabic
|
||||
assert multi_font_manager.has_all_glyphs(
|
||||
'NotoSansArabic-Regular', word.text
|
||||
), f"NotoSansArabic cannot render '{word.text}'"
|
||||
|
||||
def test_arabic_rtl_handling(self, arabic_hocr):
|
||||
"""Test that RTL direction is correctly parsed from hOCR."""
|
||||
parser = HocrParser(arabic_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
for para in page.paragraphs:
|
||||
if para.language in ('ara', 'per'):
|
||||
# Arabic paragraphs should have RTL direction
|
||||
assert para.direction == 'rtl', \
|
||||
"Arabic paragraph should have RTL direction"
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# CJK Script Tests
|
||||
# =============================================================================
|
||||
|
||||
|
||||
def _cjk_font_works(multi_font_manager) -> bool:
|
||||
"""Check if CJK font is working (not corrupted)."""
|
||||
return multi_font_manager.has_all_glyphs('NotoSansCJK-Regular', '你')
|
||||
|
||||
|
||||
class TestCJKScript:
|
||||
"""Tests for CJK scripts (Chinese, Japanese, Korean)."""
|
||||
|
||||
@pytest.fixture
|
||||
def cjk_hocr(self):
|
||||
"""Return path to CJK HOCR test file."""
|
||||
return RESOURCES / "cjk.hocr"
|
||||
|
||||
def test_render_cjk_basic(self, cjk_hocr, multi_font_manager, tmp_path):
|
||||
"""Test rendering CJK script text."""
|
||||
if not _cjk_font_works(multi_font_manager):
|
||||
pytest.skip("CJK font not available or corrupted")
|
||||
|
||||
parser = HocrParser(cjk_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
assert page is not None
|
||||
paras = list(page.paragraphs)
|
||||
assert len(paras) == 4 # Chinese Simplified, Traditional, Japanese, Korean
|
||||
|
||||
# Check languages
|
||||
languages = [p.language for p in paras]
|
||||
assert 'chi_sim' in languages
|
||||
assert 'chi_tra' in languages
|
||||
assert 'jpn' in languages
|
||||
assert 'kor' in languages
|
||||
|
||||
# Render to PDF
|
||||
output_pdf = tmp_path / "cjk_output.pdf"
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
assert output_pdf.exists()
|
||||
assert output_pdf.stat().st_size > 0
|
||||
|
||||
# Extract text and verify CJK content
|
||||
text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True)
|
||||
|
||||
# Chinese: 你好 世界 (Hello world)
|
||||
assert '你好' in text or '世界' in text
|
||||
# Japanese: こんにちは (Hello)
|
||||
assert 'こんにちは' in text or '世界' in text
|
||||
# Korean: 안녕하세요 (Hello)
|
||||
assert '안녕하세요' in text or '세계' in text
|
||||
|
||||
def test_cjk_font_selection(self, cjk_hocr, multi_font_manager):
|
||||
"""Test that NotoSansCJK is selected for CJK text."""
|
||||
if not _cjk_font_works(multi_font_manager):
|
||||
pytest.skip("CJK font not available or corrupted")
|
||||
|
||||
parser = HocrParser(cjk_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
cjk_languages = {'chi_sim', 'chi_tra', 'jpn', 'kor', 'zho', 'chi'}
|
||||
|
||||
for line in page.lines:
|
||||
for word in line.children:
|
||||
if word.text and line.language in cjk_languages:
|
||||
font = multi_font_manager.select_font_for_word(
|
||||
word.text, line.language
|
||||
)
|
||||
assert font is not None
|
||||
# CJK text should use NotoSansCJK
|
||||
assert multi_font_manager.has_all_glyphs(
|
||||
'NotoSansCJK-Regular', word.text
|
||||
), f"NotoSansCJK cannot render '{word.text}'"
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Devanagari Script Tests
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class TestDevanagariScript:
|
||||
"""Tests for Devanagari script (Hindi, Sanskrit, etc.)."""
|
||||
|
||||
@pytest.fixture
|
||||
def devanagari_hocr(self):
|
||||
"""Return path to Devanagari HOCR test file."""
|
||||
return RESOURCES / "devanagari.hocr"
|
||||
|
||||
def test_render_devanagari_basic(
|
||||
self, devanagari_hocr, multi_font_manager, tmp_path
|
||||
):
|
||||
"""Test rendering Devanagari script text."""
|
||||
parser = HocrParser(devanagari_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
assert page is not None
|
||||
paras = list(page.paragraphs)
|
||||
assert len(paras) == 3 # Hindi paragraphs and Sanskrit
|
||||
|
||||
# Render to PDF
|
||||
output_pdf = tmp_path / "devanagari_output.pdf"
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
assert output_pdf.exists()
|
||||
assert output_pdf.stat().st_size > 0
|
||||
|
||||
# Extract text and verify Devanagari content
|
||||
text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True)
|
||||
|
||||
# Hindi: नमस्ते दुनिया (Hello world)
|
||||
assert 'नमस्ते' in text or 'दुनिया' in text
|
||||
# यह हिंदी पाठ है (This is Hindi text)
|
||||
assert 'हिंदी' in text or 'पाठ' in text
|
||||
|
||||
def test_devanagari_font_selection(self, devanagari_hocr, multi_font_manager):
|
||||
"""Test that NotoSansDevanagari is selected for Devanagari text."""
|
||||
parser = HocrParser(devanagari_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
devanagari_languages = {'hin', 'san', 'mar', 'nep'}
|
||||
|
||||
for line in page.lines:
|
||||
for word in line.children:
|
||||
if word.text and line.language in devanagari_languages:
|
||||
font = multi_font_manager.select_font_for_word(
|
||||
word.text, line.language
|
||||
)
|
||||
assert font is not None
|
||||
# Devanagari text should use NotoSansDevanagari
|
||||
assert multi_font_manager.has_all_glyphs(
|
||||
'NotoSansDevanagari-Regular', word.text
|
||||
), f"NotoSansDevanagari cannot render '{word.text}'"
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Mixed Language / Multilingual Tests
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class TestMultilingual:
|
||||
"""Tests for mixed-language documents."""
|
||||
|
||||
@pytest.fixture
|
||||
def multilingual_hocr(self):
|
||||
"""Return path to multilingual HOCR test file."""
|
||||
return RESOURCES / "multilingual.hocr"
|
||||
|
||||
def test_render_multilingual_hocr_basic(
|
||||
self, multilingual_hocr, multi_font_manager, tmp_path
|
||||
):
|
||||
"""Test rendering multilingual HOCR file with English and Arabic text."""
|
||||
parser = HocrParser(multilingual_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
assert page is not None
|
||||
assert len(list(page.paragraphs)) == 2 # English and Arabic paragraphs
|
||||
|
||||
# Check languages
|
||||
paras = list(page.paragraphs)
|
||||
assert paras[0].language == 'eng'
|
||||
assert paras[1].language == 'ara'
|
||||
|
||||
# Render to PDF
|
||||
output_pdf = tmp_path / "multilingual_output.pdf"
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
assert output_pdf.exists()
|
||||
assert output_pdf.stat().st_size > 0
|
||||
|
||||
# Extract text from PDF
|
||||
text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True)
|
||||
|
||||
# Verify both English and Arabic text are present
|
||||
assert 'English' in text or 'Text' in text or 'Here' in text
|
||||
# Arabic text: مرحبا بك
|
||||
assert 'مرحبا' in text or 'بك' in text
|
||||
|
||||
def test_render_multilingual_with_debug_options(
|
||||
self, multilingual_hocr, multi_font_manager, tmp_path
|
||||
):
|
||||
"""Test rendering with debug visualization enabled."""
|
||||
parser = HocrParser(multilingual_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
# Render with debug options
|
||||
output_pdf = tmp_path / "multilingual_debug.pdf"
|
||||
debug_options = DebugRenderOptions(
|
||||
render_baseline=True,
|
||||
render_line_bbox=True,
|
||||
render_word_bbox=True,
|
||||
)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
debug_render_options=debug_options,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
assert output_pdf.exists()
|
||||
assert output_pdf.stat().st_size > 0
|
||||
|
||||
def test_multilingual_invisible_text(
|
||||
self, multilingual_hocr, multi_font_manager, tmp_path
|
||||
):
|
||||
"""Test rendering with invisible text (default OCR mode)."""
|
||||
parser = HocrParser(multilingual_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
# Render with invisible text (standard for OCR layer)
|
||||
output_pdf = tmp_path / "multilingual_invisible.pdf"
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=300.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=True,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
assert output_pdf.exists()
|
||||
|
||||
# Text should still be extractable even though invisible
|
||||
text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True)
|
||||
assert len(text.strip()) > 0
|
||||
|
||||
def test_multilingual_font_selection(self, multilingual_hocr, multi_font_manager):
|
||||
"""Test that correct fonts are selected for each language."""
|
||||
parser = HocrParser(multilingual_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
# Get all words
|
||||
words = []
|
||||
for line in page.lines:
|
||||
for word in line.children:
|
||||
if word.text:
|
||||
words.append((word.text, line.language))
|
||||
|
||||
# Verify we have both English and Arabic words
|
||||
eng_words = [w for w, lang in words if lang == 'eng']
|
||||
ara_words = [w for w, lang in words if lang == 'ara']
|
||||
|
||||
assert len(eng_words) > 0, "Should have English words"
|
||||
assert len(ara_words) > 0, "Should have Arabic words"
|
||||
|
||||
# Test font selection
|
||||
for text, lang in words:
|
||||
font_mgr = multi_font_manager.select_font_for_word(text, lang)
|
||||
assert font_mgr is not None, f"No font selected for '{text}' ({lang})"
|
||||
|
||||
if lang == 'ara':
|
||||
assert multi_font_manager.has_all_glyphs(
|
||||
'NotoSansArabic-Regular', text
|
||||
), f"NotoSansArabic cannot render '{text}'"
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Baseline and Structure Tests
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class TestBaselineHandling:
|
||||
"""Tests for baseline and hOCR structure handling."""
|
||||
|
||||
@pytest.fixture
|
||||
def multilingual_hocr(self):
|
||||
"""Return path to multilingual HOCR test file."""
|
||||
return RESOURCES / "multilingual.hocr"
|
||||
|
||||
def test_multilingual_baseline_handling(self, multilingual_hocr):
|
||||
"""Test that baseline information is correctly parsed from hOCR."""
|
||||
parser = HocrParser(multilingual_hocr)
|
||||
page = parser.parse()
|
||||
|
||||
for line in page.lines:
|
||||
if line.baseline:
|
||||
# Baseline should be reasonable
|
||||
assert -1.0 <= line.baseline.slope <= 1.0, \
|
||||
"Baseline slope should be reasonable"
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Font Coverage Tests
|
||||
# =============================================================================
|
||||
|
||||
|
||||
class TestFontCoverage:
|
||||
"""Tests verifying font coverage for various scripts."""
|
||||
|
||||
def test_noto_sans_latin_coverage(self, multi_font_manager):
|
||||
"""Test NotoSans covers common Latin characters and diacritics."""
|
||||
latin_samples = [
|
||||
"Hello World",
|
||||
"Café résumé naïve",
|
||||
"Größe Zürich Ärger",
|
||||
"ÀÁÂÃÄÅÆÇÈÉÊË",
|
||||
"àáâãäåæçèéêë",
|
||||
]
|
||||
|
||||
for sample in latin_samples:
|
||||
assert multi_font_manager.has_all_glyphs('NotoSans-Regular', sample), \
|
||||
f"NotoSans should cover: {sample}"
|
||||
|
||||
def test_noto_sans_arabic_coverage(self, multi_font_manager):
|
||||
"""Test NotoSansArabic covers Arabic characters."""
|
||||
arabic_samples = [
|
||||
"مرحبا", # Hello
|
||||
"بالعالم", # World
|
||||
"العربية", # Arabic
|
||||
]
|
||||
|
||||
for sample in arabic_samples:
|
||||
assert multi_font_manager.has_all_glyphs(
|
||||
'NotoSansArabic-Regular', sample
|
||||
), f"NotoSansArabic should cover: {sample}"
|
||||
|
||||
def test_noto_sans_devanagari_coverage(self, multi_font_manager):
|
||||
"""Test NotoSansDevanagari covers Devanagari characters."""
|
||||
devanagari_samples = [
|
||||
"नमस्ते", # Hello
|
||||
"हिंदी", # Hindi
|
||||
"संस्कृत", # Sanskrit
|
||||
]
|
||||
|
||||
for sample in devanagari_samples:
|
||||
assert multi_font_manager.has_all_glyphs(
|
||||
'NotoSansDevanagari-Regular', sample
|
||||
), f"NotoSansDevanagari should cover: {sample}"
|
||||
|
||||
def test_noto_sans_cjk_coverage(self, multi_font_manager):
|
||||
"""Test NotoSansCJK covers CJK characters."""
|
||||
if not _cjk_font_works(multi_font_manager):
|
||||
pytest.skip("CJK font not available or corrupted")
|
||||
|
||||
cjk_samples = [
|
||||
"你好", # Chinese: Hello
|
||||
"世界", # Chinese: World
|
||||
"こんにちは", # Japanese: Hello
|
||||
"안녕하세요", # Korean: Hello
|
||||
]
|
||||
|
||||
for sample in cjk_samples:
|
||||
assert multi_font_manager.has_all_glyphs('NotoSansCJK-Regular', sample), \
|
||||
f"NotoSansCJK should cover: {sample}"
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Allow running this test directly for quick iteration
|
||||
import sys
|
||||
|
||||
sys.exit(pytest.main([__file__, "-v", "-s"]))
|
||||
+10
-10
@@ -15,12 +15,12 @@ wh_rect = [0, 0, 412, 592]
|
||||
neg_rect = [-100, -100, 512, 692]
|
||||
|
||||
mediabox_testdata = [
|
||||
('hocr', 'pdfa', 'ccitt.pdf', None, inset_rect, wh_rect),
|
||||
('fpdf2', 'pdfa', 'ccitt.pdf', None, inset_rect, wh_rect),
|
||||
('sandwich', 'pdfa', 'ccitt.pdf', None, inset_rect, wh_rect),
|
||||
('hocr', 'pdf', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
('fpdf2', 'pdf', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
('sandwich', 'pdf', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
(
|
||||
'hocr',
|
||||
'fpdf2',
|
||||
'pdfa',
|
||||
'ccitt.pdf',
|
||||
'--force-ocr',
|
||||
@@ -28,15 +28,15 @@ mediabox_testdata = [
|
||||
wh_rect,
|
||||
),
|
||||
(
|
||||
'hocr',
|
||||
'fpdf2',
|
||||
'pdf',
|
||||
'ccitt.pdf',
|
||||
'--force-ocr',
|
||||
inset_rect,
|
||||
wh_rect,
|
||||
),
|
||||
('hocr', 'pdfa', 'ccitt.pdf', '--force-ocr', neg_rect, page_rect),
|
||||
('hocr', 'pdf', 'ccitt.pdf', '--force-ocr', neg_rect, page_rect),
|
||||
('fpdf2', 'pdfa', 'ccitt.pdf', '--force-ocr', neg_rect, page_rect),
|
||||
('fpdf2', 'pdf', 'ccitt.pdf', '--force-ocr', neg_rect, page_rect),
|
||||
]
|
||||
|
||||
|
||||
@@ -69,12 +69,12 @@ def test_media_box(
|
||||
|
||||
|
||||
cropbox_testdata = [
|
||||
('hocr', 'pdfa', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
('fpdf2', 'pdfa', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
('sandwich', 'pdfa', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
('hocr', 'pdf', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
('fpdf2', 'pdf', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
('sandwich', 'pdf', 'ccitt.pdf', None, inset_rect, inset_rect),
|
||||
(
|
||||
'hocr',
|
||||
'fpdf2',
|
||||
'pdfa',
|
||||
'ccitt.pdf',
|
||||
'--force-ocr',
|
||||
@@ -82,7 +82,7 @@ cropbox_testdata = [
|
||||
inset_rect,
|
||||
),
|
||||
(
|
||||
'hocr',
|
||||
'fpdf2',
|
||||
'pdf',
|
||||
'ccitt.pdf',
|
||||
'--force-ocr',
|
||||
|
||||
+136
-108
@@ -1,7 +1,7 @@
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Unit tests for PdfTextRenderer class."""
|
||||
"""Unit tests for Fpdf2PdfRenderer class."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -15,17 +15,16 @@ from pdfminer.pdfdocument import PDFDocument
|
||||
from pdfminer.pdfinterp import PDFPageInterpreter, PDFResourceManager
|
||||
from pdfminer.pdfpage import PDFPage
|
||||
from pdfminer.pdfparser import PDFParser
|
||||
from PIL import Image
|
||||
|
||||
from ocrmypdf.font import MultiFontManager
|
||||
from ocrmypdf.fpdf_renderer import DebugRenderOptions, Fpdf2PdfRenderer
|
||||
from ocrmypdf.helpers import check_pdf
|
||||
from ocrmypdf.hocrtransform import (
|
||||
Baseline,
|
||||
BoundingBox,
|
||||
OcrClass,
|
||||
OcrElement,
|
||||
PdfTextRenderer,
|
||||
)
|
||||
from ocrmypdf.hocrtransform.pdf_renderer import DebugRenderOptions
|
||||
|
||||
|
||||
def text_from_pdf(filename: Path) -> str:
|
||||
@@ -42,6 +41,18 @@ def text_from_pdf(filename: Path) -> str:
|
||||
return output_string.getvalue()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def font_dir():
|
||||
"""Get the font directory."""
|
||||
return Path(__file__).parent.parent / "src" / "ocrmypdf" / "data"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def multi_font_manager(font_dir):
|
||||
"""Create a MultiFontManager for tests."""
|
||||
return MultiFontManager(font_dir)
|
||||
|
||||
|
||||
def create_simple_page(
|
||||
width: float = 1000,
|
||||
height: float = 500,
|
||||
@@ -93,90 +104,108 @@ def create_simple_page(
|
||||
return page
|
||||
|
||||
|
||||
class TestPdfTextRendererBasic:
|
||||
"""Basic PdfTextRenderer functionality tests."""
|
||||
class TestFpdf2PdfRendererBasic:
|
||||
"""Basic Fpdf2PdfRenderer functionality tests."""
|
||||
|
||||
def test_render_simple_page(self, tmp_path):
|
||||
def test_render_simple_page(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering a simple page with two words."""
|
||||
page = create_simple_page()
|
||||
output_pdf = tmp_path / "simple.pdf"
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
assert output_pdf.exists()
|
||||
check_pdf(str(output_pdf))
|
||||
|
||||
def test_rendered_text_extractable(self, tmp_path):
|
||||
def test_rendered_text_extractable(self, tmp_path, multi_font_manager):
|
||||
"""Test that rendered text can be extracted from the PDF."""
|
||||
page = create_simple_page()
|
||||
output_pdf = tmp_path / "extractable.pdf"
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
assert "Hello" in extracted_text
|
||||
assert "World" in extracted_text
|
||||
|
||||
def test_invisible_text_mode(self, tmp_path):
|
||||
def test_invisible_text_mode(self, tmp_path, multi_font_manager):
|
||||
"""Test that invisible_text=True creates a valid PDF."""
|
||||
page = create_simple_page()
|
||||
output_pdf = tmp_path / "invisible.pdf"
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf, invisible_text=True)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=72.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=True,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
# Text should still be extractable even when invisible
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
assert "Hello" in extracted_text
|
||||
|
||||
def test_visible_text_mode(self, tmp_path):
|
||||
def test_visible_text_mode(self, tmp_path, multi_font_manager):
|
||||
"""Test that invisible_text=False creates a valid PDF with visible text."""
|
||||
page = create_simple_page()
|
||||
output_pdf = tmp_path / "visible.pdf"
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf, invisible_text=False)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=72.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
# Text should be extractable
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
assert "Hello" in extracted_text
|
||||
|
||||
|
||||
class TestPdfTextRendererPageSize:
|
||||
class TestFpdf2PdfRendererPageSize:
|
||||
"""Test page size calculations."""
|
||||
|
||||
def test_page_dimensions(self, tmp_path):
|
||||
def test_page_dimensions(self, tmp_path, multi_font_manager):
|
||||
"""Test that page dimensions are calculated correctly."""
|
||||
# 1000x500 pixels at 72 dpi = 1000x500 points
|
||||
page = create_simple_page(width=1000, height=500)
|
||||
output_pdf = tmp_path / "dimensions.pdf"
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
assert renderer.width == pytest.approx(1000.0)
|
||||
assert renderer.height == pytest.approx(500.0)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
assert renderer.coord_transform.page_width_pt == pytest.approx(1000.0)
|
||||
assert renderer.coord_transform.page_height_pt == pytest.approx(500.0)
|
||||
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
def test_high_dpi_page(self, tmp_path):
|
||||
def test_high_dpi_page(self, tmp_path, multi_font_manager):
|
||||
"""Test page dimensions at higher DPI."""
|
||||
# 720x360 pixels at 144 dpi = 360x180 points
|
||||
page = create_simple_page(width=720, height=360)
|
||||
output_pdf = tmp_path / "high_dpi.pdf"
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=144.0)
|
||||
assert renderer.width == pytest.approx(360.0)
|
||||
assert renderer.height == pytest.approx(180.0)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=144.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
assert renderer.coord_transform.page_width_pt == pytest.approx(360.0)
|
||||
assert renderer.coord_transform.page_height_pt == pytest.approx(180.0)
|
||||
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer.render(output_pdf)
|
||||
check_pdf(str(output_pdf))
|
||||
|
||||
|
||||
class TestPdfTextRendererMultiLine:
|
||||
class TestFpdf2PdfRendererMultiLine:
|
||||
"""Test rendering of multi-line content."""
|
||||
|
||||
def test_multiple_lines(self, tmp_path):
|
||||
def test_multiple_lines(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering multiple lines of text."""
|
||||
line1_words = [
|
||||
OcrElement(
|
||||
@@ -231,8 +260,10 @@ class TestPdfTextRendererMultiLine:
|
||||
)
|
||||
|
||||
output_pdf = tmp_path / "multiline.pdf"
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
assert "Line" in extracted_text
|
||||
@@ -240,20 +271,22 @@ class TestPdfTextRendererMultiLine:
|
||||
assert "two" in extracted_text
|
||||
|
||||
|
||||
class TestPdfTextRendererTextDirection:
|
||||
class TestFpdf2PdfRendererTextDirection:
|
||||
"""Test rendering of different text directions."""
|
||||
|
||||
def test_ltr_text(self, tmp_path):
|
||||
def test_ltr_text(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering LTR text."""
|
||||
page = create_simple_page()
|
||||
output_pdf = tmp_path / "ltr.pdf"
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
|
||||
def test_rtl_text(self, tmp_path):
|
||||
def test_rtl_text(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering RTL text."""
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
@@ -281,16 +314,18 @@ class TestPdfTextRendererTextDirection:
|
||||
)
|
||||
|
||||
output_pdf = tmp_path / "rtl.pdf"
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
|
||||
|
||||
class TestPdfTextRendererBaseline:
|
||||
class TestFpdf2PdfRendererBaseline:
|
||||
"""Test baseline handling in rendering."""
|
||||
|
||||
def test_sloped_baseline(self, tmp_path):
|
||||
def test_sloped_baseline(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering with a sloped baseline."""
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
@@ -317,18 +352,20 @@ class TestPdfTextRendererBaseline:
|
||||
)
|
||||
|
||||
output_pdf = tmp_path / "sloped.pdf"
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
assert "Sloped" in extracted_text
|
||||
|
||||
|
||||
class TestPdfTextRendererTextangle:
|
||||
class TestFpdf2PdfRendererTextangle:
|
||||
"""Test textangle (rotation) handling in rendering."""
|
||||
|
||||
def test_rotated_text(self, tmp_path):
|
||||
def test_rotated_text(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering rotated text."""
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
@@ -356,32 +393,36 @@ class TestPdfTextRendererTextangle:
|
||||
)
|
||||
|
||||
output_pdf = tmp_path / "rotated.pdf"
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
assert "Rotated" in extracted_text
|
||||
|
||||
|
||||
class TestPdfTextRendererWordBreaks:
|
||||
"""Test word break injection."""
|
||||
class TestFpdf2PdfRendererWordBreaks:
|
||||
"""Test word rendering."""
|
||||
|
||||
def test_word_breaks_english(self, tmp_path):
|
||||
"""Test that word breaks are injected for English text."""
|
||||
def test_word_breaks_english(self, tmp_path, multi_font_manager):
|
||||
"""Test that words are rendered for English text."""
|
||||
page = create_simple_page()
|
||||
output_pdf = tmp_path / "english.pdf"
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
# Words should be separated
|
||||
# Words should be present
|
||||
assert "Hello" in extracted_text
|
||||
assert "World" in extracted_text
|
||||
|
||||
def test_no_word_breaks_cjk(self, tmp_path):
|
||||
"""Test that word breaks are not injected for CJK text."""
|
||||
def test_cjk_text(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering CJK text."""
|
||||
words = [
|
||||
OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
@@ -414,40 +455,47 @@ class TestPdfTextRendererWordBreaks:
|
||||
)
|
||||
|
||||
output_pdf = tmp_path / "chinese.pdf"
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
|
||||
|
||||
class TestPdfTextRendererDebugOptions:
|
||||
class TestFpdf2PdfRendererDebugOptions:
|
||||
"""Test debug rendering options."""
|
||||
|
||||
def test_debug_render_options_default(self):
|
||||
def test_debug_render_options_default(self, multi_font_manager):
|
||||
"""Test that debug options are disabled by default."""
|
||||
page = create_simple_page()
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
|
||||
assert renderer.render_options.render_paragraph_bbox is False
|
||||
assert renderer.render_options.render_baseline is False
|
||||
assert renderer.render_options.render_word_bbox is False
|
||||
assert renderer.debug_options.render_baseline is False
|
||||
assert renderer.debug_options.render_word_bbox is False
|
||||
assert renderer.debug_options.render_line_bbox is False
|
||||
|
||||
def test_debug_render_options_enabled(self, tmp_path):
|
||||
def test_debug_render_options_enabled(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering with debug options enabled."""
|
||||
page = create_simple_page()
|
||||
output_pdf = tmp_path / "debug.pdf"
|
||||
|
||||
debug_opts = DebugRenderOptions(
|
||||
render_paragraph_bbox=True,
|
||||
render_baseline=True,
|
||||
render_word_bbox=True,
|
||||
render_triangle=True,
|
||||
render_line_bbox=True,
|
||||
)
|
||||
|
||||
renderer = PdfTextRenderer(
|
||||
page=page, dpi=72.0, debug_render_options=debug_opts
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page,
|
||||
dpi=72.0,
|
||||
multi_font_manager=multi_font_manager,
|
||||
invisible_text=False,
|
||||
debug_render_options=debug_opts,
|
||||
)
|
||||
renderer.render(out_filename=output_pdf, invisible_text=False)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
# Text should still be extractable
|
||||
@@ -455,54 +503,30 @@ class TestPdfTextRendererDebugOptions:
|
||||
assert "Hello" in extracted_text
|
||||
|
||||
|
||||
class TestPdfTextRendererWithImage:
|
||||
"""Test rendering with image overlay."""
|
||||
class TestFpdf2PdfRendererErrors:
|
||||
"""Test error handling in Fpdf2PdfRenderer."""
|
||||
|
||||
def test_render_with_image(self, tmp_path):
|
||||
"""Test rendering with an image overlaid on text."""
|
||||
page = create_simple_page()
|
||||
output_pdf = tmp_path / "with_image.pdf"
|
||||
|
||||
# Create a simple test image
|
||||
image_path = tmp_path / "test.png"
|
||||
img = Image.new('RGB', (1000, 500), color='white')
|
||||
img.save(image_path)
|
||||
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(
|
||||
out_filename=output_pdf, image_filename=image_path, invisible_text=True
|
||||
)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
# Text should still be extractable under the image
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
assert "Hello" in extracted_text
|
||||
|
||||
|
||||
class TestPdfTextRendererErrors:
|
||||
"""Test error handling in PdfTextRenderer."""
|
||||
|
||||
def test_invalid_ocr_class(self):
|
||||
def test_invalid_ocr_class(self, multi_font_manager):
|
||||
"""Test that non-page elements are rejected."""
|
||||
line = OcrElement(
|
||||
ocr_class=OcrClass.LINE, bbox=BoundingBox(left=0, top=0, right=100, bottom=50)
|
||||
)
|
||||
|
||||
with pytest.raises(ValueError, match="ocr_page"):
|
||||
PdfTextRenderer(page=line, dpi=72.0)
|
||||
Fpdf2PdfRenderer(page=line, dpi=72.0, multi_font_manager=multi_font_manager)
|
||||
|
||||
def test_page_without_bbox(self):
|
||||
def test_page_without_bbox(self, multi_font_manager):
|
||||
"""Test that pages without bbox are rejected."""
|
||||
page = OcrElement(ocr_class=OcrClass.PAGE)
|
||||
|
||||
with pytest.raises(ValueError, match="bounding box"):
|
||||
PdfTextRenderer(page=page, dpi=72.0)
|
||||
Fpdf2PdfRenderer(page=page, dpi=72.0, multi_font_manager=multi_font_manager)
|
||||
|
||||
|
||||
class TestPdfTextRendererLineTypes:
|
||||
class TestFpdf2PdfRendererLineTypes:
|
||||
"""Test rendering of different line types."""
|
||||
|
||||
def test_header_line(self, tmp_path):
|
||||
def test_header_line(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering header lines."""
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
@@ -529,14 +553,16 @@ class TestPdfTextRendererLineTypes:
|
||||
)
|
||||
|
||||
output_pdf = tmp_path / "header.pdf"
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
assert "Header" in extracted_text
|
||||
|
||||
def test_caption_line(self, tmp_path):
|
||||
def test_caption_line(self, tmp_path, multi_font_manager):
|
||||
"""Test rendering caption lines."""
|
||||
word = OcrElement(
|
||||
ocr_class=OcrClass.WORD,
|
||||
@@ -563,8 +589,10 @@ class TestPdfTextRendererLineTypes:
|
||||
)
|
||||
|
||||
output_pdf = tmp_path / "caption.pdf"
|
||||
renderer = PdfTextRenderer(page=page, dpi=72.0)
|
||||
renderer.render(out_filename=output_pdf)
|
||||
renderer = Fpdf2PdfRenderer(
|
||||
page=page, dpi=72.0, multi_font_manager=multi_font_manager
|
||||
)
|
||||
renderer.render(output_pdf)
|
||||
|
||||
check_pdf(str(output_pdf))
|
||||
extracted_text = text_from_pdf(output_pdf)
|
||||
|
||||
@@ -15,7 +15,7 @@ from ocrmypdf.pdfinfo import PdfInfo
|
||||
|
||||
from .conftest import check_ocrmypdf, have_unpaper, run_ocrmypdf
|
||||
|
||||
RENDERERS = ['hocr', 'sandwich']
|
||||
RENDERERS = ['fpdf2', 'sandwich']
|
||||
|
||||
|
||||
def test_deskew(resources, outdir):
|
||||
@@ -79,7 +79,7 @@ def test_remove_background(resources, outdir):
|
||||
@pytest.mark.parametrize(
|
||||
"pdf", ['palette.pdf', 'cmyk.pdf', 'ccitt.pdf', 'jbig2.pdf', 'lichtenstein.pdf']
|
||||
)
|
||||
@pytest.mark.parametrize("renderer", ['sandwich', 'hocr'])
|
||||
@pytest.mark.parametrize("renderer", ['sandwich', 'fpdf2'])
|
||||
@pytest.mark.parametrize("output_type", ['pdf', 'pdfa'])
|
||||
def test_exotic_image(pdf, renderer, output_type, resources, outdir):
|
||||
outfile = outdir / f'test_{pdf}_{renderer}.pdf'
|
||||
|
||||
@@ -24,7 +24,7 @@ from .conftest import check_ocrmypdf, run_ocrmypdf_api
|
||||
|
||||
# pylintx: disable=unused-variable
|
||||
|
||||
RENDERERS = ['hocr', 'sandwich']
|
||||
RENDERERS = ['fpdf2', 'sandwich']
|
||||
|
||||
|
||||
def compare_images_monochrome(
|
||||
@@ -167,7 +167,7 @@ def test_rotated_skew_timeout(resources, outpdf):
|
||||
input_file,
|
||||
outpdf,
|
||||
'--pdf-renderer',
|
||||
'hocr',
|
||||
'fpdf2',
|
||||
'--deskew',
|
||||
'--tesseract-timeout',
|
||||
'0',
|
||||
@@ -198,7 +198,7 @@ def test_rotate_deskew_ocr_timeout(resources, outdir):
|
||||
'--tesseract-timeout',
|
||||
'0',
|
||||
'--pdf-renderer',
|
||||
'hocr',
|
||||
'fpdf2',
|
||||
'--rasterizer',
|
||||
'ghostscript', # Use Ghostscript for consistent dimensions
|
||||
)
|
||||
@@ -291,7 +291,7 @@ def test_page_rotate_tag(page_rotate_angle, resources, outdir, caplog):
|
||||
|
||||
|
||||
@pytest.mark.parametrize('page_rotate_angle', (0, 90, 180, 270))
|
||||
@pytest.mark.parametrize('renderer', ['sandwich', 'hocr'])
|
||||
@pytest.mark.parametrize('renderer', ['sandwich', 'fpdf2'])
|
||||
@pytest.mark.parametrize('output_type', ['pdf', 'pdfa'])
|
||||
def test_rotate_and_crop(
|
||||
resources, outdir, page_rotate_angle, renderer, output_type, caplog
|
||||
|
||||
@@ -0,0 +1,337 @@
|
||||
# SPDX-FileCopyrightText: 2025 James R. Barlow
|
||||
# SPDX-License-Identifier: MPL-2.0
|
||||
|
||||
"""Unit tests for SystemFontProvider and ChainedFontProvider."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from ocrmypdf.font import (
|
||||
BuiltinFontProvider,
|
||||
ChainedFontProvider,
|
||||
FontManager,
|
||||
SystemFontProvider,
|
||||
)
|
||||
|
||||
|
||||
# --- SystemFontProvider Platform Detection Tests ---
|
||||
|
||||
|
||||
class TestSystemFontProviderPlatform:
|
||||
"""Test platform detection in SystemFontProvider."""
|
||||
|
||||
def test_get_platform_linux(self):
|
||||
"""Test Linux platform detection."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'linux'):
|
||||
assert provider._get_platform() == 'linux'
|
||||
|
||||
def test_get_platform_darwin(self):
|
||||
"""Test macOS platform detection."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'darwin'):
|
||||
assert provider._get_platform() == 'darwin'
|
||||
|
||||
def test_get_platform_windows(self):
|
||||
"""Test Windows platform detection."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'win32'):
|
||||
assert provider._get_platform() == 'windows'
|
||||
|
||||
def test_get_platform_freebsd(self):
|
||||
"""Test FreeBSD platform detection."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'freebsd13'):
|
||||
assert provider._get_platform() == 'freebsd'
|
||||
|
||||
|
||||
class TestSystemFontProviderDirectories:
|
||||
"""Test font directory resolution."""
|
||||
|
||||
def test_linux_font_dirs(self):
|
||||
"""Test Linux font directories."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'linux'):
|
||||
provider._font_dirs = None # Reset cache
|
||||
dirs = provider._get_font_dirs()
|
||||
assert Path('/usr/share/fonts') in dirs
|
||||
assert Path('/usr/local/share/fonts') in dirs
|
||||
|
||||
def test_darwin_font_dirs(self):
|
||||
"""Test macOS font directories."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'darwin'):
|
||||
provider._font_dirs = None # Reset cache
|
||||
dirs = provider._get_font_dirs()
|
||||
assert Path('/Library/Fonts') in dirs
|
||||
assert Path('/System/Library/Fonts') in dirs
|
||||
|
||||
def test_windows_font_dirs_with_windir(self):
|
||||
"""Test Windows font directory from WINDIR env var."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'win32'):
|
||||
with patch.dict('os.environ', {'WINDIR': r'D:\Windows'}):
|
||||
provider._font_dirs = None # Reset cache
|
||||
dirs = provider._get_font_dirs()
|
||||
# Check that Fonts subdir of WINDIR is included
|
||||
# Use str comparison to avoid Path normalization issues across platforms
|
||||
dir_strs = [str(d) for d in dirs]
|
||||
assert any('Fonts' in d for d in dir_strs)
|
||||
|
||||
def test_windows_font_dirs_default(self):
|
||||
"""Test Windows font directory with default path."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'win32'):
|
||||
with patch.dict('os.environ', {}, clear=True):
|
||||
provider._font_dirs = None # Reset cache
|
||||
dirs = provider._get_font_dirs()
|
||||
# Check that Windows\Fonts is included (default fallback)
|
||||
dir_strs = [str(d) for d in dirs]
|
||||
assert any('Windows' in d and 'Fonts' in d for d in dir_strs)
|
||||
|
||||
def test_windows_font_dirs_with_localappdata(self):
|
||||
"""Test Windows user fonts directory from LOCALAPPDATA env var."""
|
||||
provider = SystemFontProvider()
|
||||
with patch.object(sys, 'platform', 'win32'):
|
||||
with patch.dict(
|
||||
'os.environ',
|
||||
{'WINDIR': r'C:\Windows', 'LOCALAPPDATA': r'C:\Users\Test\AppData\Local'},
|
||||
):
|
||||
provider._font_dirs = None # Reset cache
|
||||
dirs = provider._get_font_dirs()
|
||||
dir_strs = [str(d) for d in dirs]
|
||||
# Should have both system and user font directories
|
||||
assert len(dirs) == 2
|
||||
assert any('Windows' in d and 'Fonts' in d for d in dir_strs)
|
||||
assert any('AppData' in d and 'Local' in d and 'Fonts' in d for d in dir_strs)
|
||||
|
||||
def test_font_dirs_cached(self):
|
||||
"""Test that font directories are cached."""
|
||||
provider = SystemFontProvider()
|
||||
dirs1 = provider._get_font_dirs()
|
||||
dirs2 = provider._get_font_dirs()
|
||||
assert dirs1 is dirs2 # Same object, not recomputed
|
||||
|
||||
|
||||
class TestSystemFontProviderLazyLoading:
|
||||
"""Test lazy loading behavior."""
|
||||
|
||||
def test_no_scanning_on_init(self):
|
||||
"""Test that no directory scanning happens during initialization."""
|
||||
provider = SystemFontProvider()
|
||||
# Caches should be empty
|
||||
assert len(provider._font_cache) == 0
|
||||
assert len(provider._not_found) == 0
|
||||
|
||||
def test_get_font_unknown_name_returns_none(self):
|
||||
"""Test that unknown font names return None."""
|
||||
provider = SystemFontProvider()
|
||||
result = provider.get_font('UnknownFont-Regular')
|
||||
assert result is None
|
||||
# Unknown fonts are added to not_found to cache the negative result
|
||||
assert 'UnknownFont-Regular' in provider._not_found
|
||||
|
||||
def test_negative_cache(self):
|
||||
"""Test that not-found results are cached."""
|
||||
provider = SystemFontProvider()
|
||||
# Mock _find_font_file to return None
|
||||
with patch.object(provider, '_find_font_file', return_value=None):
|
||||
result1 = provider.get_font('NotoSansCJK-Regular')
|
||||
assert result1 is None
|
||||
assert 'NotoSansCJK-Regular' in provider._not_found
|
||||
|
||||
# Second call should not call _find_font_file again
|
||||
provider._find_font_file = MagicMock(return_value=None)
|
||||
result2 = provider.get_font('NotoSansCJK-Regular')
|
||||
assert result2 is None
|
||||
provider._find_font_file.assert_not_called()
|
||||
|
||||
def test_positive_cache(self):
|
||||
"""Test that found fonts are cached."""
|
||||
provider = SystemFontProvider()
|
||||
font_dir = Path(__file__).parent.parent / "src" / "ocrmypdf" / "data"
|
||||
font_path = font_dir / "NotoSans-Regular.ttf"
|
||||
|
||||
if not font_path.exists():
|
||||
pytest.skip("Test font not available")
|
||||
|
||||
with patch.object(provider, '_find_font_file', return_value=font_path):
|
||||
result1 = provider.get_font('NotoSans-Regular')
|
||||
assert result1 is not None
|
||||
assert 'NotoSans-Regular' in provider._font_cache
|
||||
|
||||
# Second call should use cache
|
||||
provider._find_font_file = MagicMock()
|
||||
result2 = provider.get_font('NotoSans-Regular')
|
||||
assert result2 is result1
|
||||
provider._find_font_file.assert_not_called()
|
||||
|
||||
|
||||
class TestSystemFontProviderAvailableFonts:
|
||||
"""Test get_available_fonts method."""
|
||||
|
||||
def test_returns_all_patterns(self):
|
||||
"""Test that get_available_fonts returns all known font patterns."""
|
||||
provider = SystemFontProvider()
|
||||
fonts = provider.get_available_fonts()
|
||||
assert 'NotoSans-Regular' in fonts
|
||||
assert 'NotoSansCJK-Regular' in fonts
|
||||
assert 'NotoSansArabic-Regular' in fonts
|
||||
assert 'NotoSansThai-Regular' in fonts
|
||||
|
||||
def test_fallback_font_raises(self):
|
||||
"""Test that get_fallback_font raises NotImplementedError."""
|
||||
provider = SystemFontProvider()
|
||||
with pytest.raises(NotImplementedError):
|
||||
provider.get_fallback_font()
|
||||
|
||||
|
||||
# --- ChainedFontProvider Tests ---
|
||||
|
||||
|
||||
class TestChainedFontProvider:
|
||||
"""Test ChainedFontProvider."""
|
||||
|
||||
def test_requires_at_least_one_provider(self):
|
||||
"""Test that empty provider list raises error."""
|
||||
with pytest.raises(ValueError, match="At least one provider"):
|
||||
ChainedFontProvider([])
|
||||
|
||||
def test_get_font_tries_providers_in_order(self):
|
||||
"""Test that get_font tries providers in order."""
|
||||
provider1 = MagicMock()
|
||||
provider1.get_font.return_value = None
|
||||
|
||||
provider2 = MagicMock()
|
||||
mock_font = MagicMock()
|
||||
provider2.get_font.return_value = mock_font
|
||||
|
||||
chain = ChainedFontProvider([provider1, provider2])
|
||||
result = chain.get_font('TestFont')
|
||||
|
||||
provider1.get_font.assert_called_once_with('TestFont')
|
||||
provider2.get_font.assert_called_once_with('TestFont')
|
||||
assert result is mock_font
|
||||
|
||||
def test_get_font_stops_on_first_match(self):
|
||||
"""Test that get_font stops after first successful match."""
|
||||
mock_font = MagicMock()
|
||||
provider1 = MagicMock()
|
||||
provider1.get_font.return_value = mock_font
|
||||
|
||||
provider2 = MagicMock()
|
||||
|
||||
chain = ChainedFontProvider([provider1, provider2])
|
||||
result = chain.get_font('TestFont')
|
||||
|
||||
provider1.get_font.assert_called_once()
|
||||
provider2.get_font.assert_not_called()
|
||||
assert result is mock_font
|
||||
|
||||
def test_get_font_returns_none_if_all_fail(self):
|
||||
"""Test that get_font returns None if all providers fail."""
|
||||
provider1 = MagicMock()
|
||||
provider1.get_font.return_value = None
|
||||
|
||||
provider2 = MagicMock()
|
||||
provider2.get_font.return_value = None
|
||||
|
||||
chain = ChainedFontProvider([provider1, provider2])
|
||||
result = chain.get_font('TestFont')
|
||||
|
||||
assert result is None
|
||||
|
||||
def test_get_available_fonts_combines_providers(self):
|
||||
"""Test that get_available_fonts combines all providers."""
|
||||
provider1 = MagicMock()
|
||||
provider1.get_available_fonts.return_value = ['Font1', 'Font2']
|
||||
|
||||
provider2 = MagicMock()
|
||||
provider2.get_available_fonts.return_value = ['Font2', 'Font3']
|
||||
|
||||
chain = ChainedFontProvider([provider1, provider2])
|
||||
fonts = chain.get_available_fonts()
|
||||
|
||||
assert fonts == ['Font1', 'Font2', 'Font3'] # Deduplicated, order preserved
|
||||
|
||||
def test_get_fallback_font_from_first_provider(self):
|
||||
"""Test that get_fallback_font uses first available fallback."""
|
||||
mock_font = MagicMock()
|
||||
provider1 = MagicMock()
|
||||
provider1.get_fallback_font.return_value = mock_font
|
||||
|
||||
provider2 = MagicMock()
|
||||
|
||||
chain = ChainedFontProvider([provider1, provider2])
|
||||
result = chain.get_fallback_font()
|
||||
|
||||
assert result is mock_font
|
||||
provider2.get_fallback_font.assert_not_called()
|
||||
|
||||
def test_get_fallback_font_skips_not_implemented(self):
|
||||
"""Test that get_fallback_font skips providers that raise."""
|
||||
provider1 = MagicMock()
|
||||
provider1.get_fallback_font.side_effect = NotImplementedError()
|
||||
|
||||
mock_font = MagicMock()
|
||||
provider2 = MagicMock()
|
||||
provider2.get_fallback_font.return_value = mock_font
|
||||
|
||||
chain = ChainedFontProvider([provider1, provider2])
|
||||
result = chain.get_fallback_font()
|
||||
|
||||
assert result is mock_font
|
||||
|
||||
def test_get_fallback_font_raises_if_none_available(self):
|
||||
"""Test that get_fallback_font raises if no provider has fallback."""
|
||||
provider1 = MagicMock()
|
||||
provider1.get_fallback_font.side_effect = NotImplementedError()
|
||||
|
||||
provider2 = MagicMock()
|
||||
provider2.get_fallback_font.side_effect = KeyError()
|
||||
|
||||
chain = ChainedFontProvider([provider1, provider2])
|
||||
with pytest.raises(RuntimeError, match="No fallback font available"):
|
||||
chain.get_fallback_font()
|
||||
|
||||
|
||||
class TestChainedFontProviderIntegration:
|
||||
"""Integration tests with real providers."""
|
||||
|
||||
@pytest.fixture
|
||||
def font_dir(self):
|
||||
"""Return path to font directory."""
|
||||
return Path(__file__).parent.parent / "src" / "ocrmypdf" / "data"
|
||||
|
||||
def test_builtin_then_system_chain(self, font_dir):
|
||||
"""Test chaining BuiltinFontProvider with SystemFontProvider."""
|
||||
builtin = BuiltinFontProvider(font_dir)
|
||||
system = SystemFontProvider()
|
||||
|
||||
chain = ChainedFontProvider([builtin, system])
|
||||
|
||||
# Should find NotoSans from builtin
|
||||
font = chain.get_font('NotoSans-Regular')
|
||||
assert font is not None
|
||||
|
||||
# Should get fallback from builtin
|
||||
fallback = chain.get_fallback_font()
|
||||
assert fallback is not None
|
||||
|
||||
def test_system_fonts_extend_builtin(self, font_dir):
|
||||
"""Test that system fonts add to builtin fonts."""
|
||||
builtin = BuiltinFontProvider(font_dir)
|
||||
system = SystemFontProvider()
|
||||
|
||||
chain = ChainedFontProvider([builtin, system])
|
||||
|
||||
builtin_fonts = set(builtin.get_available_fonts())
|
||||
chain_fonts = set(chain.get_available_fonts())
|
||||
|
||||
# Chain should have at least as many fonts as builtin
|
||||
assert chain_fonts >= builtin_fonts
|
||||
@@ -49,7 +49,9 @@ def test_skip_pages_does_not_replicate(resources, basename, outdir):
|
||||
def test_content_preservation(resources, outpdf):
|
||||
infile = resources / 'masks.pdf'
|
||||
|
||||
check_ocrmypdf(infile, outpdf, '--pdf-renderer', 'hocr', '--tesseract-timeout', '0')
|
||||
check_ocrmypdf(
|
||||
infile, outpdf, '--pdf-renderer', 'fpdf2', '--tesseract-timeout', '0'
|
||||
)
|
||||
|
||||
info = pdfinfo.PdfInfo(outpdf)
|
||||
page = info[0]
|
||||
|
||||
Reference in New Issue
Block a user