diff --git a/tests/test_hocr_parser.py b/tests/test_hocr_parser.py index ddc22eab..c7f02374 100644 --- a/tests/test_hocr_parser.py +++ b/tests/test_hocr_parser.py @@ -41,7 +41,7 @@ def simple_hocr(tmp_path) -> Path: """) hocr_file = tmp_path / "simple.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') return hocr_file @@ -74,7 +74,7 @@ def multiline_hocr(tmp_path) -> Path: """) hocr_file = tmp_path / "multiline.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') return hocr_file @@ -96,7 +96,7 @@ def rtl_hocr(tmp_path) -> Path: """) hocr_file = tmp_path / "rtl.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') return hocr_file @@ -118,7 +118,7 @@ def rotated_hocr(tmp_path) -> Path: """) hocr_file = tmp_path / "rotated.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') return hocr_file @@ -149,7 +149,7 @@ def header_hocr(tmp_path) -> Path: """) hocr_file = tmp_path / "header.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') return hocr_file @@ -171,7 +171,7 @@ def font_info_hocr(tmp_path) -> Path: """) hocr_file = tmp_path / "font_info.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') return hocr_file @@ -346,14 +346,16 @@ class TestHocrParserErrors: def test_invalid_xml(self, tmp_path): hocr_file = tmp_path / "invalid.hocr" - hocr_file.write_text("not closed") + hocr_file.write_text("not closed", encoding='utf-8') with pytest.raises(HocrParseError): HocrParser(hocr_file) def test_missing_ocr_page(self, tmp_path): hocr_file = tmp_path / "no_page.hocr" - hocr_file.write_text("

No ocr_page

") + hocr_file.write_text( + "

No ocr_page

", encoding='utf-8' + ) parser = HocrParser(hocr_file) with pytest.raises(HocrParseError, match="No ocr_page"): @@ -362,7 +364,8 @@ class TestHocrParserErrors: def test_missing_page_bbox(self, tmp_path): hocr_file = tmp_path / "no_bbox.hocr" hocr_file.write_text( - "
No bbox
" + "
No bbox
", + encoding='utf-8', ) parser = HocrParser(hocr_file) @@ -391,7 +394,7 @@ class TestHocrParserEdgeCases: """) hocr_file = tmp_path / "empty_word.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') parser = HocrParser(hocr_file) page = parser.parse() @@ -418,7 +421,7 @@ class TestHocrParserEdgeCases: """) hocr_file = tmp_path / "whitespace_word.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') parser = HocrParser(hocr_file) page = parser.parse() @@ -446,7 +449,7 @@ class TestHocrParserEdgeCases: """) hocr_file = tmp_path / "no_line_bbox.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') parser = HocrParser(hocr_file) page = parser.parse() @@ -473,7 +476,7 @@ class TestHocrParserEdgeCases: """) hocr_file = tmp_path / "unicode.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') parser = HocrParser(hocr_file) page = parser.parse() @@ -495,7 +498,7 @@ class TestHocrParserEdgeCases: """) hocr_file = tmp_path / "direct_words.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') parser = HocrParser(hocr_file) page = parser.parse() @@ -521,7 +524,7 @@ class TestHocrParserEdgeCases: """) hocr_file = tmp_path / "no_namespace.hocr" - hocr_file.write_text(content) + hocr_file.write_text(content, encoding='utf-8') parser = HocrParser(hocr_file) page = parser.parse() diff --git a/tests/test_multilingual_direct.py b/tests/test_multilingual_direct.py index 382922d6..fd130141 100644 --- a/tests/test_multilingual_direct.py +++ b/tests/test_multilingual_direct.py @@ -76,7 +76,11 @@ class TestLatinScript: assert output_pdf.stat().st_size > 0 # Extract text and verify - text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True) + text = subprocess.check_output( + ['pdftotext', '-enc', 'UTF-8', str(output_pdf), '-'], + text=True, + encoding='utf-8', + ) # English words assert 'quick' in text or 'brown' in text or 'fox' in text @@ -141,7 +145,11 @@ class TestArabicScript: assert output_pdf.stat().st_size > 0 # Extract text and verify Arabic content - text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True) + text = subprocess.check_output( + ['pdftotext', '-enc', 'UTF-8', str(output_pdf), '-'], + text=True, + encoding='utf-8', + ) # Arabic words: مرحبا بالعالم (Hello world) assert 'مرحبا' in text or 'بالعالم' in text @@ -173,8 +181,9 @@ class TestArabicScript: for para in page.paragraphs: if para.language in ('ara', 'per'): # Arabic paragraphs should have RTL direction - assert para.direction == 'rtl', \ - "Arabic paragraph should have RTL direction" + assert ( + para.direction == 'rtl' + ), "Arabic paragraph should have RTL direction" # ============================================================================= @@ -228,7 +237,11 @@ class TestCJKScript: assert output_pdf.stat().st_size > 0 # Extract text and verify CJK content - text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True) + text = subprocess.check_output( + ['pdftotext', '-enc', 'UTF-8', str(output_pdf), '-'], + text=True, + encoding='utf-8', + ) # Chinese: 你好 世界 (Hello world) assert '你好' in text or '世界' in text @@ -298,7 +311,11 @@ class TestDevanagariScript: assert output_pdf.stat().st_size > 0 # Extract text and verify Devanagari content - text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True) + text = subprocess.check_output( + ['pdftotext', '-enc', 'UTF-8', str(output_pdf), '-'], + text=True, + encoding='utf-8', + ) # Hindi: नमस्ते दुनिया (Hello world) assert 'नमस्ते' in text or 'दुनिया' in text @@ -367,7 +384,11 @@ class TestMultilingual: assert output_pdf.stat().st_size > 0 # Extract text from PDF - text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True) + text = subprocess.check_output( + ['pdftotext', '-enc', 'UTF-8', str(output_pdf), '-'], + text=True, + encoding='utf-8', + ) # Verify both English and Arabic text are present assert 'English' in text or 'Text' in text or 'Here' in text @@ -420,7 +441,11 @@ class TestMultilingual: assert output_pdf.exists() # Text should still be extractable even though invisible - text = subprocess.check_output(['pdftotext', str(output_pdf), '-'], text=True) + text = subprocess.check_output( + ['pdftotext', '-enc', 'UTF-8', str(output_pdf), '-'], + text=True, + encoding='utf-8', + ) assert len(text.strip()) > 0 def test_multilingual_font_selection(self, multilingual_hocr, multi_font_manager): @@ -474,8 +499,9 @@ class TestBaselineHandling: for line in page.lines: if line.baseline: # Baseline should be reasonable - assert -1.0 <= line.baseline.slope <= 1.0, \ - "Baseline slope should be reasonable" + assert ( + -1.0 <= line.baseline.slope <= 1.0 + ), "Baseline slope should be reasonable" # ============================================================================= @@ -497,8 +523,9 @@ class TestFontCoverage: ] for sample in latin_samples: - assert multi_font_manager.has_all_glyphs('NotoSans-Regular', sample), \ - f"NotoSans should cover: {sample}" + assert multi_font_manager.has_all_glyphs( + 'NotoSans-Regular', sample + ), f"NotoSans should cover: {sample}" def test_noto_sans_arabic_coverage(self, multi_font_manager): """Test NotoSansArabic covers Arabic characters.""" @@ -539,8 +566,9 @@ class TestFontCoverage: ] for sample in cjk_samples: - assert multi_font_manager.has_all_glyphs('NotoSansCJK-Regular', sample), \ - f"NotoSansCJK should cover: {sample}" + assert multi_font_manager.has_all_glyphs( + 'NotoSansCJK-Regular', sample + ), f"NotoSansCJK should cover: {sample}" if __name__ == "__main__":