diff --git a/src/ocrmypdf/hocrtransform/__init__.py b/src/ocrmypdf/hocrtransform/__init__.py index 7dce171a..9502d8c5 100755 --- a/src/ocrmypdf/hocrtransform/__init__.py +++ b/src/ocrmypdf/hocrtransform/__init__.py @@ -1,18 +1,59 @@ -# SPDX-FileCopyrightText: 2023 James R. Barlow +# SPDX-FileCopyrightText: 2023-2025 James R. Barlow # SPDX-License-Identifier: MIT -"""Transform .hocr and page image to text PDF.""" +"""Transform OCR output to text-only PDFs. + +This package provides tools for: +1. Parsing OCR output (hOCR format) into generic OcrElement structures +2. Rendering OcrElement structures to searchable PDF text layers + +The architecture separates parsing from rendering, allowing: +- Support for multiple OCR input formats (hOCR, ALTO, custom engines) +- Independent improvements to text rendering +- Reuse of the OcrElement data model for other purposes + +Main components: +- OcrElement: Generic dataclass representing OCR output structure +- HocrParser: Parses hOCR files into OcrElement trees +- PdfTextRenderer: Renders OcrElement trees to PDF text layers +- HocrTransform: Backward-compatible wrapper combining parser and renderer +""" from __future__ import annotations from ocrmypdf.hocrtransform._hocr import ( - DebugRenderOptions, HocrTransform, HocrTransformError, ) +from ocrmypdf.hocrtransform.hocr_parser import ( + HocrParseError, + HocrParser, +) +from ocrmypdf.hocrtransform.ocr_element import ( + Baseline, + BoundingBox, + FontInfo, + OcrClass, + OcrElement, +) +from ocrmypdf.hocrtransform.pdf_renderer import ( + DebugRenderOptions, + PdfTextRenderer, +) __all__ = ( + # Backward-compatible API 'HocrTransform', 'HocrTransformError', 'DebugRenderOptions', + # New separated components + 'HocrParser', + 'HocrParseError', + 'PdfTextRenderer', + # OCR element data model + 'OcrElement', + 'OcrClass', + 'BoundingBox', + 'Baseline', + 'FontInfo', ) diff --git a/src/ocrmypdf/hocrtransform/_hocr.py b/src/ocrmypdf/hocrtransform/_hocr.py index 05d8c74e..317cd850 100644 --- a/src/ocrmypdf/hocrtransform/_hocr.py +++ b/src/ocrmypdf/hocrtransform/_hocr.py @@ -1,58 +1,33 @@ # SPDX-FileCopyrightText: 2010 Jonathan Brinley # SPDX-FileCopyrightText: 2013-2014 Julien Pfefferkorn -# SPDX-FileCopyrightText: 2023 James R. Barlow -# SPDX-FileCopyrightText: 2025 Odin Dahlström +# SPDX-FileCopyrightText: 2023-2025 James R. Barlow +# SPDX-FileCopyrightText: 2025 Odin Dahlstr\u00f6m # SPDX-License-Identifier: MIT -"""hOCR transform implementation.""" +"""hOCR transform implementation. + +This module provides backward-compatible HocrTransform class that wraps the +new separated HocrParser and PdfTextRenderer components. +""" from __future__ import annotations import logging -import os -import re -import unicodedata -from dataclasses import dataclass -from itertools import pairwise -from math import atan, pi +import warnings from pathlib import Path -from xml.etree import ElementTree -from pikepdf import Matrix, Name, Rectangle -from pikepdf.canvas import ( - BLACK, - BLUE, - CYAN, - DARKGREEN, - GREEN, - MAGENTA, - RED, - Canvas, - Text, - TextDirection, -) +from pikepdf import Name from ocrmypdf.hocrtransform._font import EncodableFont as Font from ocrmypdf.hocrtransform._font import GlyphlessFont +from ocrmypdf.hocrtransform.hocr_parser import HocrParseError, HocrParser +from ocrmypdf.hocrtransform.pdf_renderer import ( + DebugRenderOptions, + PdfTextRenderer, +) log = logging.getLogger(__name__) -INCH = 72.0 - -Element = ElementTree.Element - - -@dataclass -class DebugRenderOptions: - """A class for managing rendering options.""" - - render_paragraph_bbox: bool = False - render_baseline: bool = False - render_triangle: bool = False - render_line_bbox: bool = False - render_word_bbox: bool = False - render_space_bbox: bool = False - class HocrTransformError(Exception): """Error while applying hOCR transform.""" @@ -63,33 +38,10 @@ class HocrTransform: For details of the hOCR format, see: http://kba.github.io/hocr-spec/1.2/. - """ - box_pattern = re.compile( - r''' - bbox \s+ - (\d+) \s+ # left: uint - (\d+) \s+ # top: uint - (\d+) \s+ # right: uint - (\d+) # bottom: uint - ''', - re.VERBOSE, - ) - baseline_pattern = re.compile( - r''' - baseline \s+ - ([\-\+]?\d*\.?\d*) \s+ # +/- decimal float - ([\-\+]?\d+) # +/- int - ''', - re.VERBOSE, - ) - textangle_pattern = re.compile( - r''' - textangle \s+ - ([\-\+]?\d*\.?\d*) # +/- decimal float - ''', - re.VERBOSE, - ) + This class provides backward compatibility with existing code. Internally, + it uses the new HocrParser and PdfTextRenderer components. + """ def __init__( self, @@ -101,9 +53,22 @@ class HocrTransform: font: Font = GlyphlessFont(), debug_render_options: DebugRenderOptions | None = None, ): - """Initialize the HocrTransform object.""" + """Initialize the HocrTransform object. + + Args: + hocr_filename: Path to the hOCR file + dpi: Resolution of the source image in dots per inch + debug: Deprecated; use debug_render_options instead + fontname: PDF font name to use + font: Font implementation for encoding and metrics + debug_render_options: Options for debug visualization + """ if debug: - log.warning("Use debug_render_options instead", DeprecationWarning) + warnings.warn( + "Use debug_render_options instead of debug parameter", + DeprecationWarning, + stacklevel=2, + ) self.render_options = DebugRenderOptions( render_baseline=debug, render_triangle=debug, @@ -114,74 +79,26 @@ class HocrTransform: ) else: self.render_options = debug_render_options or DebugRenderOptions() + self.dpi = dpi - self.hocr = ElementTree.parse(os.fspath(hocr_filename)) self._fontname = fontname self._font = font + self._hocr_filename = Path(hocr_filename) - # if the hOCR file has a namespace, ElementTree requires its use to - # find elements - matches = re.match(r'({.*})html', self.hocr.getroot().tag) - self.xmlns = '' - if matches: - self.xmlns = matches.group(1) + # Parse the hOCR file + try: + parser = HocrParser(hocr_filename) + self._page = parser.parse() + except HocrParseError as e: + raise HocrTransformError(str(e)) from e - for div in self.hocr.findall(self._child_xpath('div', 'ocr_page')): - coords = self.element_coordinates(div) - if not coords: - raise HocrTransformError("hocr file is missing page dimensions") - self.width = (coords.urx - coords.llx) / (self.dpi / INCH) - self.height = (coords.ury - coords.lly) / (self.dpi / INCH) - # Stop after first div that has page coordinates - break + if self._page.bbox is None: + raise HocrTransformError("hocr file is missing page dimensions") - def _get_element_text(self, element: Element) -> str: - """Return the textual content of the element and its children.""" - text = element.text if element.text is not None else '' - for child in element: - text += self._get_element_text(child) - text += element.tail if element.tail is not None else '' - return text - - @classmethod - def element_coordinates(cls, element: Element) -> Rectangle | None: - """Get coordinates of the bounding box around an element.""" - matches = cls.box_pattern.search(element.attrib.get('title', '')) - if not matches: - return None - return Rectangle( - float(matches.group(1)), # llx = left - float(matches.group(2)), # lly = top - float(matches.group(3)), # urx = right - float(matches.group(4)), # ury = bottom - ) - - @classmethod - def baseline(cls, element: Element) -> tuple[float, float]: - """Get baseline's slope and intercept.""" - matches = cls.baseline_pattern.search(element.attrib.get('title', '')) - if not matches: - return (0.0, 0.0) - return float(matches.group(1)), int(matches.group(2)) - - @classmethod - def textangle(cls, element: Element) -> float: - """Get text angle of an element.""" - matches = cls.textangle_pattern.search(element.attrib.get('title', '')) - if not matches: - return 0.0 - return float(matches.group(1)) - - def _child_xpath(self, html_tag: str, html_class: str | None = None) -> str: - xpath = f".//{self.xmlns}{html_tag}" - if html_class: - xpath += f"[@class='{html_class}']" - return xpath - - @classmethod - def normalize_text(cls, s: str) -> str: - """Normalize the given text using the NFKC normalization form.""" - return unicodedata.normalize("NFKC", s) + # Calculate page size in PDF points + INCH = 72.0 + self.width = self._page.bbox.width / (self.dpi / INCH) + self.height = self._page.bbox.height / (self.dpi / INCH) def to_pdf( self, @@ -198,7 +115,7 @@ class HocrTransform: file. It can have a lower resolution, different color mode, etc. - Arguments: + Args: out_filename: Path of PDF to write. image_filename: Image to use for this file. If omitted, the OCR text is shown. @@ -206,335 +123,25 @@ class HocrTransform: selectable but never drawn. If False, text is visible and may be seen if the image is skipped or deleted in Acrobat. """ - # create the PDF file - # page size in points (1/72 in.) - canvas = Canvas(page_size=(self.width, self.height)) - canvas.add_font(self._fontname, self._font) - page_matrix = ( - Matrix() - .translated(0, self.height) - .scaled(1, -1) - .scaled(INCH / self.dpi, INCH / self.dpi) + renderer = PdfTextRenderer( + page=self._page, + dpi=self.dpi, + fontname=self._fontname, + font=self._font, + debug_render_options=self.render_options, ) - log.debug(page_matrix) - with canvas.do.save_state(cm=page_matrix): - self._debug_draw_paragraph_boxes(canvas) - found_lines = False - for par in self.hocr.iterfind(self._child_xpath('p', 'ocr_par')): - for line in ( - element - for element in par.iterfind(self._child_xpath('span')) - if 'class' in element.attrib - and element.attrib['class'] - in {'ocr_header', 'ocr_line', 'ocr_textfloat', 'ocr_caption'} - ): - found_lines = True - direction = self._get_text_direction(par) - inject_word_breaks = self._get_inject_word_breaks(par) - self._do_line( - canvas, - line, - "ocrx_word", - invisible_text, - direction, - inject_word_breaks, - ) - if not found_lines: - # Tesseract did not report any lines (just words) - root = self.hocr.find(self._child_xpath('div', 'ocr_page')) - direction = self._get_text_direction(root) - self._do_line( - canvas, - root, - "ocrx_word", - invisible_text, - direction, - True, - ) - # put the image on the page, scaled to fill the page - if image_filename is not None: - canvas.do.draw_image( - image_filename, 0, 0, width=self.width, height=self.height - ) + renderer.render( + out_filename=out_filename, + image_filename=image_filename, + invisible_text=invisible_text, + ) - # finish up the page and save it - canvas.to_pdf().save(out_filename) + @property + def page(self): + """Get the parsed OcrElement page. - def _get_text_direction(self, par): - """Get the text direction of the paragraph. - - Arabic, Hebrew, Persian, are right-to-left languages. - When the paragraph element is None, defaults to left-to-right. + Returns: + The root OcrElement representing the parsed page """ - if par is None: - return TextDirection.LTR - - return ( - TextDirection.RTL - if par.attrib.get('dir', 'ltr') == 'rtl' - else TextDirection.LTR - ) - - def _get_inject_word_breaks(self, par): - """Determine whether word breaks should be injected. - - In Chinese, Japanese, and Korean, word breaks are not injected, because - words are usually one or two characters and separators are usually explicit. - In all other languages, we inject word breaks to help word segmentation. - """ - lang = par.attrib.get('lang', '') - log.debug(lang) - if lang in {'chi_sim', 'chi_tra', 'jpn', 'kor'}: - return False - return True - - @classmethod - def polyval(cls, poly, x): # pragma: no cover - """Calculate the value of a polynomial at a point.""" - return x * poly[0] + poly[1] - - def _do_line( - self, - canvas: Canvas, - line: Element | None, - elemclass: str, - invisible_text: bool, - text_direction: TextDirection, - inject_word_breaks: bool, - ): - """Render the text for a given line. - - The canvas's coordinate system must be configured so that hOCR pixel - coordinates are mapped to PDF coordinates. - """ - if line is None: - return - # line_min_aabb (which is created from the "bbox" hOCR property) is so named - # because a Rectangle instance is always an AABB (it has no orientation). - # However, this means that for non-zero values of the "textangle" hOCR - # property, line_min_aabb is not the true bounding box of the hOCR line, - # but rather the minimum AABB that encloses the bounding box of the line. - # The true bounding box of the line must be seen as an OBB, due to the - # existance of the "textangle" hOCR property. - line_min_aabb = self.element_coordinates(line) - if not line_min_aabb: - return - if line_min_aabb.ury <= line_min_aabb.lly: - log.error( - "line box is invalid so we cannot render it: box=%s text=%s", - line_min_aabb, - self._get_element_text(line), - ) - return - self._debug_draw_line_bbox(canvas, line_min_aabb) - - # Even though line_min_aabb is not the true bounding box of the line, - # it is still possible to derive an AABB (Rectangle) from it that is - # the same size as the true bounding box of the line, - # if we use a coordinate system that is axis-aligned with respect to - # the rotation of the OBB (textangle). - # line_size_aabb_matrix is a transform matrix for such a coordinate - # system, and line_size_aabb is thus an AABB with the same - # size as the true bounding box of the line. - top_left_corner = (line_min_aabb.llx, line_min_aabb.lly) - line_size_aabb_matrix = ( - Matrix() - .translated(*top_left_corner) - # Note: negative sign (textangle is counter-clockwise, see hOCR spec) - .rotated(-self.textangle(line)) - ) - line_size_aabb = line_size_aabb_matrix.inverse().transform(line_min_aabb) - - slope, intercept = self.baseline(line) - if abs(slope) < 0.005: - slope = 0.0 - slope_angle = atan(slope) - - # Final PDF-perspective (bottom-left corner) transform matrix for the - # text baseline, which has an intercept and slope relative to the OBB. - # See "bbox", "textangle" and "baseline" in the hOCR spec for more details. - baseline_matrix = ( - line_size_aabb_matrix - # Translate from hOCR perspective (top-left corner) to PDF perspective - # (bottom-left corner). - # Note: it would be incorrect to use line_min_aabb.height here because - # it is not the true height of the OBB of the line, if textangle != 0. - .translated(0, line_size_aabb.height) - .translated(0, intercept) - .rotated(slope_angle / pi * 180) - ) - - with canvas.do.save_state(cm=baseline_matrix): - text = Text(direction=text_direction) - fontsize = line_size_aabb.height + intercept - text.font(self._fontname, fontsize) - text.render_mode(3 if invisible_text else 0) - - self._debug_draw_baseline( - canvas, baseline_matrix.inverse().transform(line_min_aabb), 0 - ) - - canvas.do.fill_color(BLACK) # text in black - elements = line.findall(self._child_xpath('span', elemclass)) - for elem, next_elem in pairwise(elements + [None]): - self._do_line_word( - canvas, - baseline_matrix, - text, - fontsize, - elem, - next_elem, - text_direction, - inject_word_breaks, - ) - canvas.do.draw_text(text) - - def _do_line_word( - self, - canvas: Canvas, - line_matrix: Matrix, - text: Text, - fontsize: float, - elem: Element | None, - next_elem: Element | None, - text_direction: TextDirection, - inject_word_breaks: bool, - ): - """Render the text for a single word.""" - if elem is None: - return - elemtxt = self.normalize_text(self._get_element_text(elem).strip()) - if elemtxt == '': - return - - hocr_box = self.element_coordinates(elem) - if hocr_box is None: - return - box = line_matrix.inverse().transform(hocr_box) - font_width = self._font.text_width(elemtxt, fontsize) - - # Debug sketches - self._debug_draw_word_triangle(canvas, box) - self._debug_draw_word_bbox(canvas, box) - - # If this word is 0 units wide, our best bet seems to be to suppress this text - if text_direction == TextDirection.RTL: - log.info("RTL: %s", elemtxt) - if font_width > 0: - if text_direction == TextDirection.LTR: - text.text_transform(Matrix(1, 0, 0, -1, box.llx, 0)) - elif text_direction == TextDirection.RTL: - text.text_transform(Matrix(-1, 0, 0, -1, box.llx + box.width, 0)) - text.horiz_scale(100 * box.width / font_width) - text.show(self._font.text_encode(elemtxt)) - - # Get coordinates of the next word (if there is one) - hocr_next_box = ( - self.element_coordinates(next_elem) if next_elem is not None else None - ) - if hocr_next_box is None: - return - # Render a space between this word and the next word. The explicit space helps - # PDF viewers identify the word break, and horizontally scaling it to - # occupy the space the between the words helps the PDF viewer - # avoid combiningthewordstogether. - if not inject_word_breaks: - return - next_box = line_matrix.inverse().transform(hocr_next_box) - if text_direction == TextDirection.LTR: - space_box = Rectangle(box.urx, box.lly, next_box.llx, next_box.ury) - elif text_direction == TextDirection.RTL: - space_box = Rectangle(next_box.urx, box.lly, box.llx, next_box.ury) - self._debug_draw_space_bbox(canvas, space_box) - space_width = self._font.text_width(' ', fontsize) - if space_width > 0 and space_box.width > 0: - if text_direction == TextDirection.LTR: - text.text_transform(Matrix(1, 0, 0, -1, space_box.llx, 0)) - elif text_direction == TextDirection.RTL: - text.text_transform( - Matrix(-1, 0, 0, -1, space_box.llx + space_box.width, 0) - ) - text.horiz_scale(100 * space_box.width / space_width) - text.show(self._font.text_encode(' ')) - - def _debug_draw_paragraph_boxes(self, canvas: Canvas, color=CYAN): - """Draw boxes around paragraphs in the document.""" - if not self.render_options.render_paragraph_bbox: # pragma: no cover - return - with canvas.do.save_state(): - # draw box around paragraph - canvas.do.stroke_color(color).line_width(0.1) - for elem in self.hocr.iterfind(self._child_xpath('p', 'ocr_par')): - elemtxt = self._get_element_text(elem).strip() - if len(elemtxt) == 0: - continue - ocr_par = self.element_coordinates(elem) - if ocr_par is None: - continue - canvas.do.rect( - ocr_par.llx, ocr_par.lly, ocr_par.width, ocr_par.height, fill=False - ) - - def _debug_draw_line_bbox(self, canvas: Canvas, line_box: Rectangle, color=BLUE): - """Render the bounding box of a text line.""" - if not self.render_options.render_line_bbox: # pragma: no cover - return - with canvas.do.save_state(): - canvas.do.stroke_color(color).line_width(0.15).rect( - line_box.llx, line_box.lly, line_box.width, line_box.height, fill=False - ) - - def _debug_draw_word_triangle( - self, canvas: Canvas, box: Rectangle, color=RED, line_width=0.1 - ): - """Render a triangle that conveys word height and drawing direction.""" - if not self.render_options.render_triangle: # pragma: no cover - return - with canvas.do.save_state(): - canvas.do.stroke_color(color).line_width(line_width).line( - box.llx, box.lly, box.urx, box.lly - ).line(box.urx, box.lly, box.llx, box.ury).line( - box.llx, box.lly, box.llx, box.ury - ) - - def _debug_draw_word_bbox( - self, canvas: Canvas, box: Rectangle, color=GREEN, line_width=0.1 - ): - """Render a box depicting the word.""" - if not self.render_options.render_word_bbox: # pragma: no cover - return - with canvas.do.save_state(): - canvas.do.stroke_color(color).line_width(line_width).rect( - box.llx, box.lly, box.width, box.height, fill=False - ) - - def _debug_draw_space_bbox( - self, canvas: Canvas, box: Rectangle, color=DARKGREEN, line_width=0.1 - ): - """Render a box depicting the space between two words.""" - if not self.render_options.render_space_bbox: # pragma: no cover - return - with canvas.do.save_state(): - canvas.do.fill_color(color).line_width(line_width).rect( - box.llx, box.lly, box.width, box.height, fill=True - ) - - def _debug_draw_baseline( - self, - canvas: Canvas, - line_box: Rectangle, - baseline_lly, - color=MAGENTA, - line_width=0.25, - ): - """Render the text baseline.""" - if not self.render_options.render_baseline: - return - with canvas.do.save_state(): - canvas.do.stroke_color(color).line_width(line_width).line( - line_box.llx, - baseline_lly, - line_box.urx, - baseline_lly, - ) + return self._page diff --git a/src/ocrmypdf/hocrtransform/hocr_parser.py b/src/ocrmypdf/hocrtransform/hocr_parser.py new file mode 100644 index 00000000..898da4c8 --- /dev/null +++ b/src/ocrmypdf/hocrtransform/hocr_parser.py @@ -0,0 +1,521 @@ +# SPDX-FileCopyrightText: 2010 Jonathan Brinley +# SPDX-FileCopyrightText: 2013-2014 Julien Pfefferkorn +# SPDX-FileCopyrightText: 2023-2025 James R. Barlow +# SPDX-License-Identifier: MIT + +"""Parser for hOCR format files. + +This module provides functionality to parse hOCR files (HTML-based OCR format) +and convert them to the engine-agnostic OcrElement tree structure. + +For details of the hOCR format, see: +http://kba.github.io/hocr-spec/1.2/ +""" + +from __future__ import annotations + +import logging +import os +import re +import unicodedata +from pathlib import Path +from typing import Literal, cast +from xml.etree import ElementTree + +from ocrmypdf.hocrtransform.ocr_element import ( + Baseline, + BoundingBox, + FontInfo, + OcrClass, + OcrElement, +) + +TextDirection = Literal["ltr", "rtl"] + +log = logging.getLogger(__name__) + +Element = ElementTree.Element + + +class HocrParseError(Exception): + """Error while parsing hOCR file.""" + + +class HocrParser: + """Parser for hOCR format files. + + Converts hOCR XML/HTML files into OcrElement trees. + + The hOCR format uses HTML with special class attributes (ocr_page, ocr_line, + ocrx_word, etc.) and a title attribute containing properties like bbox, + baseline, and confidence scores. + """ + + # Regex patterns for parsing hOCR title attributes + _bbox_pattern = re.compile( + r''' + bbox \s+ + (\d+) \s+ # left: uint + (\d+) \s+ # top: uint + (\d+) \s+ # right: uint + (\d+) # bottom: uint + ''', + re.VERBOSE, + ) + + _baseline_pattern = re.compile( + r''' + baseline \s+ + ([\-\+]?\d*\.?\d*) \s+ # slope: +/- decimal float + ([\-\+]?\d+) # intercept: +/- int + ''', + re.VERBOSE, + ) + + _textangle_pattern = re.compile( + r''' + textangle \s+ + ([\-\+]?\d*\.?\d*) # angle: +/- decimal float + ''', + re.VERBOSE, + ) + + _x_wconf_pattern = re.compile( + r''' + x_wconf \s+ + (\d+) # confidence: uint (0-100) + ''', + re.VERBOSE, + ) + + _x_fsize_pattern = re.compile( + r''' + x_fsize \s+ + (\d*\.?\d+) # font size: float + ''', + re.VERBOSE, + ) + + _x_font_pattern = re.compile( + r''' + x_font \s+ + (\S+) # font name: non-whitespace string + ''', + re.VERBOSE, + ) + + _ppageno_pattern = re.compile( + r''' + ppageno \s+ + (\d+) # page number: uint + ''', + re.VERBOSE, + ) + + _scan_res_pattern = re.compile( + r''' + scan_res \s+ + (\d+) \s+ # x resolution + (\d+) # y resolution + ''', + re.VERBOSE, + ) + + def __init__(self, hocr_file: str | Path): + """Initialize the parser with an hOCR file. + + Args: + hocr_file: Path to the hOCR file to parse + + Raises: + HocrParseError: If the file cannot be parsed + """ + self._hocr_path = Path(hocr_file) + try: + self._tree = ElementTree.parse(os.fspath(hocr_file)) + except ElementTree.ParseError as e: + raise HocrParseError(f"Failed to parse hOCR file: {e}") from e + + # Detect XML namespace + root_tag = self._tree.getroot().tag + matches = re.match(r'({.*})html', root_tag) + self._xmlns = matches.group(1) if matches else '' + + def parse(self) -> OcrElement: + """Parse the hOCR file and return an OcrElement tree. + + Returns: + The root OcrElement (ocr_page) containing the document structure + + Raises: + HocrParseError: If no ocr_page element is found + """ + # Find the first ocr_page element + page_div = self._tree.find(self._xpath('div', 'ocr_page')) + if page_div is None: + raise HocrParseError("No ocr_page element found in hOCR file") + + return self._parse_page(page_div) + + def _xpath(self, html_tag: str, html_class: str | None = None) -> str: + """Build an XPath expression for finding elements. + + Args: + html_tag: HTML tag name (e.g., 'div', 'span', 'p') + html_class: Optional class attribute to match + + Returns: + XPath expression string + """ + xpath = f".//{self._xmlns}{html_tag}" + if html_class: + xpath += f"[@class='{html_class}']" + return xpath + + def _parse_page(self, page_elem: Element) -> OcrElement: + """Parse an ocr_page element. + + Args: + page_elem: The XML element with class="ocr_page" + + Returns: + OcrElement representing the page + """ + title = page_elem.attrib.get('title', '') + + bbox = self._parse_bbox(title) + if bbox is None: + raise HocrParseError("ocr_page missing bbox") + + # Parse page-level properties + page_number = self._parse_ppageno(title) + dpi = self._parse_scan_res(title) + + page = OcrElement( + ocr_class=OcrClass.PAGE, + bbox=bbox, + page_number=page_number, + dpi=dpi, + ) + + # Parse child paragraphs + for par_elem in page_elem.iterfind(self._xpath('p', 'ocr_par')): + paragraph = self._parse_paragraph(par_elem) + if paragraph is not None: + page.children.append(paragraph) + + # If no paragraphs found, check for words directly under page + # (some Tesseract output structures) + if not page.children: + for word_elem in page_elem.iterfind(self._xpath('span', 'ocrx_word')): + word = self._parse_word(word_elem) + if word is not None: + page.children.append(word) + + return page + + def _parse_paragraph(self, par_elem: Element) -> OcrElement | None: + """Parse an ocr_par element. + + Args: + par_elem: The XML element with class="ocr_par" + + Returns: + OcrElement representing the paragraph, or None if empty + """ + title = par_elem.attrib.get('title', '') + bbox = self._parse_bbox(title) + + # Get direction and language from attributes + dir_attr = par_elem.attrib.get('dir') + direction: TextDirection | None = ( + cast(TextDirection, dir_attr) if dir_attr in ('ltr', 'rtl') else None + ) + + language = par_elem.attrib.get('lang') + + paragraph = OcrElement( + ocr_class=OcrClass.PARAGRAPH, + bbox=bbox, + direction=direction, + language=language, + ) + + # Parse child lines + line_classes = { + 'ocr_line', + 'ocr_header', + 'ocr_footer', + 'ocr_caption', + 'ocr_textfloat', + } + for span_elem in par_elem.iterfind(self._xpath('span')): + elem_class = span_elem.attrib.get('class', '') + if elem_class in line_classes: + line = self._parse_line(span_elem, elem_class, direction, language) + if line is not None: + paragraph.children.append(line) + + # Return None if paragraph is empty + if not paragraph.children: + return None + + return paragraph + + def _parse_line( + self, + line_elem: Element, + ocr_class: str, + parent_direction: TextDirection | None, + parent_language: str | None, + ) -> OcrElement | None: + """Parse a line element (ocr_line, ocr_header, etc.). + + Args: + line_elem: The XML element representing the line + ocr_class: The hOCR class of the line + parent_direction: Text direction inherited from parent + parent_language: Language inherited from parent + + Returns: + OcrElement representing the line, or None if empty + """ + title = line_elem.attrib.get('title', '') + bbox = self._parse_bbox(title) + + if bbox is None: + return None + + baseline = self._parse_baseline(title) + textangle = self._parse_textangle(title) + + # Inherit direction and language from parent if not specified + dir_attr = line_elem.attrib.get('dir') + if dir_attr in ('ltr', 'rtl'): + direction: TextDirection | None = cast(TextDirection, dir_attr) + else: + direction = parent_direction + + language = line_elem.attrib.get('lang') or parent_language + + line = OcrElement( + ocr_class=ocr_class, + bbox=bbox, + baseline=baseline, + textangle=textangle, + direction=direction, + language=language, + ) + + # Parse child words + for word_elem in line_elem.iterfind(self._xpath('span', 'ocrx_word')): + word = self._parse_word(word_elem) + if word is not None: + line.children.append(word) + + # Return None if line has no words + if not line.children: + return None + + return line + + def _parse_word(self, word_elem: Element) -> OcrElement | None: + """Parse an ocrx_word element. + + Args: + word_elem: The XML element with class="ocrx_word" + + Returns: + OcrElement representing the word, or None if empty + """ + title = word_elem.attrib.get('title', '') + bbox = self._parse_bbox(title) + + # Get the text content + text = self._get_element_text(word_elem) + text = self._normalize_text(text) + + if not text: + return None + + # Parse confidence (x_wconf is 0-100, convert to 0.0-1.0) + confidence = self._parse_x_wconf(title) + if confidence is not None: + confidence = confidence / 100.0 + + # Parse font info + font = self._parse_font_info(title) + + return OcrElement( + ocr_class=OcrClass.WORD, + bbox=bbox, + text=text, + confidence=confidence, + font=font, + ) + + def _get_element_text(self, element: Element) -> str: + """Get the full text content of an element including children. + + Args: + element: XML element + + Returns: + Combined text content + """ + text = element.text if element.text is not None else '' + for child in element: + text += self._get_element_text(child) + text += element.tail if element.tail is not None else '' + return text + + @staticmethod + def _normalize_text(text: str) -> str: + """Normalize text using NFKC normalization. + + This splits ligatures and combines diacritics. + + Args: + text: Raw text + + Returns: + Normalized text, stripped of leading/trailing whitespace + """ + return unicodedata.normalize("NFKC", text).strip() + + def _parse_bbox(self, title: str) -> BoundingBox | None: + """Parse a bbox from an hOCR title attribute. + + Args: + title: The title attribute value + + Returns: + BoundingBox or None if not found + """ + match = self._bbox_pattern.search(title) + if not match: + return None + + try: + return BoundingBox( + left=float(match.group(1)), + top=float(match.group(2)), + right=float(match.group(3)), + bottom=float(match.group(4)), + ) + except ValueError: + return None + + def _parse_baseline(self, title: str) -> Baseline | None: + """Parse baseline from an hOCR title attribute. + + Args: + title: The title attribute value + + Returns: + Baseline or None if not found + """ + match = self._baseline_pattern.search(title) + if not match: + return None + + try: + return Baseline( + slope=float(match.group(1)) if match.group(1) else 0.0, + intercept=float(match.group(2)), + ) + except ValueError: + return None + + def _parse_textangle(self, title: str) -> float | None: + """Parse textangle from an hOCR title attribute. + + Args: + title: The title attribute value + + Returns: + Angle in degrees or None if not found + """ + match = self._textangle_pattern.search(title) + if not match: + return None + + try: + return float(match.group(1)) + except ValueError: + return None + + def _parse_x_wconf(self, title: str) -> float | None: + """Parse word confidence from an hOCR title attribute. + + Args: + title: The title attribute value + + Returns: + Confidence (0-100) or None if not found + """ + match = self._x_wconf_pattern.search(title) + if not match: + return None + + try: + return float(match.group(1)) + except ValueError: + return None + + def _parse_ppageno(self, title: str) -> int | None: + """Parse physical page number from an hOCR title attribute. + + Args: + title: The title attribute value + + Returns: + Page number or None if not found + """ + match = self._ppageno_pattern.search(title) + if not match: + return None + + try: + return int(match.group(1)) + except ValueError: + return None + + def _parse_scan_res(self, title: str) -> float | None: + """Parse scan resolution (DPI) from an hOCR title attribute. + + Args: + title: The title attribute value + + Returns: + DPI (using first value if x and y differ) or None if not found + """ + match = self._scan_res_pattern.search(title) + if not match: + return None + + try: + # Use the first (x) resolution value + return float(match.group(1)) + except ValueError: + return None + + def _parse_font_info(self, title: str) -> FontInfo | None: + """Parse font information from an hOCR title attribute. + + Args: + title: The title attribute value + + Returns: + FontInfo or None if no font info found + """ + font_match = self._x_font_pattern.search(title) + size_match = self._x_fsize_pattern.search(title) + + if not font_match and not size_match: + return None + + return FontInfo( + name=font_match.group(1) if font_match else None, + size=float(size_match.group(1)) if size_match else None, + ) diff --git a/src/ocrmypdf/hocrtransform/ocr_element.py b/src/ocrmypdf/hocrtransform/ocr_element.py new file mode 100644 index 00000000..fc825919 --- /dev/null +++ b/src/ocrmypdf/hocrtransform/ocr_element.py @@ -0,0 +1,267 @@ +# SPDX-FileCopyrightText: 2025 James R. Barlow +# SPDX-License-Identifier: MPL-2.0 + +"""OCR element dataclasses for representing OCR output structure. + +This module provides a generic, engine-agnostic representation of OCR output. +The OcrElement dataclass can represent structural units from any OCR source +(hOCR, ALTO, custom engines, etc.) in a unified format suitable for rendering. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Literal + + +@dataclass +class BoundingBox: + """An axis-aligned bounding box in pixel coordinates. + + Coordinates use top-left origin (standard for images and hOCR). + + Attributes: + left: Left edge x-coordinate + top: Top edge y-coordinate + right: Right edge x-coordinate + bottom: Bottom edge y-coordinate + """ + + left: float + top: float + right: float + bottom: float + + @property + def width(self) -> float: + """Width of the bounding box.""" + return self.right - self.left + + @property + def height(self) -> float: + """Height of the bounding box.""" + return self.bottom - self.top + + def __post_init__(self): + """Validate bounding box coordinates.""" + if self.right < self.left: + raise ValueError( + f"Invalid bounding box: right ({self.right}) < left ({self.left})" + ) + if self.bottom < self.top: + raise ValueError( + f"Invalid bounding box: bottom ({self.bottom}) < top ({self.top})" + ) + + +@dataclass +class Baseline: + """Text baseline information. + + The baseline is represented as a linear equation: y = slope * x + intercept. + This describes the line along which text characters sit, relative to the + bottom-left corner of the line's bounding box. + + In hOCR, the baseline is specified relative to the bottom of the line's bbox, + with the intercept being the vertical offset from the bottom and the slope + representing rotation (positive = ascending left-to-right). + + Attributes: + slope: Slope of the baseline (rise over run) + intercept: Y-intercept of the baseline (vertical offset from bbox bottom) + """ + + slope: float = 0.0 + intercept: float = 0.0 + + +@dataclass +class FontInfo: + """Font information for text rendering. + + Attributes: + name: Font family name (e.g., "Times New Roman") + size: Font size in points + bold: Whether the font is bold + italic: Whether the font is italic + monospace: Whether the font is monospace + serif: Whether the font is serif (vs sans-serif) + smallcaps: Whether the font uses small caps + underline: Whether the text is underlined + """ + + name: str | None = None + size: float | None = None + bold: bool = False + italic: bool = False + monospace: bool = False + serif: bool = False + smallcaps: bool = False + underline: bool = False + + +@dataclass +class OcrElement: + """A generic OCR element representing any structural unit of OCR output. + + OcrElements form a tree structure where pages contain paragraphs, paragraphs + contain lines, lines contain words, etc. The specific hierarchy depends on + the OCR engine, but this dataclass can represent any of these levels. + + The ocr_class field uses hOCR naming conventions (ocr_page, ocr_par, ocr_line, + ocrx_word, etc.) as a common vocabulary, but elements from other sources can + map to these classes. + + Common hOCR classes: + - ocr_page: The root element for a page + - ocr_carea: A content/column area + - ocr_par: A paragraph + - ocr_line: A line of text + - ocr_header: A header line + - ocr_footer: A footer line + - ocr_caption: A caption line + - ocr_textfloat: A floating text element + - ocrx_word: A single word + + Attributes: + ocr_class: The element type (e.g., "ocr_page", "ocr_line", "ocrx_word") + bbox: Axis-aligned bounding box in source pixel coordinates (top-left origin) + poly: Polygon vertices for oriented/non-rectangular bounds + text: Text content (primarily for leaf nodes like words) + confidence: OCR confidence score (0.0-1.0) + children: Child elements (hierarchical structure) + direction: Text direction ("ltr" or "rtl") + language: Language code (e.g., "eng", "deu", "chi_sim") + baseline: Text baseline information (slope and intercept) + textangle: Text rotation angle in degrees (counter-clockwise from horizontal) + font: Font information (name, size, style) + dpi: Image resolution in dots per inch (typically for page-level) + page_number: Physical page number (0-indexed) + logical_page_number: Logical page number (as printed on the page) + """ + + ocr_class: str + + # Bounding boxes + bbox: BoundingBox | None = None + poly: list[tuple[float, float]] | None = None + + # Text content + text: str = "" + + # Confidence (0.0-1.0) + confidence: float | None = None + + # Children (hierarchical structure) + children: list[OcrElement] = field(default_factory=list) + + # Text direction and language + direction: Literal["ltr", "rtl"] | None = None + language: str | None = None + + # Baseline (for lines) + baseline: Baseline | None = None + + # Rotation angle in degrees (counter-clockwise) + textangle: float | None = None + + # Font information + font: FontInfo | None = None + + # Page-level properties + dpi: float | None = None + page_number: int | None = None + logical_page_number: int | None = None + + def iter_by_class(self, *ocr_classes: str) -> list[OcrElement]: + """Iterate over all descendants matching the given class(es). + + Args: + *ocr_classes: One or more ocr_class values to match + + Returns: + List of all matching descendant elements (depth-first order) + """ + result = [] + if self.ocr_class in ocr_classes: + result.append(self) + for child in self.children: + result.extend(child.iter_by_class(*ocr_classes)) + return result + + def find_by_class(self, *ocr_classes: str) -> OcrElement | None: + """Find the first descendant matching the given class(es). + + Args: + *ocr_classes: One or more ocr_class values to match + + Returns: + The first matching element, or None if not found + """ + if self.ocr_class in ocr_classes: + return self + for child in self.children: + result = child.find_by_class(*ocr_classes) + if result is not None: + return result + return None + + def get_text_recursive(self) -> str: + """Get the combined text of this element and all descendants. + + Returns: + Combined text content, with words separated by spaces + """ + if self.text: + return self.text + texts = [child.get_text_recursive() for child in self.children] + return " ".join(t for t in texts if t) + + @property + def words(self) -> list[OcrElement]: + """Get all word elements (ocrx_word) in this element's subtree.""" + return self.iter_by_class("ocrx_word") + + @property + def lines(self) -> list[OcrElement]: + """Get all line elements in this element's subtree.""" + return self.iter_by_class( + "ocr_line", "ocr_header", "ocr_footer", "ocr_caption", "ocr_textfloat" + ) + + @property + def paragraphs(self) -> list[OcrElement]: + """Get all paragraph elements (ocr_par) in this element's subtree.""" + return self.iter_by_class("ocr_par") + + +# Type alias for text direction +TextDirection = Literal["ltr", "rtl"] + + +# hOCR class constants for convenience +class OcrClass: + """Constants for common OCR element classes.""" + + # Page-level + PAGE = "ocr_page" + CAREA = "ocr_carea" + + # Block-level + PARAGRAPH = "ocr_par" + + # Line-level + LINE = "ocr_line" + HEADER = "ocr_header" + FOOTER = "ocr_footer" + CAPTION = "ocr_caption" + TEXTFLOAT = "ocr_textfloat" + + # Word-level + WORD = "ocrx_word" + + # Character-level + CHAR = "ocrx_cinfo" + + # Line types (for convenience) + LINE_TYPES = frozenset({LINE, HEADER, FOOTER, CAPTION, TEXTFLOAT}) diff --git a/src/ocrmypdf/hocrtransform/pdf_renderer.py b/src/ocrmypdf/hocrtransform/pdf_renderer.py new file mode 100644 index 00000000..b103ad85 --- /dev/null +++ b/src/ocrmypdf/hocrtransform/pdf_renderer.py @@ -0,0 +1,544 @@ +# SPDX-FileCopyrightText: 2010 Jonathan Brinley +# SPDX-FileCopyrightText: 2013-2014 Julien Pfefferkorn +# SPDX-FileCopyrightText: 2023-2025 James R. Barlow +# SPDX-FileCopyrightText: 2025 Odin Dahlstr\u00f6m +# SPDX-License-Identifier: MIT + +"""PDF text renderer for OcrElement structures. + +This module provides functionality to render OcrElement trees to PDF files, +creating text layers that can be overlaid on scanned document images. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from itertools import pairwise +from math import atan, pi +from pathlib import Path + +from pikepdf import Matrix, Name, Rectangle +from pikepdf.canvas import ( + BLACK, + BLUE, + CYAN, + DARKGREEN, + GREEN, + MAGENTA, + RED, + Canvas, + Text, + TextDirection, +) + +from ocrmypdf.hocrtransform._font import EncodableFont as Font +from ocrmypdf.hocrtransform._font import GlyphlessFont +from ocrmypdf.hocrtransform.ocr_element import OcrClass, OcrElement + +log = logging.getLogger(__name__) + +INCH = 72.0 + +# CJK languages where word breaks should not be injected +CJK_LANGUAGES = frozenset({'chi_sim', 'chi_tra', 'jpn', 'kor'}) + + +@dataclass +class DebugRenderOptions: + """Options for debug visualization during rendering. + + When enabled, these options draw colored boxes and lines to visualize + the OCR structure, which is helpful for debugging layout issues. + + Attributes: + render_paragraph_bbox: Draw boxes around paragraphs (cyan) + render_baseline: Draw text baselines (magenta) + render_triangle: Draw direction triangles at word positions (red) + render_line_bbox: Draw boxes around lines (blue) + render_word_bbox: Draw boxes around words (green) + render_space_bbox: Draw boxes for inter-word spaces (dark green) + """ + + render_paragraph_bbox: bool = False + render_baseline: bool = False + render_triangle: bool = False + render_line_bbox: bool = False + render_word_bbox: bool = False + render_space_bbox: bool = False + + +class PdfTextRenderer: + """Renders OcrElement trees to PDF text layers. + + This class takes an OcrElement tree (typically parsed from hOCR or + another OCR format) and renders it to a PDF file. The text is positioned + according to the bounding boxes in the OcrElement structure, allowing + it to be overlaid on scanned document images. + + The renderer supports: + - Invisible text mode for selectable but hidden text + - Text direction (LTR and RTL) + - Baseline-aware positioning + - Text rotation (textangle) + - Word break injection for better PDF viewer segmentation + - Debug visualization options + """ + + def __init__( + self, + *, + page: OcrElement, + dpi: float, + fontname: Name = Name("/f-0-0"), + font: Font | None = None, + debug_render_options: DebugRenderOptions | None = None, + ): + """Initialize the PDF text renderer. + + Args: + page: The root OcrElement (should be ocr_page) + dpi: Resolution of the source image in dots per inch + fontname: PDF font name to use + font: Font implementation for encoding and metrics + debug_render_options: Options for debug visualization + """ + if page.ocr_class != OcrClass.PAGE: + raise ValueError(f"Expected ocr_page element, got {page.ocr_class}") + + if page.bbox is None: + raise ValueError("Page element must have a bounding box") + + self.page = page + self.dpi = dpi + self._fontname = fontname + self._font = font or GlyphlessFont() + self.render_options = debug_render_options or DebugRenderOptions() + + # Calculate page size in PDF points (1/72 inch) + self.width = page.bbox.width / (self.dpi / INCH) + self.height = page.bbox.height / (self.dpi / INCH) + + def render( + self, + *, + out_filename: Path, + image_filename: Path | None = None, + invisible_text: bool = True, + ) -> None: + """Render the OCR elements to a PDF file. + + Creates a PDF file with text positioned according to the OcrElement + bounding boxes. Optionally overlays an image on top of the text. + + Args: + out_filename: Path to write the PDF file + image_filename: Optional image to composite on top of text + invisible_text: If True, text is selectable but not visible. + If False, text is visible (useful for debugging). + """ + canvas = Canvas(page_size=(self.width, self.height)) + canvas.add_font(self._fontname, self._font) + + # Transform from hOCR pixel coordinates (top-left origin) to + # PDF coordinates (bottom-left origin) + page_matrix = ( + Matrix() + .translated(0, self.height) + .scaled(1, -1) + .scaled(INCH / self.dpi, INCH / self.dpi) + ) + + log.debug("Page matrix: %s", page_matrix) + + with canvas.do.save_state(cm=page_matrix): + self._render_debug_paragraph_boxes(canvas) + self._render_page_content(canvas, invisible_text) + + # Overlay image if provided + if image_filename is not None: + canvas.do.draw_image( + image_filename, 0, 0, width=self.width, height=self.height + ) + + canvas.to_pdf().save(out_filename) + + def _render_page_content(self, canvas: Canvas, invisible_text: bool) -> None: + """Render all text content from the page. + + Args: + canvas: The PDF canvas to render to + invisible_text: Whether text should be invisible + """ + found_lines = False + + # Iterate through paragraphs and their lines + for paragraph in self.page.paragraphs: + direction = self._get_text_direction(paragraph) + inject_word_breaks = self._should_inject_word_breaks(paragraph) + + for line in paragraph.lines: + found_lines = True + self._render_line( + canvas, + line, + invisible_text, + direction, + inject_word_breaks, + ) + + # Fallback: if no lines found in paragraphs, check for lines/words + # directly under page (some OCR output structures) + if not found_lines: + direction = self._get_text_direction(self.page) + inject_word_breaks = True + + # Try to find lines directly under page + for line in self.page.lines: + found_lines = True + self._render_line( + canvas, + line, + invisible_text, + direction, + inject_word_breaks, + ) + + # If still no lines, render words directly + if not found_lines: + for word in self.page.words: + self._render_standalone_word(canvas, word, invisible_text) + + def _get_text_direction(self, element: OcrElement) -> TextDirection: + """Get the text direction for an element. + + Args: + element: OcrElement to check + + Returns: + TextDirection.LTR or TextDirection.RTL + """ + if element.direction == "rtl": + return TextDirection.RTL + return TextDirection.LTR + + def _should_inject_word_breaks(self, element: OcrElement) -> bool: + """Determine whether word breaks should be injected. + + Word breaks are not injected for CJK languages where words are + typically one or two characters and separators are explicit. + + Args: + element: OcrElement to check (typically a paragraph) + + Returns: + True if word breaks should be injected + """ + language = element.language or '' + return language not in CJK_LANGUAGES + + def _render_line( + self, + canvas: Canvas, + line: OcrElement, + invisible_text: bool, + text_direction: TextDirection, + inject_word_breaks: bool, + ) -> None: + """Render a line of text. + + Args: + canvas: The PDF canvas (with page coordinate transform active) + line: The line element to render + invisible_text: Whether text should be invisible + text_direction: LTR or RTL text direction + inject_word_breaks: Whether to add spaces between words + """ + if line.bbox is None: + return + + # Validate line bbox + if line.bbox.height <= 0: + log.error( + "line box is invalid so we cannot render it: box=%s text=%s", + line.bbox, + line.get_text_recursive(), + ) + return + + # Convert BoundingBox to Rectangle for pikepdf operations + line_min_aabb = Rectangle( + line.bbox.left, + line.bbox.top, + line.bbox.right, + line.bbox.bottom, + ) + + self._render_debug_line_bbox(canvas, line_min_aabb) + + # Calculate the line's oriented bounding box transform + # The bbox from hOCR is the minimum AABB enclosing the rotated text + textangle = line.textangle or 0.0 + + top_left_corner = (line_min_aabb.llx, line_min_aabb.lly) + line_size_aabb_matrix = ( + Matrix() + .translated(*top_left_corner) + # Note: negative sign (textangle is counter-clockwise, see hOCR spec) + .rotated(-textangle) + ) + line_size_aabb = line_size_aabb_matrix.inverse().transform(line_min_aabb) + + # Get baseline information + slope = 0.0 + intercept = 0.0 + if line.baseline is not None: + slope = line.baseline.slope + intercept = line.baseline.intercept + + if abs(slope) < 0.005: + slope = 0.0 + slope_angle = atan(slope) + + # Create the baseline transform matrix + # Translate from hOCR perspective (top-left) to PDF perspective (bottom-left) + baseline_matrix = ( + line_size_aabb_matrix.translated(0, line_size_aabb.height) + .translated(0, intercept) + .rotated(slope_angle / pi * 180) + ) + + with canvas.do.save_state(cm=baseline_matrix): + text = Text(direction=text_direction) + fontsize = line_size_aabb.height + intercept + text.font(self._fontname, fontsize) + text.render_mode(3 if invisible_text else 0) + + self._render_debug_baseline( + canvas, baseline_matrix.inverse().transform(line_min_aabb), 0 + ) + + canvas.do.fill_color(BLACK) + + # Get words and render with inter-word spaces + words = line.children + for word, next_word in pairwise(words + [None]): + if word is not None: + self._render_word( + canvas, + baseline_matrix, + text, + fontsize, + word, + next_word, + text_direction, + inject_word_breaks, + ) + + canvas.do.draw_text(text) + + def _render_word( + self, + canvas: Canvas, + line_matrix: Matrix, + text: Text, + fontsize: float, + word: OcrElement, + next_word: OcrElement | None, + text_direction: TextDirection, + inject_word_breaks: bool, + ) -> None: + """Render a single word. + + Args: + canvas: The PDF canvas + line_matrix: Transform matrix for the line + text: Text object to add glyphs to + fontsize: Font size in points + word: The word element to render + next_word: The next word (for space calculation) or None + text_direction: LTR or RTL text direction + inject_word_breaks: Whether to add space after this word + """ + if word.bbox is None or not word.text: + return + + # Convert to Rectangle for transform + hocr_box = Rectangle( + word.bbox.left, word.bbox.top, word.bbox.right, word.bbox.bottom + ) + box = line_matrix.inverse().transform(hocr_box) + font_width = float(self._font.text_width(word.text, fontsize)) + + # Debug rendering + self._render_debug_word_triangle(canvas, box) + self._render_debug_word_bbox(canvas, box) + + # Skip zero-width words + if font_width <= 0: + return + + if text_direction == TextDirection.RTL: + log.info("RTL: %s", word.text) + + # Position and scale the word + if text_direction == TextDirection.LTR: + text.text_transform(Matrix(1, 0, 0, -1, box.llx, 0)) + elif text_direction == TextDirection.RTL: + text.text_transform(Matrix(-1, 0, 0, -1, box.llx + box.width, 0)) + + text.horiz_scale(100 * box.width / font_width) + text.show(self._font.text_encode(word.text)) + + # Render space to next word + if not inject_word_breaks or next_word is None or next_word.bbox is None: + return + + next_hocr_box = Rectangle( + next_word.bbox.left, + next_word.bbox.top, + next_word.bbox.right, + next_word.bbox.bottom, + ) + next_box = line_matrix.inverse().transform(next_hocr_box) + + if text_direction == TextDirection.LTR: + space_box = Rectangle(box.urx, box.lly, next_box.llx, next_box.ury) + elif text_direction == TextDirection.RTL: + space_box = Rectangle(next_box.urx, box.lly, box.llx, next_box.ury) + + self._render_debug_space_bbox(canvas, space_box) + + space_width = float(self._font.text_width(' ', fontsize)) + if space_width > 0 and space_box.width > 0: + if text_direction == TextDirection.LTR: + text.text_transform(Matrix(1, 0, 0, -1, space_box.llx, 0)) + elif text_direction == TextDirection.RTL: + text.text_transform( + Matrix(-1, 0, 0, -1, space_box.llx + space_box.width, 0) + ) + text.horiz_scale(100 * space_box.width / space_width) + text.show(self._font.text_encode(' ')) + + def _render_standalone_word( + self, canvas: Canvas, word: OcrElement, invisible_text: bool + ) -> None: + """Render a word that is not part of a line structure. + + This is a fallback for OCR output that doesn't have line structure. + + Args: + canvas: The PDF canvas + word: The word element to render + invisible_text: Whether text should be invisible + """ + if word.bbox is None or not word.text: + return + + # Simple rendering without baseline adjustment + box = Rectangle( + word.bbox.left, word.bbox.top, word.bbox.right, word.bbox.bottom + ) + + fontsize = box.height + font_width = float(self._font.text_width(word.text, fontsize)) + + if font_width <= 0: + return + + text = Text() + text.font(self._fontname, fontsize) + text.render_mode(3 if invisible_text else 0) + text.text_transform(Matrix(1, 0, 0, -1, box.llx, box.ury)) + text.horiz_scale(100 * box.width / font_width) + text.show(self._font.text_encode(word.text)) + + canvas.do.fill_color(BLACK) + canvas.do.draw_text(text) + + # Debug rendering methods + + def _render_debug_paragraph_boxes(self, canvas: Canvas, color=CYAN) -> None: + """Draw boxes around paragraphs.""" + if not self.render_options.render_paragraph_bbox: + return + + with canvas.do.save_state(): + canvas.do.stroke_color(color).line_width(0.1) + for paragraph in self.page.paragraphs: + if paragraph.bbox is None: + continue + if not paragraph.get_text_recursive(): + continue + canvas.do.rect( + paragraph.bbox.left, + paragraph.bbox.top, + paragraph.bbox.width, + paragraph.bbox.height, + fill=False, + ) + + def _render_debug_line_bbox( + self, canvas: Canvas, line_box: Rectangle, color=BLUE + ) -> None: + """Render the bounding box of a text line.""" + if not self.render_options.render_line_bbox: + return + with canvas.do.save_state(): + canvas.do.stroke_color(color).line_width(0.15).rect( + line_box.llx, line_box.lly, line_box.width, line_box.height, fill=False + ) + + def _render_debug_word_triangle( + self, canvas: Canvas, box: Rectangle, color=RED, line_width=0.1 + ) -> None: + """Render a triangle that conveys word height and direction.""" + if not self.render_options.render_triangle: + return + with canvas.do.save_state(): + canvas.do.stroke_color(color).line_width(line_width).line( + box.llx, box.lly, box.urx, box.lly + ).line(box.urx, box.lly, box.llx, box.ury).line( + box.llx, box.lly, box.llx, box.ury + ) + + def _render_debug_word_bbox( + self, canvas: Canvas, box: Rectangle, color=GREEN, line_width=0.1 + ) -> None: + """Render a box depicting the word.""" + if not self.render_options.render_word_bbox: + return + with canvas.do.save_state(): + canvas.do.stroke_color(color).line_width(line_width).rect( + box.llx, box.lly, box.width, box.height, fill=False + ) + + def _render_debug_space_bbox( + self, canvas: Canvas, box: Rectangle, color=DARKGREEN, line_width=0.1 + ) -> None: + """Render a box depicting the space between words.""" + if not self.render_options.render_space_bbox: + return + with canvas.do.save_state(): + canvas.do.fill_color(color).line_width(line_width).rect( + box.llx, box.lly, box.width, box.height, fill=True + ) + + def _render_debug_baseline( + self, + canvas: Canvas, + line_box: Rectangle, + baseline_lly: float, + color=MAGENTA, + line_width=0.25, + ) -> None: + """Render the text baseline.""" + if not self.render_options.render_baseline: + return + with canvas.do.save_state(): + canvas.do.stroke_color(color).line_width(line_width).line( + line_box.llx, + baseline_lly, + line_box.urx, + baseline_lly, + )