Begin transforming Tesseract into pluggable OCR engine

This commit is contained in:
James R. Barlow
2020-05-14 03:54:21 -07:00
parent 41eb54cc0a
commit 8174089c8b
6 changed files with 140 additions and 33 deletions
+12 -25
View File
@@ -379,11 +379,8 @@ def get_orientation_correction(preview, page_context):
"""
orient_conf = tesseract.get_orientation(
preview,
engine_mode=page_context.options.tesseract_oem,
timeout=page_context.options.tesseract_timeout,
tesseract_env=page_context.options.tesseract_env,
orient_conf = page_context.plugin_manager.hook.get_ocr_engine().get_orientation(
preview, page_context.options
)
correction = orient_conf.angle % 360
@@ -531,22 +528,17 @@ def create_ocr_image(image, page_context):
return output_file
def ocr_tesseract_hocr(input_file, page_context):
def ocr_engine_hocr(input_file, page_context):
hocr_out = page_context.get_path('ocr_hocr.hocr')
hocr_text_out = page_context.get_path('ocr_hocr.txt')
options = page_context.options
tesseract.generate_hocr(
ocr_engine = page_context.plugin_manager.hook.get_ocr_engine()
ocr_engine.generate_hocr(
input_file=input_file,
output_hocr=hocr_out,
output_text=hocr_text_out,
languages=options.language,
engine_mode=options.tesseract_oem,
tessconfig=options.tesseract_config,
timeout=options.tesseract_timeout,
pagesegmode=options.tesseract_pagesegmode,
user_words=options.user_words,
user_patterns=options.user_patterns,
tesseract_env=options.tesseract_env,
options=options,
)
return (hocr_out, hocr_text_out)
@@ -610,22 +602,17 @@ def render_hocr_page(hocr, page_context):
return output_file
def ocr_tesseract_textonly_pdf(input_image, page_context):
def ocr_engine_textonly_pdf(input_image, page_context):
output_pdf = page_context.get_path('ocr_tess.pdf')
output_text = page_context.get_path('ocr_tess.txt')
options = page_context.options
tesseract.generate_pdf(
ocr_engine = page_context.plugin_manager.hook.get_ocr_engine()
ocr_engine.generate_pdf(
input_file=input_image,
output_pdf=output_pdf,
output_text=output_text,
languages=options.language,
engine_mode=options.tesseract_oem,
tessconfig=options.tesseract_config,
timeout=options.tesseract_timeout,
pagesegmode=options.tesseract_pagesegmode,
user_words=options.user_words,
user_patterns=options.user_patterns,
tesseract_env=options.tesseract_env,
options=options,
)
return (output_pdf, output_text)
+3 -1
View File
@@ -26,9 +26,11 @@ import pluggy
from ocrmypdf import pluginspec
def get_plugin_manager(plugins: List[str]):
def get_plugin_manager(plugins: List[str], builtins=True):
pm = pluggy.PluginManager('ocrmypdf')
pm.add_hookspecs(pluginspec)
if builtins:
plugins.insert(0, 'ocrmypdf.builtin_plugins')
for name in plugins:
if name.endswith('.py'):
# Import by filename
+4 -6
View File
@@ -42,8 +42,8 @@ from ocrmypdf._pipeline import (
is_ocr_required,
merge_sidecars,
metadata_fixup,
ocr_tesseract_hocr,
ocr_tesseract_textonly_pdf,
ocr_engine_hocr,
ocr_engine_textonly_pdf,
optimize_pdf,
preprocess_clean,
preprocess_deskew,
@@ -176,13 +176,11 @@ def exec_page_sync(page_context):
)
if options.pdf_renderer == 'hocr':
(hocr_out, text_out) = ocr_tesseract_hocr(ocr_image_out, page_context)
(hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context)
ocr_out = render_hocr_page(hocr_out, page_context)
if options.pdf_renderer == 'sandwich':
(ocr_out, text_out) = ocr_tesseract_textonly_pdf(
ocr_image_out, page_context
)
(ocr_out, text_out) = ocr_engine_textonly_pdf(ocr_image_out, page_context)
return PageResult(
pageno=page_context.pageno,
+18
View File
@@ -0,0 +1,18 @@
# © 2020 James R. Barlow: github.com/jbarlow83
#
# This file is part of OCRmyPDF.
#
# OCRmyPDF is free software: you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# OCRmyPDF is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# GNU General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
from ocrmypdf.builtin_plugins.tesseract_ocr import *
@@ -0,0 +1,68 @@
# © 2020 James R. Barlow: github.com/jbarlow83
#
# This file is part of OCRmyPDF.
#
# OCRmyPDF is free software: you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# OCRmyPDF is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# GNU General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
from ocrmypdf import hookimpl
from ocrmypdf.exec import tesseract
from ocrmypdf.pluginspec import OcrEngine
class TesseractOcrEngine(OcrEngine):
def languages(self):
return tesseract.get_languages()
def get_orientation(self, input_file, options):
return tesseract.get_orientation(
input_file,
engine_mode=options.tesseract_oem,
timeout=options.tesseract_timeout,
tesseract_env=options.tesseract_env,
)
def generate_hocr(self, input_file, output_hocr, output_text, options):
tesseract.generate_hocr(
input_file=input_file,
output_hocr=output_hocr,
output_text=output_text,
languages=options.language,
engine_mode=options.tesseract_oem,
tessconfig=options.tesseract_config,
timeout=options.tesseract_timeout,
pagesegmode=options.tesseract_pagesegmode,
user_words=options.user_words,
user_patterns=options.user_patterns,
tesseract_env=options.tesseract_env,
)
def generate_pdf(self, input_file, output_pdf, output_text, options):
tesseract.generate_pdf(
input_file=input_file,
output_pdf=output_pdf,
output_text=output_text,
languages=options.language,
engine_mode=options.tesseract_oem,
tessconfig=options.tesseract_config,
timeout=options.tesseract_timeout,
pagesegmode=options.tesseract_pagesegmode,
user_words=options.user_words,
user_patterns=options.user_patterns,
tesseract_env=options.tesseract_env,
)
@hookimpl
def get_ocr_engine():
return TesseractOcrEngine()
+35 -1
View File
@@ -15,9 +15,11 @@
# You should have received a copy of the GNU General Public License
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
from abc import ABC, abstractmethod
from argparse import ArgumentParser, Namespace
from collections import namedtuple
from pathlib import Path
from typing import Optional
from typing import AbstractSet, Optional
import pluggy
from PIL import Image
@@ -87,3 +89,35 @@ def filter_page_image(page: 'PageContext', image_filename: Path) -> Path:
Note that the ocrmypdf image optimization stage may ultimately chose a
different format.
"""
OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence'))
class OcrEngine(ABC):
@abstractmethod
def languages(self) -> AbstractSet[str]:
"""Returns set of languages that are supported."""
@abstractmethod
def get_orientation(
self, input_file: Path, options: Namespace
) -> OrientationConfidence:
"""Returns the orientation of the image."""
@abstractmethod
def generate_hocr(
self, input_file: Path, output_hocr: Path, output_text: Path, options: Namespace
) -> None:
pass
@abstractmethod
def generate_pdf(
self, input_file: Path, output_pdf: Path, output_text: Path, options: Namespace
) -> None:
pass
@hookspec(firstresult=True)
def get_ocr_engine() -> OcrEngine:
pass