diff --git a/.gitignore b/.gitignore index 58549bc1..ad618018 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,17 @@ tmp/ log/ -*.pyc \ No newline at end of file +*.pyc +tests/output/ +.ruffus_history.sqlite +*.sublime-* +/*.pdf +build/ +dist/ +*.egg-info/ +venv/ +*/test/output +bin/ +include/ +lib/ +pip-selfcheck.json +pyvenv.cfg diff --git a/LICENSE.md b/LICENSE.md deleted file mode 100644 index 03fae0b8..00000000 --- a/LICENSE.md +++ /dev/null @@ -1,19 +0,0 @@ -Copyright (c) 2013 fritz-hh from Github - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in -all copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN -THE SOFTWARE. diff --git a/LICENSE.rst b/LICENSE.rst new file mode 100644 index 00000000..ec3f0f9e --- /dev/null +++ b/LICENSE.rst @@ -0,0 +1,20 @@ +Copyright (c) 2013-2015, The OCRmyPDF Authors + +Permission is hereby granted, free of charge, to any person obtaining a +copy of this software and associated documentation files (the +"Software"), to deal in the Software without restriction, including +without limitation the rights to use, copy, modify, merge, publish, +distribute, sublicense, and/or sell copies of the Software, and to +permit persons to whom the Software is furnished to do so, subject to +the following conditions: + +The above copyright notice and this permission notice shall be included +in all copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS +OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF +MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. +IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY +CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION OF CONTRACT, +TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE +SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. diff --git a/MANIFEST.in b/MANIFEST.in new file mode 100644 index 00000000..9d6a9042 --- /dev/null +++ b/MANIFEST.in @@ -0,0 +1,4 @@ +recursive-include ocrmypdf/jhove/bin *.jar +recursive-include ocrmypdf/jhove/lib *.jar +recursive-include ocrmypdf/jhove/conf *.conf +recursive-exclude tests/output * diff --git a/OCRmyPDF.sh b/OCRmyPDF.sh index 8ffa824b..d5236acd 100644 --- a/OCRmyPDF.sh +++ b/OCRmyPDF.sh @@ -3,314 +3,4 @@ # Copyright (c) 2013-14: fritz-hh from Github (https://github.com/fritz-hh) ############################################################################## -# Determine real path of this script, following symlinks if present -! command -v python2 > /dev/null && echo "Please install python v2.x. Exiting..." && exit 1 -BASEPATH="$(dirname $(python2 -c "import os; print os.path.realpath(\"$0\")"))" - -# Import required scripts -. "$BASEPATH/src/config.sh" - -# Set variables corresponding to the input parameters -ARGUMENTS="$@" - -START=`date +%s` - -usage() { - cat << EOF --------------------------------------------------------------------------------------- -Script aimed at generating a searchable PDF file from a PDF file containing only images. -(The script performs optical character recognition of each respective page using the -tesseract engine) - -Copyright: fritz-hh from Github (https://github.com/fritz-hh) -Version: $VERSION - -Usage: OCRmyPDF.sh [-h] [-v] [-g] [-k] [-d] [-c] [-i] [-o dpi] [-f|-s] [-l lan1[+lan2...]] [-C filename] inputfile outputfile - --h : Display this help message --v : Increase the verbosity (this option can be used more than once) (e.g. -vvv) --k : Do not delete the temporary files --g : Activate debug mode: - - Generates a PDF file containing each page twice (once with the image, once without the image - but with the OCRed text as well as the detected bounding boxes) - - Set the verbosity to the highest possible - - Do not delete the temporary files --d : Deskew each page before performing OCR --c : Clean each page before performing OCR --i : Incorporate the cleaned image in the final PDF file (by default the original image, or the deskewed image if the -d option is set) --o : If the resolution of an image is lower than dpi value provided as argument, provide the OCR engine with - an oversampled image having the latter dpi value. This can improve the OCR results but can lead to a larger output PDF file. - (default: no oversampling performed) --f : Force to OCR the whole document, even if some page already contain font data. - (which should not be the case for PDF files built from scnanned images) - Any text data will be rendered to raster format and then fed through OCR. --s : If pages contain font data, do not OCR that page, but include the page (as is) in the final output. --l : Language(s) of the PDF file. The language should be set correctly in order to get good OCR results. - Any language supported by tesseract is supported (Tesseract uses 3-character ISO 639-2 language codes) - Multiple languages may be specified, separated by '+' characters. - (The default language is defined in the config file) --C : Pass an additional configuration file to the tesseract OCR engine. - (this option can be used more than once) - Note 1: The configuration file must be available in the "tessdata/configs" folder of your tesseract installation -inputfile : PDF file to be OCRed -outputfile : The PDF/A file that will be generated --------------------------------------------------------------------------------------- -EOF -} - - -################################################# -# Get an absolute path from a relative path to a file -# -# Param1 : Relative path -# Returns: 1 if the folder in which the file is located does not exist -# 0 otherwise -################################################# -absolutePath() { - local wdsave absolutepath - wdsave="$(pwd)" - ! cd "$(dirname "$1")" 1> /dev/null 2> /dev/null && return 1 - absolutepath="$(pwd)/$(basename "$1")" - cd "$wdsave" - echo "$absolutepath" - return 0 -} - - -# Initialization the configuration parameters with default values -VERBOSITY="$LOG_ERR" # default verbosity level -LAN="$DEFAULT_LANGUAGES" # default language(s) of the PDF file (required to get good OCR results) -KEEP_TMP="0" # 0=no, 1=yes (keep the temporary files) -PREPROCESS_DESKEW="0" # 0=no, 1=yes (deskew image) -PREPROCESS_CLEAN="0" # 0=no, 1=yes (clean image to improve OCR) -PREPROCESS_CLEANTOPDF="0" # 0=no, 1=yes (put cleaned image in final PDF) -OVERSAMPLING_DPI="0" # 0=do not perform oversampling (dpi value under which oversampling should be performed) -PDF_NOIMG="0" # 0=no, 1=yes (generates each PDF page twice, with and without image) -FORCE_OCR="0" # 0=do not force, 1=force (force to OCR the whole document, even if some page already contain font data) -SKIP_TEXT="0" # 0=do not skip text pages, 1=skip text pages -TESS_CFG_FILES="" # list of additional configuration files to be used by tesseract - -# Parse optional command line arguments -while getopts ":hvgkdcio:fsl:C:" opt; do - case $opt in - h) usage ; exit 0 ;; - v) VERBOSITY=$(($VERBOSITY+1)) ;; - k) KEEP_TMP="1" ;; - g) PDF_NOIMG="1"; VERBOSITY="$LOG_DEBUG"; KEEP_TMP="1" ;; - d) PREPROCESS_DESKEW="1" ;; - c) PREPROCESS_CLEAN="1" ;; - i) PREPROCESS_CLEANTOPDF="1" ;; - o) OVERSAMPLING_DPI="$OPTARG" ;; - f) FORCE_OCR="1" ;; - s) SKIP_TEXT="1" ;; - l) LAN="$OPTARG" ;; - C) TESS_CFG_FILES="$OPTARG $TESS_CFG_FILES" ;; - \?) - echo "Invalid option: -$OPTARG" - usage - exit $EXIT_BAD_ARGS ;; - :) - echo "Option -$OPTARG requires an argument" - usage - exit $EXIT_BAD_ARGS ;; - esac -done - -# Remove the optional arguments parsed above. -shift $((OPTIND-1)) - -# Check if the number of mandatory parameters provided is as expected -if [ "$#" -ne "2" ]; then - echo "Exactly two mandatory argument shall be provided ($# arguments provided)" - usage - exit $EXIT_BAD_ARGS -fi - -# Ensure that -f and -s are not both set -if [ "$SKIP_TEXT" -eq "1" -a "$FORCE_OCR" -eq "1" ]; then - echo "Options -f and -s are mutually exclusive; choose one or the other" - usage - exit $EXIT_BAD_ARGS -fi - - -[ ! -f "$1" ] && echo "The input file does not exist. Exiting..." && exit $EXIT_BAD_ARGS -FILE_INPUT_PDF="`absolutePath "$1"`" - -! absolutePath "$2" >/dev/null \ - && echo "The folder in which the output file should be generated does not exist. Exiting..." && exit $EXIT_BAD_ARGS -[ -d "$2" ] && echo "Please enter the path of the file to be generated (and not a path to a folder). Exitíng..." && exit $EXIT_BAD_ARGS -[ -f "$2" ] && echo "The output file already exists. Exiting..." && exit $EXIT_BAD_ARGS -FILE_OUTPUT_PDFA="`absolutePath "$2"`" - - - -# set script path as working directory -cd "$BASEPATH" - -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "$TOOLNAME version: $VERSION" -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Arguments: $ARGUMENTS" - -# check if the required utilities are installed -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Checking if all dependencies are installed" -! command -v identify > /dev/null && echo "Please install ImageMagick. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! command -v parallel > /dev/null && echo "Please install GNU Parallel. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! command -v pdfimages > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! command -v pdftoppm > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! command -v pdffonts > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! command -v pdfseparate > /dev/null && echo "Please install or update poppler-utils to at least 0.24.5. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -[ $PREPROCESS_CLEAN -eq 1 ] && ! command -v unpaper > /dev/null && echo "Please install unpaper. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! command -v tesseract > /dev/null && echo "Please install tesseract and tesseract-data. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! python2 -c 'import lxml' 2>/dev/null && echo "Please install the python library lxml. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! python2 -c 'import sys, reportlab; (getattr(reportlab, "Version", "0.0") >= "3.0") or sys.exit(1)' 2>/dev/null \ - && echo "Please install the python library reportlab. Exiting..." && exit $EXIT_MISSING_DEPENDENCY - -! command -v gs > /dev/null && echo "Please install ghostscript. Exiting..." && exit $EXIT_MISSING_DEPENDENCY -! command -v java > /dev/null && echo "Please install java. Exiting..." && exit $EXIT_MISSING_DEPENDENCY - - -# ensure the right tesseract version is installed -# older versions are known to produce malformed hocr output and should not be used -# Even 3.02.01 fails in few cases (see issue #28). I decided to allow this version anyway because -# 3.02.02 is not yet available for some widespread linux distributions -reqtessversion="3.02.01" -tessversion=`tesseract -v 2>&1 | grep "tesseract" | sed s/[^0-9.]//g` -tesstooold=$(echo "`echo $tessversion | sed s/[.]//2`-`echo $reqtessversion | sed s/[.]//2` < 0" | bc) -[ "$tesstooold" -eq "1" ] \ - && echo "Please install tesseract ${reqtessversion} or newer (currently installed version is ${tessversion})" && exit $EXIT_MISSING_DEPENDENCY - -# ensure the right GNU parallel version is installed -# older version do not support -q flag (required to escape special characters) -reqparallelversion="20121122" -parallelversion=`parallel --minversion 0` -! parallel --minversion "$reqparallelversion" > /dev/null \ - && echo "Please install GNU parallel ${reqparallelversion} or newer (currently installed version is ${parallelversion})" && exit $EXIT_MISSING_DEPENDENCY - -# ensure pdftoppm is provided by poppler-utils, not the older xpdf version -! pdftoppm -v 2>&1 | grep -q 'Poppler' && echo "Please remove xpdf and install poppler-utils. Exiting..." && $EXIT_MISSING_DEPENDENCY - - - -# Display the version of the tools if log level is LOG_DEBUG -if [ $VERBOSITY -ge $LOG_DEBUG ]; then - echo "--------------------------------" - echo "ImageMagick version:" - identify --version - echo "--------------------------------" - echo "GNU Parallel version:" - parallel --version - echo "--------------------------------" - echo "Poppler-utils version:" - pdfimages -v - pdftoppm -v - pdffonts -v - pdfseparate -v - echo "--------------------------------" - echo "unpaper version:" - unpaper --version - echo "--------------------------------" - echo "tesseract version:" - tesseract --version - echo "--------------------------------" - echo "python2 version:" - python2 --version - echo "--------------------------------" - echo "Ghostscript version:" - gs --version - echo "--------------------------------" - echo "Java version:" - java -version - echo "--------------------------------" -fi - - - -# check if the language(s) passed to tesseract are all supported -for currentlan in `echo "$LAN" | sed 's/+/ /g'`; do - if ! tesseract --list-langs 2>&1 | grep "^$currentlan\$" > /dev/null; then - echo "The language \"$currentlan\" is not supported by tesseract." - tesseract --list-langs 2>&1 | tr '\n' ' '; echo - echo "Exiting..." - exit $EXIT_BAD_ARGS - fi -done - - - -# Initialize path to temporary files using mktemp -# Goal: save tmp file in a sub-folder of the $TMPDIR environment variable (or in "/tmp" if unset) -# Unfortunately, Linux mktemp is not compatible with FreeBSD/OSX mktemp -# Linux version requires no arg -# FreeBSD requires '-t prefix' to be used so that $TMPDIR is taken into account -# But in Linux '-t template' is handled differently than in FreeBSD -# Therefore different calls must be used for Linux and for FreeBSD -prefix="$(date +"%Y%m%d_%H%M").filename.$(basename "$FILE_INPUT_PDF" | sed 's/[.][^.]*$//')" # prefix made of date, time and pdf file name without extension -TMP_FLD=`mktemp -d 2>/dev/null || mktemp -d -t "${prefix}" 2>/dev/null` # try Linux syntax first, if it fails try FreeBSD/OSX -if [ $? -ne 0 ]; then - if [ -z "$TMPDIR" ]; then - echo "Could not create folder for temporary files. Please ensure you have sufficient right and \"/tmp\" exists" - else - echo "Could not create folder for temporary files. Please ensure you have sufficient right and \"$TMPDIR\" exists" - fi - exit $EXIT_FILE_ACCESS_ERROR -fi -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Created temporary folder: \"$TMP_FLD\"" - -FILE_TMP="${TMP_FLD}/tmp.txt" # temporary file with a very short lifetime (may be used for several things) -FILE_PAGES_INFO="${TMP_FLD}/pages-info.txt" # for each page: page #; width in pt; height in pt -FILE_VALIDATION_LOG="${TMP_FLD}/pdf_validation.log" # log file containing the results of the validation of the PDF/A file - - - -# get the size of each pdf page (width / height) in pt (i.e. inch/72) -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Input file: Extracting size of each page (in pt)" -! identify -format "%w %h\n" "$FILE_INPUT_PDF" > "$FILE_TMP" \ - && echo "Could not get size of PDF pages. Exiting..." && exit $EXIT_BAD_INPUT_FILE -# removing empty lines (last one should be) and add page # before each line -sed '/^$/d' "$FILE_TMP" | awk '{printf "%04d %s\n", NR, $0}' > "$FILE_PAGES_INFO" -numpages=`tail -n 1 "$FILE_PAGES_INFO" | cut -f1 -d" "` - -# process each page of the input pdf file -parallel --gnu -q -k --halt-on-error 1 "$OCR_PAGE" "$FILE_INPUT_PDF" "{}" "$numpages" "$TMP_FLD" \ - "$VERBOSITY" "$LAN" "$KEEP_TMP" "$PREPROCESS_DESKEW" "$PREPROCESS_CLEAN" "$PREPROCESS_CLEANTOPDF" "$OVERSAMPLING_DPI" \ - "$PDF_NOIMG" "$FORCE_OCR" "$SKIP_TEXT" "$TESS_CFG_FILES" < "$FILE_PAGES_INFO" -ret_code="$?" -[ $ret_code -ne 0 ] && exit $ret_code - -# concatenate all pages and convert the pdf file to match PDF/A format -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Concatenating all pages to the final PDF/A file" -! gs -dQUIET -dPDFA -dBATCH -dNOPAUSE -dUseCIEColor \ - -sProcessColorModel=DeviceCMYK -sDEVICE=pdfwrite -sPDFACompatibilityPolicy=2 \ - -sOutputFile="$FILE_OUTPUT_PDFA" "${TMP_FLD}/"*ocred*.pdf 1> /dev/null 2> /dev/null \ - && echo "Could not concatenate all pages to the final PDF/A file. Exiting..." && exit $EXIT_OTHER_ERROR - -# validate generated pdf file (compliance to PDF/A) -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Checking compliance to PDF/A standard" -! java -jar "$JHOVE" -c "$JHOVE_CFG" -m PDF-hul "$FILE_OUTPUT_PDFA" 2> /dev/null 1> "$FILE_VALIDATION_LOG" \ - && echo "Unexpected error while checking compliance to PDF/A file. Exiting..." && exit $EXIT_OTHER_ERROR -grep -i "Status|Message" "$FILE_VALIDATION_LOG" # summary of the validation -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "The full validation log is available here: \"$FILE_VALIDATION_LOG\"" -# check the validation results -pdf_valid=1 -grep -i 'ErrorMessage' "$FILE_VALIDATION_LOG" && pdf_valid=0 -grep -i 'Status.*not valid' "$FILE_VALIDATION_LOG" && pdf_valid=0 -grep -i 'Status.*Not well-formed' "$FILE_VALIDATION_LOG" && pdf_valid=0 -! grep -i 'Profile:.*PDF/A-1' "$FILE_VALIDATION_LOG" > /dev/null && echo "PDF file profile is not PDF/A-1" && pdf_valid=0 -[ $pdf_valid -ne 1 ] && echo "Output file: The generated PDF/A file is INVALID" -[ $pdf_valid -eq 1 ] && [ $VERBOSITY -ge $LOG_INFO ] && echo "Output file: The generated PDF/A file is VALID" - - - - -# delete temporary files -if [ $KEEP_TMP -eq 0 ]; then - [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Deleting temporary files" - rm -r -f "${TMP_FLD}" -fi - - -END=`date +%s` -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Script took $(($END-$START)) seconds" - - -[ $pdf_valid -ne 1 ] && exit $EXIT_INVALID_OUTPUT_PDFA || exit 0 +python3 -m ocrmypdf.main "$@" diff --git a/README.md b/README.md deleted file mode 100644 index 8570c67b..00000000 --- a/README.md +++ /dev/null @@ -1,69 +0,0 @@ -OCRmyPDF -======== - -OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched - -To get the script usage, call: sh ./OCRmyPDF.sh -h - -Main features --------- - -- Generates a searchable PDF/A file from a PDF file containing only images -- Places OCRed text accurately below the image to ease copy / paste -- Keeps the exact resolution of the original embedded images - - or if requested oversamples the images before OCRing so as to get better results -- If requested deskews and / or clean the image before performing OCR -- Validates the generated file against the PDF/A specification using jhove -- Provides debug mode to enable easy verification of the OCR results -- Processes several pages in parallel if more than one CPU core is available - -For details: please consult the release notes - -Motivation ----------- - -I searched the web for a free command line tool to OCR PDF files on linux/unix: -I found many, but none of them were really satisfying. -- Either they produced PDF files with misplaced text under the image (making copy/paste impossible) -- Or they did not display correctly some escaped html characters located in the hocr file produced by the OCR engine -- Or they changed the resolution of the embedded images -- Or they generated PDF file having a ridiculous big size -- Or they crashed when trying to OCR some of my PDF files -- Or they did not produce valid PDF files (even though they were readable with my current PDF reader) -- On top of that none of them produced PDF/A files (format dedicated for long time storage / archiving) - -... so I decided to develop my own tool (using various existing scripts as an inspiration) - -Install -------- - -Download OCRmyPDF here: https://github.com/fritz-hh/OCRmyPDF/releases - -Copy the file in onto your linux/unix machine and extract it. - -Run: "sh ./OCRmyPDF.sh -h" to get the script usage - -If not yet installed, the script will notify you about dependencies that need to be installed. -The script requires specific versions of the dependencies. Older version than the ones mentioned in the release notes are likely not to be compatible to OCRmyPDF. - -Support -------- - -In case you detect an issue, please: - -- Check if your issue is already known -- If no problem report exists on github, please create one here: https://github.com/fritz-hh/OCRmyPDF/issues -- Describe your problem thoroughly -- Append the console output of the script when running the debug mode (-g option) -- If possible provide your input PDF file as well as the content of the temporary folder (using a file sharing service like www.file-upload.net) - -Press & Media -------------- - -- c't 1-2014, page 59: Detailed presentation of OCRmyPDF v1.0 in the leading german IT magazine c't (http://www.heise.de/ct/inhalt/2014/1/58/) -- heise Open Source, 09/2014: Texterkennung mit OCRmyPDF (http://www.heise.de/-2356670) - -Disclaimer ----------- - -The software is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. diff --git a/README.rst b/README.rst new file mode 100644 index 00000000..26f2f5a2 --- /dev/null +++ b/README.rst @@ -0,0 +1,93 @@ +OCRmyPDF +======== + +OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to +be searched. + +Main features +------------- + +- Generates a searchable + `PDF/A `__ file from a PDF + file containing only images +- Places OCRed text accurately below the image to ease copy / paste +- Keeps the exact resolution of the original embedded images + + - or if requested oversamples the images before OCRing so as to get + better results + +- If requested deskews and/or cleans the image before performing OCR +- Validates the generated file against the PDF/A-1b specification using + `JHOVE `__ +- Provides debug mode to enable easy verification of the OCR results +- Processes several pages in parallel when more than one CPU core is + available +- Uses Tesseract OCR engine + +For details: please consult the release notes + +Motivation +---------- + +I searched the web for a free command line tool to OCR PDF files on +Linux/UNIX: I found many, but none of them were really satisfying. - +Either they produced PDF files with misplaced text under the image +(making copy/paste impossible) - Or they did not display correctly some +escaped HTML characters located in the hocr file produced by the OCR +engine - Or they changed the resolution of the embedded images - Or they +generated PDF file having a ridiculous big size - Or they crashed when +trying to OCR some of my PDF files - Or they did not produce valid PDF +files (even though they were readable with my current PDF reader) - On +top of that none of them produced PDF/A files (format dedicated for long +time storage / archiving) + +... so I decided to develop my own tool (using various existing scripts +as an inspiration) + +Install +------- + +Download OCRmyPDF here: https://github.com/fritz-hh/OCRmyPDF/releases + +To install, extract the release files and run:: + + pip install . + +Run:: + + ocrmypdf --help + +If not yet installed, the script will notify you about dependencies that +need to be installed. The script requires specific versions of the +dependencies. Older version than the ones mentioned in the release notes +are likely not to be compatible to OCRmyPDF. + +Support +------- + +In case you detect an issue, please: + +- Check if your issue is already known +- If no problem report exists on github, please create one here: + https://github.com/fritz-hh/OCRmyPDF/issues +- Describe your problem thoroughly +- Append the console output of the script when running the debug mode + (-v 1 option) +- If possible provide your input PDF file as well as the content of the + temporary folder (using a file sharing service like + www.file-upload.net) + +Press & Media +------------- + +- `c't 1-2014, page 59 `__: + Detailed presentation of OCRmyPDF v1.0 in the leading German IT + magazine c't +- `heise Open Source, 09/2014: Texterkennung mit + OCRmyPDF `__ + +Disclaimer +---------- + +The software is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR +CONDITIONS OF ANY KIND, either express or implied. diff --git a/RELEASE_NOTES.md b/RELEASE_NOTES.md deleted file mode 100644 index 83d0a376..00000000 --- a/RELEASE_NOTES.md +++ /dev/null @@ -1,335 +0,0 @@ -RELEASE NOTES -============= - -Please always read this file before installing the package - -Download software here: https://github.com/fritz-hh/OCRmyPDF/releases - -v2.2-stable (2014-09-29): -======= - -New features ------------- - -- None - -Changes -------- - -- Update to jhove v1.11 -- Request the python library reportlab v3.0 or newer (So that we could remove a patch to the previous version of reportlab leading to issues for some users) - -Fixes ------ - -- Fix bug on Mac OS X (resolution of simlink to OCRmyPDF.sh script) (thanks to jbarlow83) -- Check if the input pdf file exists before to continue - -Tested with ------------ - -- Operating system: FreeBSD 9.2 -- Dependencies: - - parallel 20140822 - - poppler-utils 0.24.5 - - ImageMagick 6.8.9-4 2014-09-17 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.8 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_65" - -v2.1-stable (2014-09-20): -======= - -New features ------------- - -- None - -Changes -------- - -- None - -Fixes ------ - -- Allow execution via simlink -- Add support for tesseract 3.03 -- Add support for newer version of reportlab -- Lowered minimum version of gnu parallel -- Various typo - -Tested with ------------ - -- Operating system: FreeBSD 9.1 -- Dependencies: - - parallel 20130222 - - poppler-utils 0.22.2 - - ImageMagick 6.8.0-7 2013-03-30 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.3 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_17" - -v2.0-stable (2014-01-25): -======= - -New features ------------- - -- Check if the language(s) passed using the -l option is supported by tesseract (fixes #60) - -Changes -------- - -- Allow OCRmyPDF to be used with tesseract 3.02.01, even though OCR might fail for few PDF file (see issue #28). Rationale: For some linux distribution, no newer version than tesseract 3.02.01 is available - -Fixes ------ - -- More robust algorithm for checking the version of the installed tesseract package - -Tested with ------------ - -- Operating system: FreeBSD 9.1 -- Dependencies: - - parallel 20130222 - - poppler-utils 0.22.2 - - ImageMagick 6.8.0-7 2013-03-30 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.3 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_17" - -v2.0-rc2 (2014-01-16): -======= - -New features ------------- - -- None - -Changes -------- - -- Size reduction of final PDF file: (fixes #50) - - Support for monochrome (Black&White) images (massive size reduction in final PDF: >80%) - - Reduced size of grayscale images (by 13% on test PDF file) -- Preventing fi, fl ligatures does not require anymore to pass an additional config file to tesseract using the -C option (fixes #58) -- Location of temporary folder according to content of environment variable TMPDIR. -- Dependency to pdftk removed -- Check for compatible versions of dependencies: (fixes #51) - - parallel and tesseract - - python libraries reportlab and lxml - -Fixes ------ - -- Improved portability with various shells (dash, bash, tcsh) and OS (FreeBSD, MAC OSX, Linux) (fixes #59) -- Corrected bug in case the input PDF file contains a space character (fixes #48) -- Prevent spurious error message in case there is no image in a PDF page -- Prevent collision of temporary folder names (fixes #57) - -Tested with ------------ - -- Operating system: FreeBSD 9.1 -- Dependencies: - - parallel 20130222 - - poppler-utils 0.22.2 - - ImageMagick 6.8.0-7 2013-03-30 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.3 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_17" - -v2.0-rc1 (2014-01-07): -==== - -New features ------------- - -- Huge performance improvement on machines having multiple CPU/cores (processing of several pages concurrently) (fixes #18) -- By default prevent from processing a PDF file already containing fonts (i.e. text)(it can be overridden with the -f flag) (fixes #16) -- Warn if the resolution is too low to get reasonable OCR results (fixes #37) -- New option (-o) to perform automatic oversampling if the image resolution is too low. This can improve OCR results. -- Warn if using a tesseract version older than v3.02.02 (as older versions are known to produce invalid output) (fixes #41) -- Echo version of the installed dependencies (e.g. tesseract) in debug mode in order to ease support (fixes #35) -- Echo the arguments passed to the script in debug mode to ease support - -Changes -------- - -- In debug mode: The debug page is now placed after the respective "normal" page -- Reduced disk space usage in temporary folder if -d (deskew) or -c (cleanup) options are not selected -- New file src/config.sh containing various configuration parameters -- Documentation of the tesseract config file "tess-cfg/no_ligature" improved -- Improved consistency of the temporary file names - -Fixes ------ - -- Improved robustness: - - in case vertical resolution differs from horizontal resolution (fixes #38) - - in case a PDF page contains more than one image (fixes #36) -- Fix a problem occurring if python 3 is the standard interpreter (fixes #33) -- Fix a problem occurring if the input PDF file contains special characters like "#" (fixes #34) - -Tested with ------------ - -- Operating system: FreeBSD 9.1 -- Dependencies: - - parallel 20130222 - - poppler-utils 0.22.2 - - ImageMagick 6.8.0-7 2013-03-30 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.3 - - pdftk 1.45 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_17" - -v1.1-stable (2014-01-06): -==== - -New features ------------- - -- N/A - -Changes -------- - -- N/A - -Fixes ------ - -- Fixed syntax error (bashism) leading to an error message on certain systems (fixes #42) - -Tested with ------------ - -- Operating system: FreeBSD 9.1 -- Dependencies: - - poppler-utils 0.22.2 - - ImageMagick 6.8.0-7 2013-03-30 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.3 - - pdftk 1.45 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_17" - -v1.0-stable (2013-05-06): -==== - -New features ------------- - -- In debug mode: compute and echo time required for processing (fixes #26) - -Changes -------- - -- Removed feature to add metadata in final pdf file (because it lead to to final PDF file that does not comply to the PDF/A-1 format) -- Removed feature to set same owner & permissions in final PDF file than in input file -- Removed many unused jhove files (e.g. documentation, *.java and *.class files) - -Fixes ------ - -- Correction to handle correctly path and input PDF files having spaces (fixes #31) -- Resolutions (x/y) that are nearly equal are now supported (fixes #25) -- Fix compatibility issue with Ubuntu server 12.04 / Ubuntu server 10.04 / Linux Mint 13 Maya and probably other Linux distributions (fixes #27) -- Commit missing jhove files (*.jar mainly) due to wrong .gitignore - -Tested with ------------ - -- Operating system: FreeBSD 9.1 -- Dependencies: - - poppler-utils 0.22.2 - - ImageMagick 6.8.0-7 2013-03-30 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.3 - - pdftk 1.45 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_17" - -v1.0-rc2 (2013-04-29): -==== - -New features ------------- - -- Keep temporary files if debug mode is set (fixes #22) -- Set same owner & permissions in final PDF file than in input file (fixes #9) -- Added metadata in final pdf file (fixes #4) - -Changes -------- - -- N/A - -Fixes ------ - -- Fixed wrong image cropping when deskew option is activated -- Exit with error message if page size is not found in hocr file (fixes #21) -- Various minor fixes in log messages - -Tested with ------------ - -- Operating system: FreeBSD 9.1 -- Dependencies: - - poppler-utils 0.22.2 - - ImageMagick 6.8.0-7 2013-03-30 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.3 - - pdftk 1.45 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_17" - -v1.0-rc1 (2013-04-26): -==== - -New features ------------- - -- First release candidate - -Changes -------- - -- N/A - -Fixes ------ - -- N/A - -Tested with ------------ - -- Operating system: FreeBSD 9.1 -- Dependencies: - - poppler-utils 0.22.2 - - ImageMagick 6.8.0-7 2013-03-30 - - Unpaper 0.3 - - tesseract 3.02.02 - - Python 2.7.3 - - pdftk 1.45 - - ghostcript (gs): 9.06 - - java: openjdk version "1.7.0_17" diff --git a/RELEASE_NOTES.rst b/RELEASE_NOTES.rst new file mode 100644 index 00000000..7a53c3de --- /dev/null +++ b/RELEASE_NOTES.rst @@ -0,0 +1,452 @@ +RELEASE NOTES +============= + +Please always read this file before installing the package + +Download software here: https://github.com/fritz-hh/OCRmyPDF/tags + +v3.0-rc2: +========= + +New features +------------ + +- Easier installation with Python's package manager +- Now installs ``ocrmypdf`` to ``/usr/local/bin`` or equivalent for system-wide + access +- Tesseract 3.03 PDF page can be used instead for better positioning + of recognized text (``--pdf-renderer tesseract``) +- Improved command line syntax and usage help (``--help``) +- PDF metadata (title, author, keywords) are now transferred to the + output PDF +- PDF metadata can also be set from the command line (``--title``, etc.) +- Added test cases to confirm everything is working +- Added option to skip extremely large pages that take too long to OCR and are + often not OCRable (e.g. large scanned maps or diagrams); other pages are still + processed (``--skip-big``) +- Added option to kill Tesseract OCR process if it seems to be taking too long on + a page, while still processing other pages (``--tesseract-timeout``) + +Changes +------- + +- New, robust rewrite in Python 3.4+ with ruffus_ pipelines +- Now uses Ghostscript 9.14's improved color conversion model +- All "tasks" in the pipeline can be executed in parallel on any + available CPUs, increasing performance +- The ``-o DPI`` argument has been phased out, in favor of ``--oversample DPI`` +- Removed several dependencies, so it's easier to install. We no + longer use: + + - GNU parallel_ + - ImageMagick_ + - Python 2.7 + - shell scripts + +- Some new external dependencies are required: + + - MuPDF_ tools + - Ghostscript 9.14+ + - Unpaper_ 6.1 (optional) + - some automatically managed Python dependencies + +.. _ruffus: http://www.ruffus.org.uk/index.html +.. _parallel: https://www.gnu.org/software/parallel/ +.. _ImageMagick: http://www.imagemagick.org/script/index.php +.. _MuPDF: http://mupdf.com/docs/ +.. _Unpaper: https://github.com/Flameeyes/unpaper + +Compatibility notes +------------------- + +- ``./OCRmyPDF.sh`` script is still available for now +- Stacking the verbosity option like ``-vvv`` is no longer supported + +- The configuration file ``config.sh`` has been removed. Instead, you can + feed a file to the arguments for common settings: + +:: + + ocrmypdf input.pdf output.pdf @settings.txt + +where ``settings.txt`` contains, for example: + +:: + + -l deu --author 'A. Merkel' --pdf-renderer tesseract + + +Fixes +----- + +- Handling of filenames containing spaces: fixed + +Notes +----- + +- Some dependencies may work with lower versions than tested, so try + overriding dependencies if they are "in the way" to see if they work. + + +v2.2-stable (2014-09-29): +======= + +New features +------------ + +- None + +Changes +------- + +- Update to jhove v1.11 +- Request the python library reportlab v3.0 or newer (So that we could remove a patch to the previous version of reportlab leading to issues for some users) + +Fixes +----- + +- Fix bug on Mac OS X (resolution of simlink to OCRmyPDF.sh script) (thanks to jbarlow83) +- Check if the input pdf file exists before to continue + +Tested with +----------- + +- Operating system: FreeBSD 9.2 +- Dependencies: + - parallel 20140822 + - poppler-utils 0.24.5 + - ImageMagick 6.8.9-4 2014-09-17 + - Unpaper 0.3 + - tesseract 3.02.02 + - Python 2.7.8 + - ghostcript (gs): 9.06 + - java: openjdk version "1.7.0_65" + + +v2.1-stable (2014-09-20): +========================= + +New features +------------ + +- None + +Changes +------- + +- None + +Fixes +----- + +- Allow execution via simlink +- Add support for tesseract 3.03 +- Add support for newer version of reportlab +- Lowered minimum version of gnu parallel +- Various typo + +Tested with +----------- + +- Operating system: FreeBSD 9.1 +- Dependencies: +- parallel 20130222 +- poppler-utils 0.22.2 +- ImageMagick 6.8.0-7 2013-03-30 +- Unpaper 0.3 +- tesseract 3.02.02 +- Python 2.7.3 +- ghoscript (gs): 9.06 +- java: openjdk version "1.7.0\_17" + +v2.0-stable (2014-01-25): +========================= + +New features +------------ + +- Check if the language(s) passed using the -l option is supported by + tesseract (fixes #60) + +Changes +------- + +- Allow OCRmyPDF to be used with tesseract 3.02.01, even though OCR + might fail for few PDF file (see issue #28). Rationale: For some + linux distribution, no newer version than tesseract 3.02.01 is + available + +Fixes +----- + +- More robust algorithm for checking the version of the installed + tesseract package + +Tested with +----------- + +- Operating system: FreeBSD 9.1 +- Dependencies: +- parallel 20130222 +- poppler-utils 0.22.2 +- ImageMagick 6.8.0-7 2013-03-30 +- Unpaper 0.3 +- tesseract 3.02.02 +- Python 2.7.3 +- ghoscript (gs): 9.06 +- java: openjdk version "1.7.0\_17" + +v2.0-rc2 (2014-01-16): +====================== + +New features +------------ + +- None + +Changes +------- + +- Size reduction of final PDF file: (fixes #50) +- Support for monochrome (Black&White) images (massive size reduction + in final PDF: >80%) +- Reduced size of grayscale images (by 13% on test PDF file) +- Preventing fi, fl ligatures does not require anymore to pass an + additional config file to tesseract using the -C option (fixes #58) +- Location of temporary folder according to content of environment + variable TMPDIR. +- Dependency to pdftk removed +- Check for compatible versions of dependencies: (fixes #51) +- parallel and tesseract +- python libraries reportlab and lxml + +Fixes +----- + +- Improved portability with various shells (dash, bash, tcsh) and OS + (FreeBSD, MAC OSX, Linux) (fixes #59) +- Corrected bug in case the input PDF file contains a space character + (fixes #48) +- Prevent spurious error message in case there is no image in a PDF + page +- Prevent collision of temporary folder names (fixes #57) + +Tested with +----------- + +- Operating system: FreeBSD 9.1 +- Dependencies: +- parallel 20130222 +- poppler-utils 0.22.2 +- ImageMagick 6.8.0-7 2013-03-30 +- Unpaper 0.3 +- tesseract 3.02.02 +- Python 2.7.3 +- ghoscript (gs): 9.06 +- java: openjdk version "1.7.0\_17" + +v2.0-rc1 (2014-01-07): +====================== + +New features +------------ + +- Huge performance improvement on machines having multiple CPU/cores + (processing of several pages concurrently) (fixes #18) +- By default prevent from processing a PDF file already containing + fonts (i.e. text)(it can be overridden with the -f flag) (fixes #16) +- Warn if the resolution is too low to get reasonable OCR results + (fixes #37) +- New option (-o) to perform automatic oversampling if the image + resolution is too low. This can improve OCR results. +- Warn if using a tesseract version older than v3.02.02 (as older + versions are known to produce invalid output) (fixes #41) +- Echo version of the installed dependencies (e.g. tesseract) in debug + mode in order to ease support (fixes #35) +- Echo the arguments passed to the script in debug mode to ease support + +Changes +------- + +- In debug mode: The debug page is now placed after the respective + "normal" page +- Reduced disk space usage in temporary folder if -d (deskew) or -c + (cleanup) options are not selected +- New file src/config.sh containing various configuration parameters +- Documentation of the tesseract config file "tess-cfg/no\_ligature" + improved +- Improved consistency of the temporary file names + +Fixes +----- + +- Improved robustness: +- in case vertical resolution differs from horizontal resolution (fixes + #38) +- in case a PDF page contains more than one image (fixes #36) +- Fix a problem occurring if python 3 is the standard interpreter + (fixes #33) +- Fix a problem occurring if the input PDF file contains special + characters like "#" (fixes #34) + +Tested with +----------- + +- Operating system: FreeBSD 9.1 +- Dependencies: +- parallel 20130222 +- poppler-utils 0.22.2 +- ImageMagick 6.8.0-7 2013-03-30 +- Unpaper 0.3 +- tesseract 3.02.02 +- Python 2.7.3 +- pdftk 1.45 +- ghoscript (gs): 9.06 +- java: openjdk version "1.7.0\_17" + +v1.1-stable (2014-01-06): +========================= + +New features +------------ + +- N/A + +Changes +------- + +- N/A + +Fixes +----- + +- Fixed syntax error (bashism) leading to an error message on certain + systems (fixes #42) + +Tested with +----------- + +- Operating system: FreeBSD 9.1 +- Dependencies: +- poppler-utils 0.22.2 +- ImageMagick 6.8.0-7 2013-03-30 +- Unpaper 0.3 +- tesseract 3.02.02 +- Python 2.7.3 +- pdftk 1.45 +- ghoscript (gs): 9.06 +- java: openjdk version "1.7.0\_17" + +v1.0-stable (2013-05-06): +========================= + +New features +------------ + +- In debug mode: compute and echo time required for processing (fixes + #26) + +Changes +------- + +- Removed feature to add metadata in final pdf file (because it lead to + to final PDF file that does not comply to the PDF/A-1 format) +- Removed feature to set same owner & permissions in final PDF file + than in input file +- Removed many unused jhove files (e.g. documentation, \*.java and + \*.class files) + +Fixes +----- + +- Correction to handle correctly path and input PDF files having spaces + (fixes #31) +- Resolutions (x/y) that are nearly equal are now supported (fixes #25) +- Fix compatibility issue with Ubuntu server 12.04 / Ubuntu server + 10.04 / Linux Mint 13 Maya and probably other Linux distributions + (fixes #27) +- Commit missing jhove files (\*.jar mainly) due to wrong .gitignore + +Tested with +----------- + +- Operating system: FreeBSD 9.1 +- Dependencies: +- poppler-utils 0.22.2 +- ImageMagick 6.8.0-7 2013-03-30 +- Unpaper 0.3 +- tesseract 3.02.02 +- Python 2.7.3 +- pdftk 1.45 +- ghoscript (gs): 9.06 +- java: openjdk version "1.7.0\_17" + +v1.0-rc2 (2013-04-29): +====================== + +New features +------------ + +- Keep temporary files if debug mode is set (fixes #22) +- Set same owner & permissions in final PDF file than in input file + (fixes #9) +- Added metadata in final pdf file (fixes #4) + +Changes +------- + +- N/A + +Fixes +----- + +- Fixed wrong image cropping when deskew option is activated +- Exit with error message if page size is not found in hocr file (fixes + #21) +- Various minor fixes in log messages + +Tested with +----------- + +- Operating system: FreeBSD 9.1 +- Dependencies: +- poppler-utils 0.22.2 +- ImageMagick 6.8.0-7 2013-03-30 +- Unpaper 0.3 +- tesseract 3.02.02 +- Python 2.7.3 +- pdftk 1.45 +- ghoscript (gs): 9.06 +- java: openjdk version "1.7.0\_17" + +v1.0-rc1 (2013-04-26): +====================== + +New features +------------ + +- First release candidate + +Changes +------- + +- N/A + +Fixes +----- + +- N/A + +Tested with +----------- + +- Operating system: FreeBSD 9.1 +- Dependencies: +- poppler-utils 0.22.2 +- ImageMagick 6.8.0-7 2013-03-30 +- Unpaper 0.3 +- tesseract 3.02.02 +- Python 2.7.3 +- pdftk 1.45 +- ghoscript (gs): 9.06 +- java: openjdk version "1.7.0\_17" diff --git a/ocrmypdf/__init__.py b/ocrmypdf/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/ocrmypdf/ghostscript.py b/ocrmypdf/ghostscript.py new file mode 100644 index 00000000..3d80ef0b --- /dev/null +++ b/ocrmypdf/ghostscript.py @@ -0,0 +1,51 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 + +from tempfile import NamedTemporaryFile +from subprocess import Popen, PIPE, check_call +from shutil import copy + + +def rasterize_pdf(input_file, output_file, xres, yres, raster_device, log): + with NamedTemporaryFile(delete=True) as tmp: + args_gs = [ + 'gs', + '-dBATCH', '-dNOPAUSE', + '-sDEVICE=%s' % raster_device, + '-o', tmp.name, + '-r{0}x{1}'.format(str(xres), str(yres)), + input_file + ] + + p = Popen(args_gs, close_fds=True, stdout=PIPE, stderr=PIPE, + universal_newlines=True) + stdout, stderr = p.communicate() + if stdout: + log.debug(stdout) + if stderr: + log.error(stderr) + + if p.returncode == 0: + copy(tmp.name, output_file) + else: + log.error('Ghostscript rendering failed') + + +def generate_pdfa(pdf_pages, output_file): + with NamedTemporaryFile(delete=True) as gs_pdf: + args_gs = [ + "gs", + "-dQUIET", + "-dBATCH", + "-dNOPAUSE", + "-sDEVICE=pdfwrite", + "-sColorConversionStrategy=/RGB", + "-sProcessColorModel=DeviceRGB", + "-dPDFA", + "-sPDFACompatibilityPolicy=2", + "-sOutputICCProfile=srgb.icc", + "-sOutputFile=" + gs_pdf.name, + ] + args_gs.extend(pdf_pages) + check_call(args_gs) + copy(gs_pdf.name, output_file) diff --git a/ocrmypdf/hocrtransform.py b/ocrmypdf/hocrtransform.py new file mode 100755 index 00000000..893c728b --- /dev/null +++ b/ocrmypdf/hocrtransform.py @@ -0,0 +1,231 @@ +#!/usr/local/bin/python3 +############################################################################## +# Copyright (c) 2013-14: fritz-hh from Github +# (https://github.com/fritz-hh) +# +# Copyright (c) 2010: Jonathan Brinley from Github +# (https://github.com/jbrinley/HocrConverter) +# Initial version by Jonathan Brinley, jonathanbrinley@gmail.com +############################################################################## +from reportlab.pdfgen.canvas import Canvas +from reportlab.lib.units import inch +from lxml import etree as ElementTree +from PIL import Image +from collections import namedtuple +import re +import argparse + + +Rect = namedtuple('Rect', ['x1', 'y1', 'x2', 'y2']) + + +class HocrTransformError(Exception): + pass + + +class HocrTransform(): + + """ + A class for converting documents from the hOCR format. + For details of the hOCR format, see: + http://docs.google.com/View?docid=dfxcv4vc_67g844kf + """ + + def __init__(self, hocrFileName, dpi): + self.dpi = dpi + self.boxPattern = re.compile(r'bbox((\s+\d+){4})') + + self.hocr = ElementTree.ElementTree() + self.hocr.parse(hocrFileName) + + # if the hOCR file has a namespace, ElementTree requires its use to + # find elements + matches = re.match(r'({.*})html', self.hocr.getroot().tag) + self.xmlns = '' + if matches: + self.xmlns = matches.group(1) + + # get dimension in pt (not pixel!!!!) of the OCRed image + self.width, self.height = None, None + for div in self.hocr.findall( + ".//%sdiv[@class='ocr_page']" % (self.xmlns)): + coords = self.element_coordinates(div) + pt_coords = self.pt_from_pixel(coords) + self.width = pt_coords.x2 - pt_coords.x1 + self.height = pt_coords.y2 - pt_coords.y1 + # there shouldn't be more than one, and if there is, we don't want + # it + break + if self.width is None or self.height is None: + raise HocrTransformError("hocr file is missing page dimensions") + + def __str__(self): + """ + Return the textual content of the HTML body + """ + if self.hocr is None: + return '' + body = self.hocr.find(".//%sbody" % (self.xmlns)) + if body: + return self._get_element_text(body) + else: + return '' + + def _get_element_text(self, element): + """ + Return the textual content of the element and its children + """ + text = '' + if element.text is not None: + text += element.text + for child in element.getchildren(): + text += self._get_element_text(child) + if element.tail is not None: + text += element.tail + return text + + def element_coordinates(self, element): + """ + Returns a tuple containing the coordinates of the bounding box around + an element + """ + out = (0, 0, 0, 0) + if 'title' in element.attrib: + matches = self.boxPattern.search(element.attrib['title']) + if matches: + coords = matches.group(1).split() + out = Rect._make(int(coords[n]) for n in range(4)) + return out + + def pt_from_pixel(self, pxl): + """ + Returns the quantity in PDF units (pt) given quantity in pixels + """ + return Rect._make( + (c / self.dpi * inch) for c in pxl) + + def replace_unsupported_chars(self, s): + """ + Given an input string, returns the corresponding string that: + - is available in the helvetica facetype + - does not contain any ligature (to allow easy search in the PDF file) + """ + # The 'u' before the character to replace indicates that it is a + # unicode character + s = s.replace(u"fl", "fl") + s = s.replace(u"fi", "fi") + return s + + def to_pdf(self, outFileName, imageFileName=None, showBoundingboxes=False, + fontname="Helvetica", invisibleText=False): + """ + Creates a PDF file with an image superimposed on top of the text. + Text is positioned according to the bounding box of the lines in + the hOCR file. + The image need not be identical to the image used to create the hOCR + file. + It can have a lower resolution, different color mode, etc. + """ + # create the PDF file + # page size in points (1/72 in.) + pdf = Canvas( + outFileName, pagesize=(self.width, self.height), pageCompression=1) + + # draw bounding box for each paragraph + # light blue for bounding box of paragraph + pdf.setStrokeColorRGB(0, 1, 1) + # light blue for bounding box of paragraph + pdf.setFillColorRGB(0, 1, 1) + pdf.setLineWidth(0) # no line for bounding box + for elem in self.hocr.findall( + ".//%sp[@class='%s']" % (self.xmlns, "ocr_par")): + + elemtxt = self._get_element_text(elem).rstrip() + if len(elemtxt) == 0: + continue + + pxl_coords = self.element_coordinates(elem) + pt = self.pt_from_pixel(pxl_coords) + + # draw the bbox border + if showBoundingboxes: + pdf.rect( + pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, + fill=1) + + # check if element with class 'ocrx_word' are available + # otherwise use 'ocr_line' as fallback + elemclass = "ocr_line" + if self.hocr.find( + ".//%sspan[@class='ocrx_word']" % (self.xmlns)) is not None: + elemclass = "ocrx_word" + + # itterate all text elements + # light green for bounding box of word/line + pdf.setStrokeColorRGB(1, 0, 0) + pdf.setLineWidth(0.5) # bounding box line width + pdf.setDash(6, 3) # bounding box is dashed + pdf.setFillColorRGB(0, 0, 0) # text in black + for elem in self.hocr.findall( + ".//%sspan[@class='%s']" % (self.xmlns, elemclass)): + + elemtxt = self._get_element_text(elem).rstrip() + + elemtxt = self.replace_unsupported_chars(elemtxt) + + if len(elemtxt) == 0: + continue + + pxl_coords = self.element_coordinates(elem) + pt = self.pt_from_pixel(pxl_coords) + + # draw the bbox border + if showBoundingboxes: + pdf.rect( + pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, + fill=0) + + text = pdf.beginText() + fontsize = pt.y2 - pt.y1 + text.setFont(fontname, fontsize) + if invisibleText: + text.setTextRenderMode(3) # Invisible (indicates OCR text) + + # set cursor to bottom left corner of bbox (adjust for dpi) + text.setTextOrigin(pt.x1, self.height - pt.y2) + + # scale the width of the text to fill the width of the bbox + text.setHorizScale( + 100 * (pt.x2 - pt.x1) / pdf.stringWidth( + elemtxt, fontname, fontsize)) + + # write the text to the page + text.textLine(elemtxt) + pdf.drawText(text) + + # put the image on the page, scaled to fill the page + if imageFileName is not None: + pdf.drawImage(imageFileName, 0, 0, + width=self.width, height=self.height) + + # finish up the page and save it + pdf.showPage() + pdf.save() + + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description='Convert hocr file to PDF') + parser.add_argument('-b', '--boundingboxes', action="store_true", + default=False, help='Show bounding boxes borders') + parser.add_argument('-r', '--resolution', type=int, + default=300, + help='Resolution of the image that was OCRed') + parser.add_argument('-i', '--image', default=None, + help='Path to the image to be placed above the text') + parser.add_argument('hocrfile', help='Path to the hocr file to be parsed') + parser.add_argument( + 'outputfile', help='Path to the PDF file to be generated') + args = parser.parse_args() + + hocr = HocrTransform(args.hocrfile, args.resolution) + hocr.to_pdf(args.outputfile, args.image, args.boundingboxes) diff --git a/jhove/COPYING b/ocrmypdf/jhove/COPYING similarity index 100% rename from jhove/COPYING rename to ocrmypdf/jhove/COPYING diff --git a/jhove/LICENSE b/ocrmypdf/jhove/LICENSE similarity index 100% rename from jhove/LICENSE rename to ocrmypdf/jhove/LICENSE diff --git a/jhove/README b/ocrmypdf/jhove/README similarity index 100% rename from jhove/README rename to ocrmypdf/jhove/README diff --git a/jhove/RELEASENOTES b/ocrmypdf/jhove/RELEASENOTES similarity index 100% rename from jhove/RELEASENOTES rename to ocrmypdf/jhove/RELEASENOTES diff --git a/jhove/bin/JhoveApp.jar b/ocrmypdf/jhove/bin/JhoveApp.jar similarity index 100% rename from jhove/bin/JhoveApp.jar rename to ocrmypdf/jhove/bin/JhoveApp.jar diff --git a/jhove/bin/JhoveView.jar b/ocrmypdf/jhove/bin/JhoveView.jar similarity index 100% rename from jhove/bin/JhoveView.jar rename to ocrmypdf/jhove/bin/JhoveView.jar diff --git a/jhove/bin/README b/ocrmypdf/jhove/bin/README similarity index 100% rename from jhove/bin/README rename to ocrmypdf/jhove/bin/README diff --git a/jhove/bin/jhove-handler.jar b/ocrmypdf/jhove/bin/jhove-handler.jar similarity index 100% rename from jhove/bin/jhove-handler.jar rename to ocrmypdf/jhove/bin/jhove-handler.jar diff --git a/jhove/bin/jhove-module.jar b/ocrmypdf/jhove/bin/jhove-module.jar similarity index 100% rename from jhove/bin/jhove-module.jar rename to ocrmypdf/jhove/bin/jhove-module.jar diff --git a/jhove/bin/jhove.jar b/ocrmypdf/jhove/bin/jhove.jar similarity index 100% rename from jhove/bin/jhove.jar rename to ocrmypdf/jhove/bin/jhove.jar diff --git a/jhove/build.xml b/ocrmypdf/jhove/build.xml similarity index 100% rename from jhove/build.xml rename to ocrmypdf/jhove/build.xml diff --git a/jhove/conf/README b/ocrmypdf/jhove/conf/README similarity index 100% rename from jhove/conf/README rename to ocrmypdf/jhove/conf/README diff --git a/jhove/conf/jhove-byteoffset=true.conf b/ocrmypdf/jhove/conf/jhove-byteoffset=true.conf similarity index 100% rename from jhove/conf/jhove-byteoffset=true.conf rename to ocrmypdf/jhove/conf/jhove-byteoffset=true.conf diff --git a/jhove/conf/jhove-withTextMD.conf b/ocrmypdf/jhove/conf/jhove-withTextMD.conf similarity index 100% rename from jhove/conf/jhove-withTextMD.conf rename to ocrmypdf/jhove/conf/jhove-withTextMD.conf diff --git a/jhove/conf/jhove.conf b/ocrmypdf/jhove/conf/jhove.conf similarity index 100% rename from jhove/conf/jhove.conf rename to ocrmypdf/jhove/conf/jhove.conf diff --git a/jhove/configure.pl b/ocrmypdf/jhove/configure.pl similarity index 100% rename from jhove/configure.pl rename to ocrmypdf/jhove/configure.pl diff --git a/jhove/gdump b/ocrmypdf/jhove/gdump similarity index 100% rename from jhove/gdump rename to ocrmypdf/jhove/gdump diff --git a/jhove/j2dump b/ocrmypdf/jhove/j2dump similarity index 100% rename from jhove/j2dump rename to ocrmypdf/jhove/j2dump diff --git a/jhove/jdump b/ocrmypdf/jhove/jdump similarity index 100% rename from jhove/jdump rename to ocrmypdf/jhove/jdump diff --git a/jhove/jhove b/ocrmypdf/jhove/jhove similarity index 100% rename from jhove/jhove rename to ocrmypdf/jhove/jhove diff --git a/jhove/jhove.tmpl b/ocrmypdf/jhove/jhove.tmpl similarity index 100% rename from jhove/jhove.tmpl rename to ocrmypdf/jhove/jhove.tmpl diff --git a/jhove/jhove_bat.tmpl b/ocrmypdf/jhove/jhove_bat.tmpl similarity index 100% rename from jhove/jhove_bat.tmpl rename to ocrmypdf/jhove/jhove_bat.tmpl diff --git a/jhove/lib/OdfModule.jar b/ocrmypdf/jhove/lib/OdfModule.jar similarity index 100% rename from jhove/lib/OdfModule.jar rename to ocrmypdf/jhove/lib/OdfModule.jar diff --git a/jhove/lib/PngModule.jar b/ocrmypdf/jhove/lib/PngModule.jar similarity index 100% rename from jhove/lib/PngModule.jar rename to ocrmypdf/jhove/lib/PngModule.jar diff --git a/jhove/md5.pl b/ocrmypdf/jhove/md5.pl similarity index 100% rename from jhove/md5.pl rename to ocrmypdf/jhove/md5.pl diff --git a/jhove/packagejhove.sh b/ocrmypdf/jhove/packagejhove.sh similarity index 100% rename from jhove/packagejhove.sh rename to ocrmypdf/jhove/packagejhove.sh diff --git a/jhove/pdump b/ocrmypdf/jhove/pdump similarity index 100% rename from jhove/pdump rename to ocrmypdf/jhove/pdump diff --git a/jhove/userhome b/ocrmypdf/jhove/userhome similarity index 100% rename from jhove/userhome rename to ocrmypdf/jhove/userhome diff --git a/ocrmypdf/leptonica.py b/ocrmypdf/leptonica.py new file mode 100644 index 00000000..a03a1af5 --- /dev/null +++ b/ocrmypdf/leptonica.py @@ -0,0 +1,331 @@ +#!/usr/bin/env python2 +# -*- coding: utf-8 -*- +# +# © 2013-15: jbarlow83 from Github (https://github.com/jbarlow83) +# +# +# Use Leptonica to detect find and remove page skew. Leptonica uses the method +# of differential square sums, which its author claim is faster and more robust +# than the Hough transform used by ImageMagick. + +from __future__ import print_function, absolute_import, division +import argparse +import ctypes as C +import sys +import os +import logging +from tempfile import TemporaryFile + +logger = logging.getLogger(__name__) + + +def stderr(*objs): + """Python 2/3 compatible print to stderr. + """ + print("leptonica.py:", *objs, file=sys.stderr) + + +from ctypes.util import find_library +lept_lib = find_library('lept') +if not lept_lib: + stderr("Could not find the Leptonica library") + sys.exit(3) +try: + lept = C.cdll.LoadLibrary(lept_lib) +except Exception: + stderr("Could not load the Leptonica library from %s", lept_lib) + sys.exit(3) + + +class _PIXCOLORMAP(C.Structure): + """struct PixColormap from Leptonica src/pix.h + """ + + _fields_ = [ + ("array", C.c_void_p), + ("depth", C.c_int32), + ("nalloc", C.c_int32), + ("n", C.c_int32) + ] + + +class _PIX(C.Structure): + """struct Pix from Leptonica src/pix.h + """ + + _fields_ = [ + ("w", C.c_uint32), + ("h", C.c_uint32), + ("d", C.c_uint32), + ("wpl", C.c_uint32), + ("refcount", C.c_uint32), + ("xres", C.c_int32), + ("yres", C.c_int32), + ("informat", C.c_int32), + ("text", C.POINTER(C.c_char)), + ("colormap", C.POINTER(_PIXCOLORMAP)), + ("data", C.POINTER(C.c_uint32)) + ] + + +PIX = C.POINTER(_PIX) + +lept.pixRead.argtypes = [C.c_char_p] +lept.pixRead.restype = PIX +lept.pixScale.argtypes = [PIX, C.c_float, C.c_float] +lept.pixScale.restype = PIX +lept.pixDeskew.argtypes = [PIX, C.c_int32] +lept.pixDeskew.restype = PIX +lept.pixFindSkew.argtypes = [PIX, C.POINTER(C.c_float), C.POINTER(C.c_float)] +lept.pixFindSkew.restype = C.c_int32 +lept.pixWriteImpliedFormat.argtypes = [C.c_char_p, PIX, C.c_int32, C.c_int32] +lept.pixWriteImpliedFormat.restype = C.c_int32 +lept.pixDestroy.argtypes = [C.POINTER(PIX)] +lept.pixDestroy.restype = None +lept.getLeptonicaVersion.argtypes = [] +lept.getLeptonicaVersion.restype = C.c_char_p + + +class LeptonicaErrorTrap(object): + """Context manager to trap errors reported by Leptonica. + + Leptonica's error return codes are unreliable to the point of being + almost useless. It does, however, write errors to stderr provided that is + not disabled at its compile time. Fortunately this is done using error + macros so it is very self-consistent. + + This context manager redirects stderr to a temporary file which is then + read and parsed for error messages. As a side benefit, debug messages + from Leptonica are also suppressed. + + """ + def __enter__(self): + self.tmpfile = TemporaryFile() + + # Save the old stderr, and redirect stderr to temporary file + self.old_stderr_fileno = os.dup(sys.stderr.fileno()) + os.dup2(self.tmpfile.fileno(), sys.stderr.fileno()) + return + + def __exit__(self, exc_type, exc_value, traceback): + # Restore old stderr + os.dup2(self.old_stderr_fileno, sys.stderr.fileno()) + + # Get data from tmpfile (in with block to ensure it is closed) + with self.tmpfile as tmpfile: + tmpfile.seek(0) # Cursor will be at end, so move back to beginning + leptonica_output = tmpfile.read().decode(errors='replace') + + # If there are Python errors, let them bubble up + if exc_type: + logger.warning(leptonica_output) + return False + + # If there are Leptonica errors, wrap them in Python excpetions + if 'Error' in leptonica_output: + if 'image file not found' in leptonica_output: + raise FileNotFoundError() + if 'pixWrite: stream not opened' in leptonica_output: + raise LeptonicaIOError() + raise LeptonicaError(leptonica_output) + + return False + + +class LeptonicaError(Exception): + pass + + +class LeptonicaIOError(LeptonicaError): + pass + + +def pixRead(filename): + """Load an image file into a PIX object. + + Leptonica can load TIFF, PNM (PBM, PGM, PPM), PNG, and JPEG. If loading + fails then the object will wrap a C null pointer. + + """ + with LeptonicaErrorTrap(): + return lept.pixRead(filename.encode(sys.getfilesystemencoding())) + + +def pixScale(pix, scalex, scaley): + """Returns the pix object rescaled according to the proportions given.""" + with LeptonicaErrorTrap(): + return lept.pixScale(pix, scalex, scaley) + + +def pixDeskew(pix, reduction_factor=0): + """Returns the deskewed pix object. + + A clone of the original is returned when the algorithm cannot find a skew + angle with sufficient confidence. + + reduction_factor -- amount to downsample (0 for default) when searching + for skew angle + + """ + with LeptonicaErrorTrap(): + return lept.pixDeskew(pix, reduction_factor) + + +def pixFindSkew(pix): + """Returns a tuple (deskew angle in degrees, confidence value). + + Returns (None, None) if no angle is available. + + """ + with LeptonicaErrorTrap(): + angle = C.c_float(0.0) + confidence = C.c_float(0.0) + result = lept.pixFindSkew(pix, C.byref(angle), C.byref(confidence)) + if result == 0: + return (angle.value, confidence.value) + else: + return (None, None) + + +def pixWriteImpliedFormat(filename, pix, jpeg_quality=0, jpeg_progressive=0): + """Write pix to the filename, with the extension indicating format. + + jpeg_quality -- quality (iff JPEG; 1 - 100, 0 for default) + jpeg_progressive -- (iff JPEG; 0 for baseline seq., 1 for progressive) + + """ + fileroot, extension = os.path.splitext(filename) + fix_pnm = False + if extension.lower() in ('.pbm', '.pgm', '.ppm'): + # Leptonica does not process handle these extensions correctly, but + # does handle .pnm correctly. Add another .pnm suffix. + filename += '.pnm' + fix_pnm = True + + with LeptonicaErrorTrap(): + lept.pixWriteImpliedFormat( + filename.encode(sys.getfilesystemencoding()), + pix, jpeg_quality, jpeg_progressive) + + if fix_pnm: + from shutil import move + move(filename, filename[:-4]) # Remove .pnm suffix + + +def pixDestroy(pix): + """Destroy the pix object. + + Function signature is pixDestroy(struct Pix **), hence C.byref() to pass + the address of the pointer. + + """ + with LeptonicaErrorTrap(): + lept.pixDestroy(C.byref(pix)) + + +def getLeptonicaVersion(): + """Get Leptonica version string. + + Caveat: Leptonica expects the caller to free this memory. We don't, + since that would involve binding to libc to access libc.free(), + a pointless effort to reclaim 100 bytes of memory. + + """ + return lept.getLeptonicaVersion().decode() + + +def deskew(infile, outfile, dpi): + try: + pix_source = pixRead(infile) + except LeptonicaIOError: + raise LeptonicaIOError("Failed to open file: %s" % infile) + + if dpi < 150: + reduction_factor = 1 # Don't downsample too much if DPI is already low + else: + reduction_factor = 0 # Use default + pix_deskewed = pixDeskew(pix_source, reduction_factor) + + try: + pixWriteImpliedFormat(outfile, pix_deskewed) + except LeptonicaIOError: + raise LeptonicaIOError("Failed to open destination file: %s" % outfile) + pixDestroy(pix_source) + pixDestroy(pix_deskewed) + + +if __name__ == '__main__': + parser = argparse.ArgumentParser( + description="Python wrapper to access Leptonica") + + subparsers = parser.add_subparsers(title='commands', + description='supported operations') + + parser_deskew = subparsers.add_parser('deskew') + parser_deskew.add_argument('-r', '--dpi', dest='dpi', action='store', + type=int, default=300, help='input resolution') + parser_deskew.add_argument('infile', help='image to deskew') + parser_deskew.add_argument('outfile', help='deskewed output image') + parser_deskew.set_defaults(func=deskew) + + args = parser.parse_args() + + if getLeptonicaVersion() != u'leptonica-1.69': + print("Unexpected leptonica version: %s" % getLeptonicaVersion()) + + args.func(args) + + +def _test_output(mode, extension, im_format): + from PIL import Image + from tempfile import NamedTemporaryFile + + with NamedTemporaryFile(prefix='test-lept-pnm', suffix=extension, delete=True) as tmpfile: + im = Image.new(mode=mode, size=(100, 100)) + im.save(tmpfile) + + pix = pixRead(tmpfile.name) + pixWriteImpliedFormat(tmpfile.name, pix) + pixDestroy(pix) + + im_roundtrip = Image.open(tmpfile.name) + assert im_roundtrip.mode == im.mode, "leptonica mode differs" + assert im_roundtrip.format == im_format, \ + "{0}: leptonica produced a {1}".format( + extension, + im_roundtrip.format) + + +def test_pnm_output(): + params = [['1', '.pbm', 'PPM'], ['L', '.pgm', 'PPM'], + ['RGB', '.ppm', 'PPM']] + for param in params: + _test_output(*param) + + +def test_skew_angle(): + from PIL import Image, ImageDraw + from tempfile import NamedTemporaryFile + + im = Image.new(mode='1', size=(1000, 1000), color=1) + + draw = ImageDraw.Draw(im) + for n in range(20): + draw.line([(50, 25 + 50*n), (950, 25 + 50*n)], width=1) + del draw + + test_angles = [0.1 * ang for ang in range(1, 10)] + \ + [float(ang) for ang in range(1, 7)] + test_angles += [-ang for ang in test_angles] + test_angles = sorted(test_angles) + + for rotate_angle in test_angles: + rotated_im = im.rotate(rotate_angle) + with NamedTemporaryFile(prefix='lept-skew', suffix='.png', delete=True) as tmpfile: + rotated_im.save(tmpfile) + pix = pixRead(tmpfile.name) + angle, confidence = pixFindSkew(pix) + pixDestroy(pix) + print('{0} {1} {2}'.format(rotate_angle, angle, confidence), file=sys.stderr) + + diff --git a/ocrmypdf/main.py b/ocrmypdf/main.py new file mode 100755 index 00000000..b2574123 --- /dev/null +++ b/ocrmypdf/main.py @@ -0,0 +1,842 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 + +from contextlib import suppress +from tempfile import NamedTemporaryFile, mkdtemp +import sys +import os +import fileinput +import re +import shutil +import warnings +import multiprocessing +import atexit +import textwrap + +import PyPDF2 as pypdf +from PIL import Image + +from subprocess import Popen, check_call, PIPE, CalledProcessError, \ + TimeoutExpired +try: + from subprocess import DEVNULL +except ImportError: + DEVNULL = open(os.devnull, 'wb') + + +from ruffus import transform, suffix, merge, active_if, regex, jobs_limit, \ + formatter, follows, split, collate, check_if_uptodate +import ruffus.cmdline as cmdline + +from .hocrtransform import HocrTransform +from .pageinfo import pdf_get_all_pageinfo +from .pdfa import generate_pdfa_def +from . import ghostscript +from . import tesseract + + +warnings.simplefilter('ignore', pypdf.utils.PdfReadWarning) + + +BASEDIR = os.path.dirname(os.path.realpath(__file__)) +JHOVE_PATH = os.path.realpath(os.path.join(BASEDIR, 'jhove')) +JHOVE_JAR = os.path.join(JHOVE_PATH, 'bin', 'JhoveApp.jar') +JHOVE_CFG = os.path.join(JHOVE_PATH, 'conf', 'jhove.conf') + +EXIT_BAD_ARGS = 1 +EXIT_BAD_INPUT_FILE = 2 +EXIT_MISSING_DEPENDENCY = 3 +EXIT_INVALID_OUTPUT_PDFA = 4 +EXIT_FILE_ACCESS_ERROR = 5 +EXIT_ALREADY_DONE_OCR = 6 +EXIT_OTHER_ERROR = 15 + +# ------------- +# External dependencies + +MINIMUM_TESS_VERSION = '3.02.02' + + +def complain(message): + print(textwrap.wrap(message), file=sys.stderr) + + +if tesseract.version() < MINIMUM_TESS_VERSION: + complain( + "Please install tesseract {0} or newer " + "(currently installed version is {1})".format( + MINIMUM_TESS_VERSION, tesseract.version())) + sys.exit(EXIT_MISSING_DEPENDENCY) + + +# ------------- +# Parser + +parser = cmdline.get_argparse( + prog="ocrmypdf", + description="Generate searchable PDF file from an image-only PDF file.", + version='3.0rc2', + fromfile_prefix_chars='@', + ignored_args=[ + 'touch_files_only', 'recreate_database', 'checksum_file_name', + 'key_legend_in_graph', 'draw_graph_horizontally', 'flowchart_format', + 'forced_tasks', 'target_tasks']) + +parser.add_argument( + 'input_file', + help="PDF file containing the images to be OCRed") +parser.add_argument( + 'output_file', + help="output searchable PDF file") +parser.add_argument( + '-l', '--language', action='append', + help="language of the file to be OCRed") + +metadata = parser.add_argument_group( + "Metadata options", + "Set output PDF/A metadata (default: use input document's title)") +metadata.add_argument( + '--title', type=str, + help="set document title (place multiple words in quotes)") +metadata.add_argument( + '--author', type=str, + help="set document author") +metadata.add_argument( + '--subject', type=str, + help="set document") +metadata.add_argument( + '--keywords', type=str, + help="set document keywords") + + +preprocessing = parser.add_argument_group( + "Preprocessing options", + "Improve OCR quality and final image") +preprocessing.add_argument( + '-d', '--deskew', action='store_true', + help="deskew each page before performing OCR") +preprocessing.add_argument( + '-c', '--clean', action='store_true', + help="clean pages with unpaper before performing OCR") +preprocessing.add_argument( + '-i', '--clean-final', action='store_true', + help="incorporate the cleaned image in the final PDF file") +preprocessing.add_argument( + '--oversample', metavar='DPI', type=int, default=0, + help="oversample images to improve OCR results slightly") + +parser.add_argument( + '-f', '--force-ocr', action='store_true', + help="force image into OCR, even if the page already contains text") +parser.add_argument( + '-s', '--skip-text', action='store_true', + help="skip OCR on any pages that already contain text") +parser.add_argument( + '--skip-big', type=float, metavar='MPixels', + help="skip OCR on pages larger than the specified amount of megapixels") +# parser.add_argument( +# '--exact-image', action='store_true', +# help="Use original page from PDF without re-rendering") + +advanced = parser.add_argument_group( + "Advanced", + "Advanced options for power users") +advanced.add_argument( + '--tesseract-config', default=[], type=list, action='append', + help="additional Tesseract configuration files") +advanced.add_argument( + '--pdf-renderer', choices=['tesseract', 'hocr'], default='hocr', + help='choose OCR PDF renderer') +advanced.add_argument( + '--tesseract-timeout', default=180.0, type=float, + help='give up on OCR after timeout') + +debugging = parser.add_argument_group( + "Debugging", + "Arguments to help with troubleshooting and debugging") +debugging.add_argument( + '-k', '--keep-temporary-files', action='store_true', + help="keep temporary files (helpful for debugging)") +debugging.add_argument( + '-g', '--debug-rendering', action='store_true', + help="render each page twice with debug information on second page") + +options = parser.parse_args() + + +# ---------- +# Languages + +if not options.language: + options.language = ['eng'] # Enforce English hegemony + +# Support v2.x "eng+deu" language syntax +if '+' in options.language[0]: + options.language = options.language[0].split('+') + +if not set(options.language).issubset(tesseract.languages()): + complain( + "The installed version of tesseract does not have language " + "data for the following requested languages: ") + for lang in (set(options.language) - tesseract.languages()): + complain(lang, file=sys.stderr) + sys.exit(EXIT_BAD_ARGS) + + +# ---------- +# Arguments + + +if any((options.deskew, options.clean, options.clean_final)): + try: + from . import unpaper + except ImportError: + complain( + "Install the 'unpaper' program to use --deskew or --clean.") + sys.exit(EXIT_BAD_ARGS) +else: + unpaper = None + +if options.debug_rendering and options.pdf_renderer == 'tesseract': + complain( + "Ignoring --debug-rendering because it is not supported with" + "--pdf-renderer=tesseract.") + +if options.force_ocr and options.skip_text: + complain( + "Error: --force-ocr and --skip-text are mutually incompatible.") + sys.exit(EXIT_BAD_ARGS) + +if options.clean and not options.clean_final \ + and options.pdf_renderer == 'tesseract': + complain( + "Tesseract PDF renderer cannot render --clean pages without " + "also performing --clean-final, so --clean-final is assumed.") + + +# ---------- +# Logging + + +_logger, _logger_mutex = cmdline.setup_logging(__name__, options.log_file, + options.verbose) + + +class WrappedLogger: + + def __init__(self, my_logger, my_mutex): + self.logger = my_logger + self.mutex = my_mutex + + def log(self, *args, **kwargs): + with self.mutex: + self.logger.log(*args, **kwargs) + + def debug(self, *args, **kwargs): + with self.mutex: + self.logger.debug(*args, **kwargs) + + def info(self, *args, **kwargs): + with self.mutex: + self.logger.info(*args, **kwargs) + + def warning(self, *args, **kwargs): + with self.mutex: + self.logger.warning(*args, **kwargs) + + def error(self, *args, **kwargs): + with self.mutex: + self.logger.error(*args, **kwargs) + + def critical(self, *args, **kwargs): + with self.mutex: + self.logger.critical(*args, **kwargs) + +_log = WrappedLogger(_logger, _logger_mutex) + + +def re_symlink(input_file, soft_link_name, log=_log): + """ + Helper function: relinks soft symbolic link if necessary + """ + # Guard against soft linking to oneself + if input_file == soft_link_name: + log.debug("Warning: No symbolic link made. You are using " + + "the original data directory as the working directory.") + return + + # Soft link already exists: delete for relink? + if os.path.lexists(soft_link_name): + # do not delete or overwrite real (non-soft link) file + if not os.path.islink(soft_link_name): + raise Exception("%s exists and is not a link" % soft_link_name) + try: + os.unlink(soft_link_name) + except: + log.debug("Can't unlink %s" % (soft_link_name)) + + if not os.path.exists(input_file): + raise Exception("trying to create a broken symlink to %s" % input_file) + + log.debug("os.symlink(%s, %s)" % (input_file, soft_link_name)) + + # Create symbolic link using absolute path + os.symlink( + os.path.abspath(input_file), + soft_link_name + ) + + +# ------------- +# The Pipeline + +manager = multiprocessing.Manager() +_pdfinfo = manager.list() +_pdfinfo_lock = manager.Lock() + +work_folder = mkdtemp(prefix="com.github.ocrmypdf.") + + +@atexit.register +def cleanup_working_files(*args): + if options.keep_temporary_files: + print("Temporary working files saved at:") + print(work_folder) + else: + with suppress(FileNotFoundError): + shutil.rmtree(work_folder) + + +@transform( + input=options.input_file, + filter=suffix('.pdf'), + output='.repaired.pdf', + output_dir=work_folder, + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def repair_pdf( + input_file, + output_file, + log, + pdfinfo, + pdfinfo_lock): + args_mutool = [ + 'mutool', 'clean', + input_file, output_file + ] + check_call(args_mutool) + + with pdfinfo_lock: + pdfinfo.extend(pdf_get_all_pageinfo(output_file)) + log.info(pdfinfo) + + +def get_pageinfo(input_file, pdfinfo, pdfinfo_lock): + pageno = int(os.path.basename(input_file)[0:6]) - 1 + with pdfinfo_lock: + pageinfo = pdfinfo[pageno].copy() + return pageinfo + + +def is_ocr_required(pageinfo, log): + page = pageinfo['pageno'] + 1 + ocr_required = True + if not pageinfo['images']: + # If the page has no images, then it contains vector content or text + # or both. It seems quite unlikely that one would find meaningful text + # from rasterizing vector content. So skip the page. + log.info( + "Page {0} has no images - skipping OCR".format(page) + ) + ocr_required = False + elif pageinfo['has_text']: + s = "Page {0} already has text! – {1}" + + if not options.force_ocr and not options.skip_text: + log.error(s.format(page, + "aborting (use --force-ocr to force OCR)")) + sys.exit(EXIT_ALREADY_DONE_OCR) + elif options.force_ocr: + log.info(s.format(page, + "rasterizing text and running OCR anyway")) + ocr_required = True + elif options.skip_text: + log.info(s.format(page, + "skipping all processing on this page")) + ocr_required = False + + if ocr_required and options.skip_big: + pixel_count = pageinfo['width_pixels'] * pageinfo['height_pixels'] + if pixel_count > (options.skip_big * 1000000): + ocr_required = False + log.info( + "Page {0} is very large; skipping due to -b".format(page)) + + return ocr_required + + +@split( + repair_pdf, + os.path.join(work_folder, '*.page.pdf'), + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def split_pages( + input_file, + output_files, + log, + pdfinfo, + pdfinfo_lock): + + for oo in output_files: + with suppress(FileNotFoundError): + os.unlink(oo) + args_pdfseparate = [ + 'pdfseparate', + input_file, + os.path.join(work_folder, '%06d.page.pdf') + ] + check_call(args_pdfseparate) + + from glob import glob + for filename in glob(os.path.join(work_folder, '*.page.pdf')): + pageinfo = get_pageinfo(filename, pdfinfo, pdfinfo_lock) + + alt_suffix = '.ocr.page.pdf' if is_ocr_required(pageinfo, log) \ + else '.skip.page.pdf' + re_symlink( + filename, + os.path.join( + work_folder, + os.path.basename(filename)[0:6] + alt_suffix)) + + +@transform( + input=split_pages, + filter=suffix('.ocr.page.pdf'), + output='.page.png', + output_dir=work_folder, + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def rasterize_with_ghostscript( + input_file, + output_file, + log, + pdfinfo, + pdfinfo_lock): + + pageinfo = get_pageinfo(input_file, pdfinfo, pdfinfo_lock) + + device = 'png16m' # 24-bit + if all(image['comp'] == 1 for image in pageinfo['images']): + if all(image['bpc'] == 1 for image in pageinfo['images']): + device = 'pngmono' + elif not any(image['color'] == 'color' + for image in pageinfo['images']): + device = 'pnggray' + + xres = max(pageinfo['xres'], options.oversample or 0) + yres = max(pageinfo['yres'], options.oversample or 0) + + ghostscript.rasterize_pdf(input_file, output_file, xres, yres, device, log) + + +@transform( + input=rasterize_with_ghostscript, + filter=suffix(".page.png"), + output=".pp-deskew.png", + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def preprocess_deskew( + input_file, + output_file, + log, + pdfinfo, + pdfinfo_lock): + + if not options.deskew: + re_symlink(input_file, output_file, log) + return + + pageinfo = get_pageinfo(input_file, pdfinfo, pdfinfo_lock) + dpi = int(pageinfo['xres']) + + unpaper.deskew(input_file, output_file, dpi, log) + + +@transform( + input=preprocess_deskew, + filter=suffix(".pp-deskew.png"), + output=".pp-clean.png", + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def preprocess_clean( + input_file, + output_file, + log, + pdfinfo, + pdfinfo_lock): + + if not options.clean: + re_symlink(input_file, output_file, log) + return + + pageinfo = get_pageinfo(input_file, pdfinfo, pdfinfo_lock) + dpi = int(pageinfo['xres']) + + unpaper.clean(input_file, output_file, dpi, log) + + +@active_if(options.pdf_renderer == 'hocr') +@transform( + input=preprocess_clean, + filter=suffix(".pp-clean.png"), + output=".hocr", + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def ocr_tesseract_hocr( + input_file, + output_file, + log, + pdfinfo, + pdfinfo_lock): + + pageinfo = get_pageinfo(input_file, pdfinfo, pdfinfo_lock) + + args_tesseract = [ + 'tesseract', + '-l', '+'.join(options.language), + input_file, + output_file, + 'hocr' + ] + options.tesseract_config + p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE, + universal_newlines=True) + try: + stdout, stderr = p.communicate(timeout=options.tesseract_timeout) + except TimeoutExpired: + p.kill() + stdout, stderr = p.communicate() + # Generate a HOCR file with no recognized text if tesseract times out + # Temporary workaround to hocrTransform not being able to function if + # it does not have a valid hOCR file. + with open(output_file, 'w', encoding="utf-8") as f: + f.write(tesseract.HOCR_TEMPLATE.format( + pageinfo['width_pixels'], + pageinfo['height_pixels'])) + else: + if stdout: + log.info(stdout) + if stderr: + log.error(stderr) + + if p.returncode != 0: + raise CalledProcessError(p.returncode, args_tesseract) + + if os.path.exists(output_file + '.html'): + # Tesseract 3.02 appends suffix ".html" on its own (.hocr.html) + shutil.move(output_file + '.html', output_file) + elif os.path.exists(output_file + '.hocr'): + # Tesseract 3.03 appends suffix ".hocr" on its own (.hocr.hocr) + shutil.move(output_file + '.hocr', output_file) + + # Tesseract 3.03 inserts source filename into hocr file without + # escaping it, creating invalid XML and breaking the parser. + # As a workaround, rewrite the hocr file, replacing the filename + # with a space. + regex_nested_single_quotes = re.compile( + r"""title='image "([^"]*)";""") + with fileinput.input(files=(output_file,), inplace=True) as f: + for line in f: + line = regex_nested_single_quotes.sub( + r"""title='image " ";""", line) + print(line, end='') # fileinput.input redirects stdout + + +@active_if(options.pdf_renderer == 'hocr') +@collate( + input=[rasterize_with_ghostscript, preprocess_deskew, preprocess_clean], + filter=regex(r".*/(\d{6})(?:\.page|\.pp-deskew|\.pp-clean)\.png"), + output=os.path.join(work_folder, r'\1.image'), + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def select_image_for_pdf( + infiles, + output_file, + log, + pdfinfo, + pdfinfo_lock): + if options.clean_final: + image_suffix = '.pp-clean.png' + elif options.deskew: + image_suffix = '.pp-deskew.png' + else: + image_suffix = '.page.png' + image = next(ii for ii in infiles if ii.endswith(image_suffix)) + + pageinfo = get_pageinfo(image, pdfinfo, pdfinfo_lock) + if all(image['enc'] == 'jpeg' for image in pageinfo['images']): + # If all images were JPEGs originally, produce a JPEG as output + Image.open(image).save(output_file, format='JPEG') + else: + re_symlink(image, output_file) + + +@active_if(options.pdf_renderer == 'hocr') +@collate( + input=[select_image_for_pdf, ocr_tesseract_hocr], + filter=regex(r".*/(\d{6})(?:\.image|\.hocr)"), + output=os.path.join(work_folder, r'\1.rendered.pdf'), + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def render_hocr_page( + infiles, + output_file, + log, + pdfinfo, + pdfinfo_lock): + hocr = next(ii for ii in infiles if ii.endswith('.hocr')) + image = next(ii for ii in infiles if ii.endswith('.image')) + + pageinfo = get_pageinfo(image, pdfinfo, pdfinfo_lock) + dpi = round(max(pageinfo['xres'], pageinfo['yres'], options.oversample)) + + hocrtransform = HocrTransform(hocr, dpi) + hocrtransform.to_pdf(output_file, imageFileName=image, + showBoundingboxes=False, invisibleText=True) + + +@active_if(options.pdf_renderer == 'hocr') +@active_if(options.debug_rendering) +@collate( + input=[select_image_for_pdf, ocr_tesseract_hocr], + filter=regex(r".*/(\d{6})(?:\.image|\.hocr)"), + output=os.path.join(work_folder, r'\1.debug.pdf'), + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def render_hocr_debug_page( + infiles, + output_file, + log, + pdfinfo, + pdfinfo_lock): + hocr = next(ii for ii in infiles if ii.endswith('.hocr')) + image = next(ii for ii in infiles if ii.endswith('.image')) + + pageinfo = get_pageinfo(image, pdfinfo, pdfinfo_lock) + dpi = round(max(pageinfo['xres'], pageinfo['yres'], options.oversample)) + + hocrtransform = HocrTransform(hocr, dpi) + hocrtransform.to_pdf(output_file, imageFileName=None, + showBoundingboxes=True, invisibleText=False) + + +@active_if(options.pdf_renderer == 'tesseract') +@collate( + input=[preprocess_clean, split_pages], + filter=regex(r".*/(\d{6})(?:\.pp-clean\.png|\.page\.pdf)"), + output=os.path.join(work_folder, r'\1.rendered.pdf'), + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def tesseract_ocr_and_render_pdf( + input_files, + output_file, + log, + pdfinfo, + pdfinfo_lock): + + input_image = next((ii for ii in input_files if ii.endswith('.png')), '') + input_pdf = next((ii for ii in input_files if ii.endswith('.pdf'))) + if not input_image: + # Skipping this page + re_symlink(input_pdf, output_file) + return + + args_tesseract = [ + 'tesseract', + '-l', '+'.join(options.language), + input_image, + os.path.splitext(output_file)[0], # Tesseract appends suffix + 'pdf' + ] + options.tesseract_config + p = Popen(args_tesseract, close_fds=True, stdout=PIPE, stderr=PIPE, + universal_newlines=True) + + try: + stdout, stderr = p.communicate(timeout=options.tesseract_timeout) + if stdout: + log.info(stdout) + if stderr: + log.error(stderr) + except TimeoutError: + p.kill() + log.info("Tesseract - page timed out") + re_symlink(input_pdf, output_file) + + +@transform( + input=repair_pdf, + filter=suffix('.repaired.pdf'), + output='.pdfa_def.ps', + output_dir=work_folder, + extras=[_log]) +def generate_postscript_stub( + input_file, + output_file, + log): + + pdf = pypdf.PdfFileReader(input_file) + + def from_document_info(key): + # pdf.documentInfo.get() DOES NOT work as expected + try: + s = pdf.documentInfo[key] + return str(s) + except KeyError: + return '' + + pdfmark = { + 'title': from_document_info('/Title'), + 'author': from_document_info('/Author'), + 'keywords': from_document_info('/Keywords'), + 'subject': from_document_info('/Subject'), + } + if options.title: + pdfmark['title'] = options.title + if options.author: + pdfmark['author'] = options.author + if options.keywords: + pdfmark['keywords'] = options.keywords + if options.subject: + pdfmark['subject'] = options.subject + + generate_pdfa_def(output_file, pdfmark) + + +@transform( + input=split_pages, + filter=suffix('.skip.page.pdf'), + output='.done.pdf', + output_dir=work_folder, + extras=[_log]) +def skip_page( + input_file, + output_file, + log): + re_symlink(input_file, output_file, log) + + +@merge( + input=[render_hocr_page, render_hocr_debug_page, skip_page, + tesseract_ocr_and_render_pdf, generate_postscript_stub], + output=os.path.join(work_folder, 'merged.pdf'), + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def merge_pages( + input_files, + output_file, + log, + pdfinfo, + pdfinfo_lock): + + def input_file_order(s): + '''Sort order: All rendered pages followed + by their debug page, if any, followed by Postscript stub. + Ghostscript documentation has the Postscript stub at the + beginning, but it works at the end and also gets document info + right that way.''' + if s.endswith('.ps'): + return 99999999 + key = int(os.path.basename(s)[0:6]) * 10 + if 'debug' in os.path.basename(s): + key += 1 + return key + + pdf_pages = sorted(input_files, key=input_file_order) + log.info(pdf_pages) + ghostscript.generate_pdfa(pdf_pages, output_file) + + +@transform( + input=merge_pages, + filter=formatter(), + output=options.output_file, + extras=[_log, _pdfinfo, _pdfinfo_lock]) +def validate_pdfa( + input_file, + output_file, + log, + pdfinfo, + pdfinfo_lock): + + args_jhove = [ + 'java', + '-jar', JHOVE_JAR, + '-c', JHOVE_CFG, + '-m', 'PDF-hul', + input_file + ] + p_jhove = Popen(args_jhove, close_fds=True, universal_newlines=True, + stdout=PIPE, stderr=DEVNULL) + stdout, _ = p_jhove.communicate() + + log.debug(stdout) + if p_jhove.returncode != 0: + log.error(stdout) + raise RuntimeError( + "Unexpected error while checking compliance to PDF/A file.") + + pdf_is_valid = True + if re.search(r'ErrorMessage', stdout, + re.IGNORECASE | re.MULTILINE): + pdf_is_valid = False + if re.search(r'^\s+Status.*not valid', stdout, + re.IGNORECASE | re.MULTILINE): + pdf_is_valid = False + if re.search(r'^\s+Status.*Not well-formed', stdout, + re.IGNORECASE | re.MULTILINE): + pdf_is_valid = False + + pdf_is_pdfa = False + if re.search(r'^\s+Profile:.*PDF/A-1', stdout, + re.IGNORECASE | re.MULTILINE): + pdf_is_pdfa = True + + if not pdf_is_valid: + log.warning('Output file: The generated PDF/A file is INVALID') + elif pdf_is_valid and not pdf_is_pdfa: + log.warning('Output file: Generated file is a VALID PDF but not PDF/A') + elif pdf_is_valid and pdf_is_pdfa: + log.info('Output file: The generated PDF/A file is VALID') + shutil.copy(input_file, output_file) + + +# @active_if(ocr_required and options.exact_image) +# @merge([render_hocr_blank_page, extract_single_page], +# os.path.join(work_folder, "%04i.merged.pdf") % pageno) +# def merge_hocr_with_original_page(infiles, output_file): +# with open(infiles[0], 'rb') as hocr_input, \ +# open(infiles[1], 'rb') as page_input, \ +# open(output_file, 'wb') as output: +# hocr_reader = pypdf.PdfFileReader(hocr_input) +# page_reader = pypdf.PdfFileReader(page_input) +# writer = pypdf.PdfFileWriter() + +# the_page = hocr_reader.getPage(0) +# the_page.mergePage(page_reader.getPage(0)) +# writer.addPage(the_page) +# writer.write(output) + + +def available_cpu_count(): + try: + return multiprocessing.cpu_count() + except NotImplementedError: + pass + + try: + import psutil + return psutil.cpu_count() + except (ImportError, AttributeError): + pass + + complain( + "Could not get CPU count. Assuming one (1) CPU." + "Use -j N to set manually.") + return 1 + + +def run_pipeline(): + cmdline.run(options, multiprocess=available_cpu_count()) + + +if __name__ == '__main__': + run_pipeline() diff --git a/ocrmypdf/pageinfo.py b/ocrmypdf/pageinfo.py new file mode 100644 index 00000000..81231653 --- /dev/null +++ b/ocrmypdf/pageinfo.py @@ -0,0 +1,137 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 + +from subprocess import Popen, PIPE +from decimal import Decimal, getcontext +import re +import sys +import PyPDF2 as pypdf + + +FRIENDLY_COLORSPACE = { + '/DeviceGray': 'gray', + '/CalGray': 'gray', + '/DeviceRGB': 'rgb', + '/CalRGB': 'rgb', + '/DeviceCMYK': 'cmyk', + '/Lab': 'lab', + '/ICCBased': 'icc', + '/Indexed': 'index', + '/Separation': 'sep', + '/DeviceN': 'devn', + '/Pattern': '-' +} + +FRIENDLY_ENCODING = { + '/CCITTFaxDecode': 'ccitt', + '/DCTDecode': 'jpeg', + '/JPXDecode': 'jpx', + '/JBIG2Decode': 'jbig2', +} + +FRIENDLY_COMP = { + 'gray': 1, + 'rgb': 3, + 'cmyk': 4, + 'lab': 3, +} + + +def _page_has_inline_images(page): + # PDF always uses \r\n for separator regardless of platform + # Really basic heuristic that might trigger the odd false positive + # This is only finds the first image and is not quite spec compliant + contents = page.getContents() + data = contents.getData() + begin_image, image_data, end_image = False, False, False + for data in re.split(b'\s+', data): + if data == b'BI': + begin_image = True + elif data == b'ID': + image_data = True + elif data == b'EI': + end_image = True + if all((begin_image, image_data, end_image)): + return True + + +def _find_page_images(page, pageinfo): + try: + page['/Resources']['/XObject'] + except KeyError: + return + + # Look for XObject (out of line images) + for xobj in page['/Resources']['/XObject']: + # PyPDF2 returns the keys as an iterator + pdfimage = page['/Resources']['/XObject'][xobj] + if pdfimage['/Subtype'] != '/Image': + continue + if '/ImageMask' in pdfimage: + if pdfimage['/ImageMask']: + continue + image = {} + image['width'] = pdfimage['/Width'] + image['height'] = pdfimage['/Height'] + image['bpc'] = pdfimage['/BitsPerComponent'] + if '/Filter' in pdfimage: + filter_ = pdfimage['/Filter'] + if isinstance(filter_, pypdf.generic.ArrayObject): + filter_ = filter_[0] + image['enc'] = FRIENDLY_ENCODING.get(filter_, 'image') + else: + image['enc'] = 'image' + if '/ColorSpace' in pdfimage: + cs = pdfimage['/ColorSpace'] + if isinstance(cs, pypdf.generic.ArrayObject): + cs = cs[0] + image['color'] = FRIENDLY_COLORSPACE.get(cs, '-') + else: + image['color'] = 'jpx' if image['enc'] == 'jpx' else '?' + + image['comp'] = FRIENDLY_COMP.get(image['color'], '?') + image['dpi_w'] = image['width'] / pageinfo['width_inches'] + image['dpi_h'] = image['height'] / pageinfo['height_inches'] + image['dpi'] = (image['dpi_w'] * image['dpi_h']) ** Decimal(0.5) + yield image + + +def _pdf_get_pageinfo(infile, page: int): + pageinfo = {} + pageinfo['pageno'] = page + pageinfo['images'] = [] + + pdf = pypdf.PdfFileReader(infile) + page = pdf.pages[page - 1] + + text = page.extractText() + pageinfo['has_text'] = (text.strip() != '') + + width_pt = page['/MediaBox'][2] - page['/MediaBox'][0] + height_pt = page['/MediaBox'][3] - page['/MediaBox'][1] + pageinfo['width_inches'] = width_pt / Decimal(72.0) + pageinfo['height_inches'] = height_pt / Decimal(72.0) + + pageinfo['images'] = [im for im in _find_page_images(page, pageinfo)] + + # Look for inline images + if _page_has_inline_images(page): + raise NotImplementedError( + "Warning: input PDF contains inline images - not supported") + + if pageinfo['images']: + xres = max(image['dpi_w'] for image in pageinfo['images']) + yres = max(image['dpi_h'] for image in pageinfo['images']) + pageinfo['xres'], pageinfo['yres'] = xres, yres + pageinfo['width_pixels'] = \ + int(round(xres * pageinfo['width_inches'])) + pageinfo['height_pixels'] = \ + int(round(yres * pageinfo['height_inches'])) + + return pageinfo + + +def pdf_get_all_pageinfo(infile): + pdf = pypdf.PdfFileReader(infile) + getcontext().prec = 6 + return [_pdf_get_pageinfo(infile, n) for n in range(pdf.numPages)] diff --git a/ocrmypdf/pdfa.py b/ocrmypdf/pdfa.py new file mode 100644 index 00000000..503830a9 --- /dev/null +++ b/ocrmypdf/pdfa.py @@ -0,0 +1,130 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 +# +# Generate a PDFA_def.ps file for Ghostscript >= 9.14 + +from __future__ import print_function, absolute_import, division +from string import Template +from subprocess import Popen, PIPE +import os +import codecs + + +# This is a template written in PostScript which is needed to create PDF/A +# files, from the Ghostscript documentation. Lines beginning with % are +# comments. Python substitution variables have a '$' prefix. +pdfa_def_template = u"""%! +% This is a sample prefix file for creating a PDF/A document. +% Feel free to modify entries marked with "Customize". +% This assumes an ICC profile to reside in the file (ISO Coated sb.icc), +% unless the user modifies the corresponding line below. + +% Define entries in the document Info dictionary : +/ICCProfile ($icc_profile) +def + +[ /Title <$title> + /Author <$author> + /Subject <$subject> + /Keywords <$keywords> + /DOCINFO pdfmark + +% Define an ICC profile : + +[/_objdef {icc_PDFA} /type /stream /OBJ pdfmark +[{icc_PDFA} +<< + /N currentpagedevice /ProcessColorModel known { + currentpagedevice /ProcessColorModel get dup /DeviceGray eq + {pop 1} { + /DeviceRGB eq + {3}{4} ifelse + } ifelse + } { + (ERROR, unable to determine ProcessColorModel) == flush + } ifelse +>> /PUT pdfmark +[{icc_PDFA} ICCProfile (r) file /PUT pdfmark + +% Define the output intent dictionary : + +[/_objdef {OutputIntent_PDFA} /type /dict /OBJ pdfmark +[{OutputIntent_PDFA} << + /Type /OutputIntent % Must be so (the standard requires). + /S /GTS_PDFA1 % Must be so (the standard requires). + /DestOutputProfile {icc_PDFA} % Must be so (see above). + /OutputConditionIdentifier ($icc_identifier) +>> /PUT pdfmark +[{Catalog} <> /PUT pdfmark +""" + + +def encode_text_string(s: str) -> str: + '''Encode text string to hex string for use in a PDF + + From PDF 32000-1:2008 a string object may be included in hexademical form + if it is enclosed in angle brackets. For general Unicode the string should + be UTF-16 (big endian) with byte order marks. A non-hexademical + representation is doable but this is preferable since it allows the output + Postscript file to be completely ASCII and no escaping of Postscript + characters is necessary. + ''' + if s == '': + return '' + utf16_bytes = s.encode('utf-16be') + ascii_hex_bytes = codecs.encode(b'\xfe\xff' + utf16_bytes, 'hex') + ascii_hex_str = ascii_hex_bytes.decode('ascii').lower() + return ascii_hex_str + + +def _get_pdfa_def(icc_profile, icc_identifier, pdfmark): + pdfmark_utf16 = {k: encode_text_string(v) for k, v in pdfmark.items()} + + t = Template(pdfa_def_template) + result = t.substitute(icc_profile=icc_profile, + icc_identifier=icc_identifier, + title=pdfmark_utf16.get('title', ''), + author=pdfmark_utf16.get('author', ''), + subject=pdfmark_utf16.get('subject', ''), + keywords=pdfmark_utf16.get('keywords', '')) + return result + + +def _get_postscript_icc_path(): + "Parse Ghostscript's help message to find where iccprofiles are stored" + + p_gs = Popen(['gs', '--help'], close_fds=True, universal_newlines=True, + stdout=PIPE, stderr=PIPE) + out, _ = p_gs.communicate() + lines = out.splitlines() + + def search_paths(lines): + seeking = True + for line in lines: + if seeking: + if line.startswith('Search path'): + seeking = False + continue + else: + if line.strip().startswith('/'): + yield from ( + path.strip() for path in line.split(':') + if path.strip() != '') + for root in search_paths(lines): + path = os.path.realpath(os.path.join(root, '../iccprofiles')) + if os.path.exists(path): + return path + + +def generate_pdfa_def(target_filename, pdfmark, icc='sRGB'): + if icc == 'sRGB': + icc_profile = os.path.join(_get_postscript_icc_path(), 'srgb.icc') + else: + raise NotImplementedError("Only supporting sRGB") + + ps = _get_pdfa_def(icc_profile, icc, pdfmark) + + # Since PostScript might not handle UTF-8 (it's hard to get a clear + # answer), insist on ascii + with open(target_filename, 'w', encoding='ascii') as f: + f.write(ps) diff --git a/ocrmypdf/tesseract.py b/ocrmypdf/tesseract.py new file mode 100644 index 00000000..9a542019 --- /dev/null +++ b/ocrmypdf/tesseract.py @@ -0,0 +1,61 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 + +from subprocess import STDOUT, CalledProcessError, check_output +import sys +import os +import re +from functools import lru_cache + + +@lru_cache(maxsize=1) +def version(): + args_tess = [ + 'tesseract', + '--version' + ] + try: + versions = check_output( + args_tess, close_fds=True, universal_newlines=True, + stderr=STDOUT) + except CalledProcessError: + print("Could not find Tesseract executable on system PATH.") + sys.exit(1) + + tesseract_version = re.match(r'tesseract\s(.+)', versions).group(1) + return tesseract_version + + +@lru_cache(maxsize=1) +def languages(): + args_tess = [ + 'tesseract', + '--list-langs' + ] + langs = check_output( + args_tess, close_fds=True, universal_newlines=True, + stderr=STDOUT) + return set(lang.strip() for lang in langs.splitlines()[1:]) + + +HOCR_TEMPLATE = ''' + + + + + + + + + +
+
+

+ + +

+
+
+ +''' diff --git a/ocrmypdf/test/test_pageinfo.py b/ocrmypdf/test/test_pageinfo.py new file mode 100644 index 00000000..19bfd65f --- /dev/null +++ b/ocrmypdf/test/test_pageinfo.py @@ -0,0 +1,107 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 + +from ocrmypdf import pageinfo +from reportlab.pdfgen.canvas import Canvas +from PIL import Image +from tempfile import NamedTemporaryFile +from contextlib import suppress +import os +import sys +import shutil +import pytest +from pkg_resources import Requirement, resource_filename + +req = Requirement.parse('ocrmypdf') + +TEST_OUTPUT = os.path.join(os.path.dirname(__file__), 'output') + + +def setup_module(): + with suppress(FileNotFoundError): + shutil.rmtree(TEST_OUTPUT) + with suppress(FileExistsError): + os.mkdir(TEST_OUTPUT) + + +def test_single_page_text(): + filename = os.path.join(TEST_OUTPUT, 'text.pdf') + pdf = Canvas(filename, pagesize=(8*72, 6*72)) + text = pdf.beginText() + text.setFont('Helvetica', 12) + text.setTextOrigin(1*72, 3*72) + text.textLine("Methink'st thou art a general offence and every" + " man should beat thee.") + pdf.drawText(text) + pdf.showPage() + pdf.save() + + pdfinfo = pageinfo.pdf_get_all_pageinfo(filename) + + assert len(pdfinfo) == 1 + page = pdfinfo[0] + + assert page['has_text'] + assert len(page['images']) == 0 + + +def test_single_page_image(): + filename = os.path.join(TEST_OUTPUT, 'image-mono.pdf') + pdf = Canvas(filename, pagesize=(72, 72)) + with NamedTemporaryFile() as im_tmp: + im = Image.new('1', (8, 8), 0) + for n in range(8): + im.putpixel((n, n), 1) + im.save(im_tmp.name, format='PNG') + # Draw image in a 72x72 pt or 1"x1" area + pdf.drawImage(im_tmp.name, 0, 0, width=72, height=72) + pdf.showPage() + pdf.save() + + pdfinfo = pageinfo.pdf_get_all_pageinfo(filename) + + assert len(pdfinfo) == 1 + page = pdfinfo[0] + + assert not page['has_text'] + assert len(page['images']) == 1 + + pdfimage = page['images'][0] + assert pdfimage['width'] == 8 + # assert pdfimage['color'] == 'gray' + + # While unexpected, this is correct + # PDF spec says /FlateDecode image must have /BitsPerComponent 8 + # So mono images get upgraded to 8-bit + assert pdfimage['bpc'] == 8 + + # DPI in a 1"x1" is the image width + assert pdfimage['dpi_w'] == 8 + assert pdfimage['dpi_h'] == 8 + + +def test_single_page_inline_image(): + filename = os.path.join(TEST_OUTPUT, 'image-mono-inline.pdf') + pdf = Canvas(filename, pagesize=(8*72, 6*72)) + with NamedTemporaryFile() as im_tmp: + im = Image.new('1', (8, 8), 0) + for n in range(8): + im.putpixel((n, n), 1) + im.save(im_tmp.name, format='PNG') + # Draw image in a 72x72 pt or 1"x1" area + pdf.drawInlineImage(im_tmp.name, 0, 0, width=72, height=72) + pdf.showPage() + pdf.save() + + with pytest.raises(NotImplementedError): + pageinfo.pdf_get_all_pageinfo(filename) + + +def test_jpeg(): + filename = resource_filename(req, 'tests/resources/c02-22.pdf') + + pdfinfo = pageinfo.pdf_get_all_pageinfo(filename) + + pdfimage = pdfinfo[0]['images'][0] + assert pdfimage['enc'] == 'jpeg' + diff --git a/ocrmypdf/unpaper.py b/ocrmypdf/unpaper.py new file mode 100644 index 00000000..a8c0d202 --- /dev/null +++ b/ocrmypdf/unpaper.py @@ -0,0 +1,84 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 +# unpaper documentation: +# https://github.com/Flameeyes/unpaper/blob/master/doc/basic-concepts.md + +from subprocess import Popen, PIPE +from tempfile import NamedTemporaryFile +import sys +import os +from functools import lru_cache + + +@lru_cache(maxsize=1) +def version(): + args_unpaper = [ + 'unpaper', + '--version' + ] + p_unpaper = Popen(args_unpaper, close_fds=True, universal_newlines=True, + stdout=PIPE, stderr=PIPE) + version, _ = p_unpaper.communicate(timeout=5) + + return version.strip() + + +try: + from PIL import Image +except ImportError: + print("Could not find Python3 imaging library", file=sys.stderr) + raise + + +def run(input_file, output_file, dpi, log, mode_args): + args_unpaper = [ + 'unpaper', + '-v', + '--dpi', str(dpi) + ] + mode_args + + SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'} + suffix = '' + + im = Image.open(input_file) + suffix = SUFFIXES[im.mode] + with NamedTemporaryFile(suffix=suffix) as input_pnm, \ + NamedTemporaryFile(suffix=suffix, mode="r+b") as output_pnm: + im.save(input_pnm, format='PPM') + im.close() + + os.unlink(output_pnm.name) + + args_unpaper.extend([input_pnm.name, output_pnm.name]) + p_unpaper = Popen( + args_unpaper, close_fds=True, + universal_newlines=True, stdout=PIPE, stderr=PIPE + ) + out, err = p_unpaper.communicate() + log.debug(out) + log.debug(err) + + Image.open(output_pnm.name).save(output_file) + + +def deskew(input_file, output_file, dpi, log): + run(input_file, output_file, dpi, log, [ + '--mask-scan-size', '100', # don't blank out narrow columns + '--no-border-align', # don't align visible content to borders + '--no-mask-center', # don't center visible content within page + '--no-grayfilter', # don't remove light gray areas + '--no-blackfilter', # don't remove solid black areas + '--no-noisefilter', # don't remove salt and pepper noise + '--no-blurfilter' # don't remove blurry objects/debris + ]) + + +def clean(input_file, output_file, dpi, log): + run(input_file, output_file, dpi, log, [ + '--mask-scan-size', '100', # don't blank out narrow columns + '--no-border-align', # don't align visible content to borders + '--no-mask-center', # don't center visible content within page + '--no-grayfilter', # don't remove light gray areas + '--no-blackfilter', # don't remove solid black areas + '--no-deskew', # don't deskew + ]) diff --git a/pipeline.svg b/pipeline.svg new file mode 100644 index 00000000..1a2e2f44 --- /dev/null +++ b/pipeline.svg @@ -0,0 +1,235 @@ + + + + + + +Pipeline: + +clustertasks + +Pipeline: + + +t0 + + + + +repair_pdf + + +t1 + + +split_pages + + +t0->t1 + + + + +t10 + + + + +generate_postscript_stub + + +t0->t10 + + + + +t2 + + + + +rasterize_with_ghostscript + + +t1->t2 + + + + +t11 + + + + +skip_page + + +t1->t11 + + + + +t9 + + + + +tesseract_ocr_and_render_pdf + + +t1->t9 + + + + +t3 + + + + +preprocess_deskew + + +t2->t3 + + + + +t6 + + + + +select_image_for_pdf + + +t2->t6 + + + + +t4 + + + + +preprocess_clean + + +t3->t4 + + + + +t3->t6 + + + + +t4->t6 + + + + +t5 + + + + +ocr_tesseract_hocr + + +t4->t5 + + + + +t4->t9 + + + + +t7 + + + + +render_hocr_page + + +t6->t7 + + + + +t8 + + + + +render_hocr_debug_page + + +t6->t8 + + + + +t5->t7 + + + + +t5->t8 + + + + +t12 + + +merge_pages + + +t7->t12 + + + + +t8->t12 + + + + +t11->t12 + + + + +t9->t12 + + + + +t10->t12 + + + + +t13 + + + + +validate_pdfa + + +t12->t13 + + + + + diff --git a/setup.cfg b/setup.cfg new file mode 100644 index 00000000..d6262b02 --- /dev/null +++ b/setup.cfg @@ -0,0 +1,2 @@ +[bdist_wheel] +python-tag = py34 \ No newline at end of file diff --git a/setup.py b/setup.py new file mode 100644 index 00000000..22ef6870 --- /dev/null +++ b/setup.py @@ -0,0 +1,214 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 + +from setuptools import setup +from subprocess import Popen, STDOUT, check_output, CalledProcessError +from string import Template +import re +import sys + + +missing_program = ''' +The program '{program}' could not be executed or was not found on your +system PATH. +''' + +unknown_version = ''' +OCRmyPDF requires '{program}' {need_version} or higher. Your system has +'{program}' but we cannot tell what version is installed. Contact the +package maintainer. +''' + +old_version = ''' +OCRmyPDF requires '{program}' {need_version} or higher. Your system appears +to have {found_version}. Please update this program. +''' + +okay_its_optional = ''' +This program is OPTIONAL, so installation of OCRmyPDF can proceed, but +some functionality may be missing. +''' + +not_okay_its_required = ''' +This program is REQUIRED for OCRmyPDF to work. Installation will abort. +''' + +osx_install_advice = ''' +If you have homebrew installed, try these command to install the missing +packages: + brew update + brew upgrade + brew install {package} +''' + +linux_install_advice = ''' +On systems with the aptitude package manager (Debian, Ubuntu), try these +commands: + sudo apt-get update + sudo apt-get install {package} + +On RPM-based systems (Red Hat, Fedora), search for instructions on +installing the RPM for {package}. +''' + + +def _error_trailer(program, package, optional): + if program == 'java': + return # You're fucked + + if optional: + print(okay_its_optional.format(**locals()), file=sys.stderr) + else: + print(not_okay_its_required.format(**locals()), file=sys.stderr) + if sys.platform.startswith('darwin'): + print(osx_install_advice.format(**locals()), file=sys.stderr) + elif sys.platform.startswith('linux'): + print(linux_install_advice.format(**locals()), file=sys.stderr) + + +def error_missing_program( + program, + package, + optional + ): + print(missing_program.format(**locals()), file=sys.stderr) + _error_trailer(**locals()) + + +def error_unknown_version( + program, + package, + optional + ): + print(unknown_version.format(**locals()), file=sys.stderr) + _error_trailer(**locals()) + + +def error_old_version( + program, + package, + optional, + need_version + ): + print(old_version.format(**locals()), file=sys.stderr) + _error_trailer(**locals()) + + +def check_external_program( + program, + minimum_version, + package, + version_check_args=['--version'], + version_scrape_regex=re.compile(r'(\d+\.\d+(?:\.\d+)?)'), + optional=False): + + print('Checking for {program} >= {minimum_version}...'.format( + program=program, minimum_version=minimum_version)) + try: + result = check_output( + [program] + version_check_args, + universal_newlines=True, stderr=STDOUT) + except CalledProcessError: + error_missing_program(program, package, optional) + if not optional: + sys.exit(1) + + try: + version = version_scrape_regex.search(result).group(1) + except AttributeError: + error_unknown_version(program, package, optional, minimum_version) + if not optional: + sys.exit(1) + + if version < minimum_version: + error_old_version(program, package, optional, minimum_version) + + print('Found {program} {version}'.format( + program=program, version=version)) + +command = next((arg for arg in sys.argv[1:] if not arg.startswith('-')), '') + +if command.startswith('install') or \ + command in ['check', 'test', 'nosetests', 'easy_install']: + check_external_program( + program='tesseract', + minimum_version='3.02.02', + package='tesseract' + ) + check_external_program( + program='gs', + minimum_version='9.14', + package='ghostscript' + ) + check_external_program( + program='unpaper', + minimum_version='6.1', + package='unpaper', + optional=True + ) + # Deprecated + check_external_program( + program='pdfseparate', + minimum_version='0.29.0', + package='poppler', + version_check_args=['-v'] + ) + check_external_program( + program='java', + minimum_version='1.5.0', + package='Java Runtime Environment', + version_check_args=['-version'] + ) + check_external_program( + program='mutool', + minimum_version='1.7a', + version_check_args=['-v'], + version_scrape_regex=re.compile(r'(\d+\.\d+[a-z]+)'), + package='mupdf-tools' + ) + +setup( + name='ocrmypdf', + version='3.0rc2', + description='OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched', + url='https://github.com/fritz-hh/OCRmyPDF', + author='J. R. Barlow', + author_email='jim@purplerock.ca', + license='Public Domain', + packages=['ocrmypdf'], + keywords=['PDF', 'OCR', 'optical character recognition', 'PDF/A', 'scanning'], + classifiers=[ + "Programming Language :: Python :: 3", + "Development Status :: 4 - Beta", + "Environment :: Console", + "Intended Audience :: End Users/Desktop", + "Intended Audience :: Science/Research", + "Intended Audience :: System Administrators", + "License :: Public Domain", + "Operating System :: MacOS :: MacOS X", + "Operating System :: POSIX", + "Operating System :: POSIX :: BSD", + "Operating System :: POSIX :: Linux", + "Topic :: Scientific/Engineering :: Image Recognition", + "Topic :: Text Processing :: Indexing", + "Topic :: Text Processing :: Linguistic", + ], + install_requires=[ + 'ruffus>=2.6.3', + 'Pillow>=2.7.0', + 'lxml>=3.4.2', + 'reportlab>=3.1.44', + 'PyPDF2>=1.25.1' + ], + entry_points={ + 'console_scripts': [ + 'ocrmypdf = ocrmypdf.main:run_pipeline' + ], + }, + eager_resources=[ + 'ocrmypdf/jhove/bin/*.jar', + 'ocrmypdf/jhove/conf/*.conf', + 'ocrmypdf/jhove/lib/*.jar' + ], + include_package_data=True, + zip_safe=False) diff --git a/src/config.sh b/src/config.sh deleted file mode 100644 index 951992c5..00000000 --- a/src/config.sh +++ /dev/null @@ -1,36 +0,0 @@ -##################################################################################### -# The following parameters might be changed by the user -##################################################################################### - -DEFAULT_LANGUAGES="eng" # Default language(s) of the PDF file. The language should be set correctly in order to get good OCR results. - # Any language supported by tesseract is supported (Tesseract uses 3-character ISO 639-2 language codes) - # Multiple languages may be specified, separated by '+' characters. - -DEFAULT_DPI=300 # dpi value used as fall back if the page dpi cannot be determined - -##################################################################################### -# Do NOT change the following parameters -##################################################################################### - -TOOLNAME="OCRmyPDF" -VERSION="v3.x" - -# possible exit codes -EXIT_BAD_ARGS="1" -EXIT_BAD_INPUT_FILE="2" -EXIT_MISSING_DEPENDENCY="3" -EXIT_INVALID_OUTPUT_PDFA="4" -EXIT_FILE_ACCESS_ERROR="5" -EXIT_OTHER_ERROR="15" - -# possible log levels -LOG_ERR="0" # only error messages -LOG_WARN="1" # error messages and warnings -LOG_INFO="2" # error messages, warnings and some infos -LOG_DEBUG="3" # debug level logging - -# various paths -SRC="./src" # location of the source folder (except source of external tools like jhove) -OCR_PAGE="$SRC/ocrPage.sh" # path to the script aimed at OCRing one page -JHOVE="./jhove/bin/JhoveApp.jar" # java SW for validating the final PDF/A -JHOVE_CFG="./jhove/conf/jhove.conf" # location of the jhove config file diff --git a/src/hocrTransform.py b/src/hocrTransform.py deleted file mode 100755 index 726a5c00..00000000 --- a/src/hocrTransform.py +++ /dev/null @@ -1,203 +0,0 @@ -#!/usr/local/bin/python2 -# coding: utf-8 -############################################################################## -# Copyright (c) 2013-14: fritz-hh from Github (https://github.com/fritz-hh) -# -# Copyright (c) 2010: Jonathan Brinley from Github (https://github.com/jbrinley/HocrConverter) -# Initial version by Jonathan Brinley, jonathanbrinley@gmail.com -############################################################################## -from reportlab.pdfgen.canvas import Canvas -from reportlab.lib.units import inch -from lxml import etree as ElementTree -from PIL import Image -import re, sys -import argparse - - -class hocrTransform(): - """ - A class for converting documents from the hOCR format. - For details of the hOCR format, see: - http://docs.google.com/View?docid=dfxcv4vc_67g844kf - """ - def __init__(self, hocrFileName, dpi): - self.dpi = dpi - self.boxPattern = re.compile('bbox((\s+\d+){4})') - - self.hocr = ElementTree.ElementTree() - self.hocr.parse(hocrFileName) - - # if the hOCR file has a namespace, ElementTree requires its use to find elements - matches = re.match('({.*})html', self.hocr.getroot().tag) - self.xmlns = '' - if matches: - self.xmlns = matches.group(1) - - # get dimension in pt (not pixel!!!!) of the OCRed image - self.width, self.height = None, None - for div in self.hocr.findall(".//%sdiv[@class='ocr_page']"%(self.xmlns)): - coords = self.element_coordinates(div) - self.width = self.px2pt(coords[2]-coords[0]) - self.height = self.px2pt(coords[3]-coords[1]) - break # there shouldn't be more than one, and if there is, we don't want it - - # no width and heigh definition in the ocr_image element of the hocr file - if self.width is None: - print("No page dimension found in the hocr file") - sys.exit(1) - - def __str__(self): - """ - Return the textual content of the HTML body - """ - if self.hocr is None: - return '' - body = self.hocr.find(".//%sbody"%(self.xmlns)) - if body: - return self._get_element_text(body).encode('utf-8') # XML gives unicode - else: - return '' - - def _get_element_text(self, element): - """ - Return the textual content of the element and its children - """ - text = '' - if element.text is not None: - text = text + element.text - for child in element.getchildren(): - text = text + self._get_element_text(child) - if element.tail is not None: - text = text + element.tail - return text - - def element_coordinates(self, element): - """ - Returns a tuple containing the coordinates of the bounding box around - an element - """ - out = (0,0,0,0) - if 'title' in element.attrib: - matches = self.boxPattern.search(element.attrib['title']) - if matches: - coords = matches.group(1).split() - out = (int(coords[0]),int(coords[1]),int(coords[2]),int(coords[3])) - return out - - def px2pt(self, pxl): - """ - Returns the length in pt given length in pxl - """ - return float(pxl)/self.dpi*inch - - def replace_unsupported_chars(self, str): - """ - Given an input string, returns the corresponding string that: - - is available in the helvetica facetype - - does not contain any ligature (to allow easy search in the PDF file) - """ - # The 'u' before the character to replace indicates that it is a unicode character - str=str.replace(u"fl","fl") - str=str.replace(u"fi","fi") - return str - - def to_pdf(self, outFileName, imageFileName, showBoundingboxes, fontname="Helvetica"): - """ - Creates a PDF file with an image superimposed on top of the text. - Text is positioned according to the bounding box of the lines in - the hOCR file. - The image need not be identical to the image used to create the hOCR file. - It can have a lower resolution, different color mode, etc. - """ - # create the PDF file - pdf = Canvas(outFileName, pagesize=(self.width, self.height), pageCompression=1) # page size in points (1/72 in.) - - # draw bounding box for each paragraph - pdf.setStrokeColorRGB(0,1,1) # light blue for bounding box of paragraph - pdf.setFillColorRGB(0,1,1) # light blue for bounding box of paragraph - pdf.setLineWidth(0) # no line for bounding box - for elem in self.hocr.findall(".//%sp[@class='%s']" % (self.xmlns, "ocr_par")): - - elemtxt=self._get_element_text(elem).rstrip() - if len(elemtxt) == 0: - continue - - coords = self.element_coordinates(elem) - x1=self.px2pt(coords[0]) - y1=self.px2pt(coords[1]) - x2=self.px2pt(coords[2]) - y2=self.px2pt(coords[3]) - - # draw the bbox border - if showBoundingboxes == True: - pdf.rect(x1, self.height-y2, x2-x1, y2-y1, fill=1) - - - # check if element with class 'ocrx_word' are available - # otherwise use 'ocr_line' as fallback - elemclass="ocr_line" - if self.hocr.find(".//%sspan[@class='ocrx_word']" %(self.xmlns)) is not None: - elemclass="ocrx_word" - - # itterate all text elements - pdf.setStrokeColorRGB(1,0,0) # light green for bounding box of word/line - pdf.setLineWidth(0.5) # bounding box line width - pdf.setDash(6,3) # bounding box is dashed - pdf.setFillColorRGB(0,0,0) # text in black - for elem in self.hocr.findall(".//%sspan[@class='%s']" % (self.xmlns, elemclass)): - - elemtxt=self._get_element_text(elem).rstrip() - - elemtxt=self.replace_unsupported_chars(elemtxt) - - if len(elemtxt) == 0: - continue - - coords = self.element_coordinates(elem) - x1=self.px2pt(coords[0]) - y1=self.px2pt(coords[1]) - x2=self.px2pt(coords[2]) - y2=self.px2pt(coords[3]) - - # draw the bbox border - if showBoundingboxes == True: - pdf.rect(x1, self.height-y2, x2-x1, y2-y1, fill=0) - - text = pdf.beginText() - fontsize=self.px2pt(coords[3]-coords[1]) - text.setFont(fontname, fontsize) - - # set cursor to bottom left corner of bbox (adjust for dpi) - text.setTextOrigin(x1, self.height-y2) - - # scale the width of the text to fill the width of the bbox - text.setHorizScale(100*(x2-x1)/pdf.stringWidth(elemtxt, fontname, fontsize)) - - # write the text to the page - text.textLine(elemtxt) - pdf.drawText(text) - - # put the image on the page, scaled to fill the page - if imageFileName != None: - im = Image.open(imageFileName) - pdf.drawInlineImage(im, 0, 0, width=self.width, height=self.height) - - # finish up the page and save it - pdf.showPage() - pdf.save() - - -if __name__ == "__main__": - parser = argparse.ArgumentParser(description='Convert hocr file to PDF') - parser.add_argument('-b', '--boundingboxes', action="store_true", default=False, help='Show bounding boxes borders') - parser.add_argument('-r', '--resolution', type=int, default=300, help='Resolution of the image that was OCRed') - parser.add_argument('-i', '--image', default=None, help='Path to the image to be placed above the text') - parser.add_argument('hocrfile', help='Path to the hocr file to be parsed') - parser.add_argument('outputfile', help='Path to the PDF file to be generated') - args = parser.parse_args() - - hocr = hocrTransform(args.hocrfile, args.resolution) - hocr.to_pdf(args.outputfile, args.image, args.boundingboxes) - - - diff --git a/src/ocrPage.sh b/src/ocrPage.sh deleted file mode 100755 index a2a53144..00000000 --- a/src/ocrPage.sh +++ /dev/null @@ -1,234 +0,0 @@ -#!/bin/sh -############################################################################## -# Script aimed at OCRing a single page of a PDF file -# -# Copyright (c) 2013-14: fritz-hh from Github (https://github.com/fritz-hh) -############################################################################## - -. "./src/config.sh" - - -# Initialization of variables passed by arguments -FILE_INPUT_PDF="$1" # PDF file containing the page to be OCRed -PAGE_INFO="$2" # Various characteristics of the page to be OCRed -NUM_PAGES="$3" # Total number of page of the PDF file (required for logging) -TMP_FLD="$4" # Folder where the temporary files should be placed -VERBOSITY="$5" # Requested verbosity -LAN="$6" # Language of the file to be OCRed -KEEP_TMP="$7" # Keep the temporary files after processing (helpful for debugging) -PREPROCESS_DESKEW="$8" # Deskew the page to be OCRed -PREPROCESS_CLEAN="$9" # Clean the page to be OCRed -PREPROCESS_CLEANTOPDF="${10}" # Put the cleaned paged in the OCRed PDF -OVERSAMPLING_DPI="${11}" # Oversampling resolution in dpi -PDF_NOIMG="${12}" # Request to generate also a PDF page containing only the OCRed text but no image (helpful for debugging) -FORCE_OCR="${13}" # Force to OCR, even if the page already contains fonts -SKIP_TEXT="${14}" # Skip OCR on pages that contain fonts and include the page anyway -TESS_CFG_FILES="${15}" # Specific configuration files to be used by Tesseract during OCRing - - - - -################################## -# Detect the characteristics of the embedded image for -# the page number provided as parameter -# -# Param 1: page number -# Param 2: PDF page width in pt -# Param 3: PDF page height in pt -# Param 4: temporary file path (Path of the file in which the output should be written) -# Output: A file containing the characteristics of the embedded image. File structure: -# DPI= -# COLOR_SPACE= -# DEPTH= -# Returns: -# - 0: if no error occurs -# - 1: in case the page already contains fonts (which should be the case for PDF generated from scanned pages) -# - 2: in case the page contains more than one image -################################## -getImgInfo() { - local page widthPDF heightPDF curImgInfo nbImg curImg propCurImg widthCurImg heightCurImg colorspaceCurImg depthCurImg dpi - - # page number - page="$1" - # width / height of PDF page (in pt) - widthPDF="$2" - heightPDF="$3" - # path of the file in which the output should be written - curImgInfo="$4" - - - [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Size ${heightPDF}x${widthPDF} (h*w in pt)" - - - # check if the page already contains fonts (which should not be the case for PDF based on scanned files - [ `pdffonts -f $page -l $page "${FILE_INPUT_PDF}" | wc -l` -gt 2 ] && echo "Page $page: Page already contains font data !!!" && return 1 - - - # extract raw image from pdf file to compute resolution - # unfortunately this image can have another orientation than in the pdf... - # so we will have to extract it again later using pdftoppm - pdfimages -f $page -l $page -j "$FILE_INPUT_PDF" "$curOrigImg" 1>&2 - # count number of extracted images - nbImg=$((`ls -1 "$curOrigImg"* 2>/dev/null | wc -l`)) - if [ $nbImg -ne "1" ]; then - [ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Expecting exactly 1 image covering the whole page (found $nbImg). Cannot compute dpi value." - return 2 - fi - # Get characteristics of the extracted image - curImg=`ls -1 "$curOrigImg"* 2>/dev/null` - propCurImg=`identify -format "%w %h %[colorspace] %[depth]" "$curImg"` - widthCurImg=`echo "$propCurImg" | cut -f1 -d" "` - heightCurImg=`echo "$propCurImg" | cut -f2 -d" "` - colorspaceCurImg=`echo "$propCurImg" | cut -f3 -d" "` - depthCurImg=`echo "$propCurImg" | cut -f4 -d" "` - [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Size ${heightCurImg}x${widthCurImg} (in pixel)" - - # compute the resolution of the image (making the assumption that x & y resolution are equal) - # and round it to the nearest integer - dpi=`echo "scale=5;sqrt($widthCurImg*72*$heightCurImg*72/$widthPDF/$heightPDF)+0.5" | bc` - dpi=`echo "scale=0;$dpi/1" | bc` - - # save the image characteristics - echo "DPI=$dpi" > "$curImgInfo" - echo "COLOR_SPACE=$colorspaceCurImg" >> "$curImgInfo" - echo "DEPTH=$depthCurImg" >> "$curImgInfo" - - return 0 -} - - -page=`echo $PAGE_INFO | cut -f1 -d" "` -[ $VERBOSITY -ge $LOG_INFO ] && echo "Processing page $page / $NUM_PAGES" - -# get width / height of PDF page (in pt) -widthPDF=`echo $PAGE_INFO | cut -f2 -d" "` -heightPDF=`echo $PAGE_INFO | cut -f3 -d" "` - -# create the name of the required temporary files -curOrigImg="$TMP_FLD/${page}.orig-img" # original image available in the current PDF page - # (the image file may have a different orientation than in the pdf file) -curHocr="$TMP_FLD/${page}.hocr" # hocr file to be generated by the OCR SW for the current page -curOCRedPDF="$TMP_FLD/${page}.ocred.pdf" # PDF file containing the image + the OCRed text for the current page -curOCRedPDFDebug="$TMP_FLD/${page}.ocred.todebug.pdf" # PDF file containing data required to find out if OCR worked correctly -curImgInfo="$TMP_FLD/${page}.orig-img-info.txt" # Detected characteristics of the embedded image - - -# auto-detect the characteristics of the embedded image -depthCurImg="8" # default color depth -colorspaceCurImg="sRGB" # default color space -dpi=$DEFAULT_DPI # default resolution - -getImgInfo "$page" "$widthPDF" "$heightPDF" "$curImgInfo" -ret_code="$?" - - -# Handle pages that already contain a text layer -if ([ "$ret_code" -eq "1" ] && [ "$FORCE_OCR" -eq "0" ] && [ "$SKIP_TEXT" -eq "0" ]); then - echo "Page $page: Exiting... (Use the -f option to force OCR, even though fonts are available in the input file. Or use -s to include the page as is (no OCR) in the output file)" && exit $EXIT_BAD_INPUT_FILE -elif ([ "$ret_code" -eq "1" ] && [ "$SKIP_TEXT" -eq "1" ]); then - [ $VERBOSITY -ge $LOG_INFO ] && echo "Page $page: Skipping processing because page contains text..." - pdfseparate -f $page -l $page "${FILE_INPUT_PDF}" "$curOCRedPDF" - exit 0 -elif ([ "$ret_code" -eq "1" ] && [ "$FORCE_OCR" -eq "1" ]); then - [ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: OCRing anyway, assuming a default resolution of $dpi dpi" -# in case the page contains more than one image, warn the user but go on with default parameters -elif [ "$ret_code" -eq "2" ]; then - [ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Continuing anyway, assuming a default resolution of $dpi dpi" -else - # read the image characteristics from the file - dpi=`cat "$curImgInfo" | grep "^DPI=" | cut -f2 -d"="` - colorspaceCurImg=`cat "$curImgInfo" | grep "^COLOR_SPACE=" | cut -f2 -d"="` - depthCurImg=`cat "$curImgInfo" | grep "^DEPTH=" | cut -f2 -d"="` -fi - -# perform oversampling if the resolution is not sufficient to get good OCR results -if [ "$dpi" -lt "$OVERSAMPLING_DPI" ]; then - [ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Low image resolution detected ($dpi dpi). Performing oversampling ($OVERSAMPLING_DPI dpi) to try to get better OCR results." - dpi="$OVERSAMPLING_DPI" -elif [ "$dpi" -lt "200" ]; then - [ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Low image resolution detected ($dpi dpi). If needed, please use the \"-o\" to try to get better OCR results." -fi - -# Identify if page image should be saved as ppm (color), pgm (gray) or pbm (b&w) -ext="ppm" # by default (color image) the extension of the extracted image is ppm -opt="" # by default (color image) no option as to be passed to pdftoppm -if [ "$colorspaceCurImg" = "Gray" ] && [ "$depthCurImg" = "1" ]; then # if monochrome (b&w) - ext="pbm" - opt="-mono" -elif [ "$colorspaceCurImg" = "Gray" ]; then # if gray - ext="pgm" - opt="-gray" -fi -curImgPixmap="$TMP_FLD/$page.$ext" -curImgPixmapDeskewed="$TMP_FLD/$page.deskewed.$ext" -curImgPixmapClean="$TMP_FLD/$page.cleaned.$ext" - -# extract current page as image with correct orientation and resolution -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Extracting image as $ext file (${dpi} dpi)" -! pdftoppm -f $page -l $page -r $dpi $opt "$FILE_INPUT_PDF" > "$curImgPixmap" \ - && echo "Could not extract page $page as $ext from \"$FILE_INPUT_PDF\". Exiting..." && exit $EXIT_OTHER_ERROR - -# if requested deskew image (without changing its size in pixel) -widthCurImg=$(($dpi*$widthPDF/72)) -heightCurImg=$(($dpi*$heightPDF/72)) -if [ "$PREPROCESS_DESKEW" -eq "1" ]; then - [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Deskewing image" - ! convert "$curImgPixmap" -deskew 40% -gravity center -extent ${widthCurImg}x${heightCurImg} "$curImgPixmapDeskewed" \ - && echo "Could not deskew \"$curImgPixmap\". Exiting..." && exit $EXIT_OTHER_ERROR -else - ln -s `basename "$curImgPixmap"` "$curImgPixmapDeskewed" -fi - -# if requested clean image with unpaper to get better OCR results -if [ "$PREPROCESS_CLEAN" -eq "1" ]; then - [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Cleaning image with unpaper" - ! unpaper --dpi $dpi --mask-scan-size 100 \ - --no-deskew --no-grayfilter --no-blackfilter --no-mask-center --no-border-align \ - "$curImgPixmapDeskewed" "$curImgPixmapClean" 1> /dev/null \ - && echo "Could not clean \"$curImgPixmapDeskewed\". Exiting..." && exit $EXIT_OTHER_ERROR -else - ln -s `basename "$curImgPixmapDeskewed"` "$curImgPixmapClean" -fi - -# perform OCR -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Performing OCR" -! tesseract -l "$LAN" "$curImgPixmapClean" "$curHocr" hocr $TESS_CFG_FILES 1> /dev/null 2> /dev/null \ - && echo "Could not OCR file \"$curImgPixmapClean\". Exiting..." && exit $EXIT_OTHER_ERROR -# Tesseract names the output files differently in some distributions. -if [ -e "$curHocr.html" ]; then - mv "$curHocr.html" "$curHocr" -elif [ -e "$curHocr.hocr" ]; then - mv "$curHocr.hocr" "$curHocr" -elif [ ! -e "$curHocr" ]; then - echo "\"$curHocr[.html|.hocr]\" not found. Exiting..." && exit $EXIT_OTHER_ERROR -fi - -# embed text and image to new pdf file -if [ "$PREPROCESS_CLEANTOPDF" -eq "1" ]; then - image4finalPDF="$curImgPixmapClean" -else - image4finalPDF="$curImgPixmapDeskewed" -fi -[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Embedding text in PDF" -! python2 $SRC/hocrTransform.py -r $dpi -i "$image4finalPDF" "$curHocr" "$curOCRedPDF" \ - && echo "Could not create PDF file from \"$curHocr\". Exiting..." && exit $EXIT_OTHER_ERROR - -# if requested generate special debug PDF page with visible OCR text -if [ $PDF_NOIMG -eq "1" ] ; then - [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Embedding text in PDF (debug page)" - ! python2 $SRC/hocrTransform.py -b -r $dpi "$curHocr" "$curOCRedPDFDebug" \ - && echo "Could not create PDF file from \"$curHocr\". Exiting..." && exit $EXIT_OTHER_ERROR -fi - -# delete temporary files created for the current page -# to avoid using to much disk space in case of PDF files having many pages -if [ $KEEP_TMP -eq 0 ]; then - rm -f "$curOrigImg"* - rm -f "$curHocr" - rm -f "$curImgPixmap" - rm -f "$curImgPixmapDeskewed" - rm -f "$curImgPixmapClean" - rm -f "$curImgInfo" -fi - -exit 0 diff --git a/src/test/OCRmyPDF_severaltimes.sh b/src/test/OCRmyPDF_severaltimes.sh deleted file mode 100644 index 683a4b9d..00000000 --- a/src/test/OCRmyPDF_severaltimes.sh +++ /dev/null @@ -1,15 +0,0 @@ -#!/bin/sh -# -# Perform OCR several times in order to find how quicly the quality decreases - -cpt=1 - -while [ $cpt -le 10 ] ; do - - echo "------- Itteration $cpt ---------" - - ! ../../OCRmyPDF.sh -vv -l deu -k ../../tmp/ocred-$(($cpt-1)).pdf ../../tmp/ocred-$cpt.pdf && exit 1 - - cpt=$(($cpt+1)) - -done \ No newline at end of file diff --git a/src/test/README.md b/src/test/README.md deleted file mode 100644 index 622c8e5c..00000000 --- a/src/test/README.md +++ /dev/null @@ -1,4 +0,0 @@ -Note -==== - -The file(s) located in this folder are aimed at testing the OCRmyPDF script diff --git a/tests/resources/LinnSequencer.jpg b/tests/resources/LinnSequencer.jpg new file mode 100644 index 00000000..bb76d0c9 Binary files /dev/null and b/tests/resources/LinnSequencer.jpg differ diff --git a/tests/resources/NOTE.rst b/tests/resources/NOTE.rst new file mode 100644 index 00000000..7e2b7fa5 --- /dev/null +++ b/tests/resources/NOTE.rst @@ -0,0 +1,17 @@ +All test resources must come from free public domain sources for +copyright reasons. + +Test files do not necessarily produce perfect (or even good) OCR +results. + ++-------------------+--------------------------------------------------------------------------------+ +| File | Source | ++===================+================================================================================+ +| graph.pdf | Wikimedia | ++-------------------+--------------------------------------------------------------------------------+ +| c02-22.pdf | Project Gutenberg: https://www.gutenberg.org/files/76/76-h/images/c02-22.jpg | ++-------------------+--------------------------------------------------------------------------------+ +| LinnSequencer.jpg | Wikimedia: https://upload.wikimedia.org/wikipedia/en/b/b7/LinnSequencer_hardware_MIDI_sequencer_brochure_page_2_300dpi.jpg | ++-------------------+--------------------------------------------------------------------------------+ +| congress.jpg | http://www.baxleystamps.com/litho/meiji/courts_1871.jpg | ++-------------------+--------------------------------------------------------------------------------+ diff --git a/src/test/Test_Issue_28.pdf b/tests/resources/Test_Issue_28.pdf similarity index 100% rename from src/test/Test_Issue_28.pdf rename to tests/resources/Test_Issue_28.pdf diff --git a/tests/resources/c02-22.pdf b/tests/resources/c02-22.pdf new file mode 100644 index 00000000..c9f77df4 Binary files /dev/null and b/tests/resources/c02-22.pdf differ diff --git a/tests/resources/congress.jpg b/tests/resources/congress.jpg new file mode 100644 index 00000000..d63d7026 Binary files /dev/null and b/tests/resources/congress.jpg differ diff --git a/tests/resources/enormous.pdf b/tests/resources/enormous.pdf new file mode 100644 index 00000000..ee7f82b1 Binary files /dev/null and b/tests/resources/enormous.pdf differ diff --git a/tests/resources/graph.pdf b/tests/resources/graph.pdf new file mode 100644 index 00000000..00db54d3 Binary files /dev/null and b/tests/resources/graph.pdf differ diff --git a/tests/resources/graph_ocred.pdf b/tests/resources/graph_ocred.pdf new file mode 100644 index 00000000..46ac8fc2 Binary files /dev/null and b/tests/resources/graph_ocred.pdf differ diff --git a/tests/resources/multipage.pdf b/tests/resources/multipage.pdf new file mode 100644 index 00000000..4fd395fe Binary files /dev/null and b/tests/resources/multipage.pdf differ diff --git a/tests/resources/skew.pdf b/tests/resources/skew.pdf new file mode 100644 index 00000000..d6eff087 Binary files /dev/null and b/tests/resources/skew.pdf differ diff --git a/tests/test_main.py b/tests/test_main.py new file mode 100644 index 00000000..770853aa --- /dev/null +++ b/tests/test_main.py @@ -0,0 +1,170 @@ +#!/usr/bin/env python3 +# © 2015 James R. Barlow: github.com/jbarlow83 + +from __future__ import print_function +from subprocess import Popen, PIPE, check_output +import os +import shutil +from contextlib import suppress +import sys +from unittest.mock import patch, create_autospec +import pytest +from ocrmypdf.pageinfo import pdf_get_all_pageinfo + + +if sys.version_info.major < 3: + print("Requires Python 3.4+") + sys.exit(1) + +TESTS_ROOT = os.path.abspath(os.path.dirname(__file__)) +PROJECT_ROOT = os.path.dirname(TESTS_ROOT) +OCRMYPDF = os.path.join(PROJECT_ROOT, 'OCRmyPDF.sh') +TEST_RESOURCES = os.path.join(PROJECT_ROOT, 'tests', 'resources') +TEST_OUTPUT = os.path.join(PROJECT_ROOT, 'tests', 'output') + + +def setup_module(): + with suppress(FileNotFoundError): + shutil.rmtree(TEST_OUTPUT) + with suppress(FileExistsError): + os.mkdir(TEST_OUTPUT) + + +def run_ocrmypdf_sh(input_file, output_file, *args): + sh_args = ['sh', OCRMYPDF] + list(args) + [input_file, output_file] + sh = Popen( + sh_args, close_fds=True, stdout=PIPE, stderr=PIPE, + universal_newlines=True) + out, err = sh.communicate() + return sh, out, err + + +def check_ocrmypdf(input_basename, output_basename, *args): + input_file = os.path.join(TEST_RESOURCES, input_basename) + output_file = os.path.join(TEST_OUTPUT, output_basename) + + sh, _, err = run_ocrmypdf_sh(input_file, output_file, *args) + assert sh.returncode == 0, err + assert os.path.exists(output_file), "Output file not created" + assert os.stat(output_file).st_size > 100, "PDF too small or empty" + return output_file + + +def test_quick(): + check_ocrmypdf('c02-22.pdf', 'test_quick.pdf') + + +def test_deskew(): + # Run with deskew + deskewed_pdf = check_ocrmypdf('skew.pdf', 'test_deskew.pdf', '-d') + + # Now render as an image again and use Leptonica to find the skew angle + # to confirm that it was deskewed + from ocrmypdf.ghostscript import rasterize_pdf + import logging + log = logging.getLogger() + + deskewed_png = os.path.join(TEST_OUTPUT, 'deskewed.png') + + rasterize_pdf( + deskewed_pdf, + deskewed_png, + xres=150, + yres=150, + raster_device='pngmono', + log=log) + + from ocrmypdf.leptonica import pixRead, pixDestroy, pixFindSkew + pix = pixRead(deskewed_png) + skew_angle, skew_confidence = pixFindSkew(pix) + pix = pixDestroy(pix) + + print(skew_angle) + assert -0.5 < skew_angle < 0.5, "Deskewing failed" + + +def test_clean(): + check_ocrmypdf('skew.pdf', 'test_clean.pdf', '-c') + + +def test_metadata(): + pdf = check_ocrmypdf( + 'c02-22.pdf', 'test_metadata.pdf', + '--title', 'Du siehst den Wald vor lauter Bäumen nicht.', + '--author', '孔子', + '--subject', 'U+1030C is: 𐌌') + + out_pdfinfo = check_output(['pdfinfo', pdf], universal_newlines=True) + lines_pdfinfo = out_pdfinfo.splitlines() + pdfinfo = {} + for line in lines_pdfinfo: + k, v = line.strip().split(':', maxsplit=1) + pdfinfo[k.strip()] = v.strip() + + assert pdfinfo['Title'] == 'Du siehst den Wald vor lauter Bäumen nicht.' + assert pdfinfo['Author'] == '孔子' + assert pdfinfo['Subject'] == 'U+1030C is: 𐌌' + assert pdfinfo.get('Keywords', '') == '' + + +def check_oversample(renderer): + oversampled_pdf = check_ocrmypdf( + 'skew.pdf', 'test_oversample_%s.pdf' % renderer, '--oversample', '300', + '--pdf-renderer', renderer) + + pdfinfo = pdf_get_all_pageinfo(oversampled_pdf) + + print(pdfinfo[0]['xres']) + assert abs(pdfinfo[0]['xres'] - 300) < 1 + + +def test_oversample(): + yield check_oversample, 'hocr' + yield check_oversample, 'tesseract' + + +def test_repeat_ocr(): + sh, _, _ = run_ocrmypdf_sh('graph_ocred.pdf', 'wontwork.pdf') + assert sh.returncode != 0 + + +def test_force_ocr(): + out = check_ocrmypdf('graph_ocred.pdf', 'test_force.pdf', '-f') + pdfinfo = pdf_get_all_pageinfo(out) + assert pdfinfo[0]['has_text'] + + +def test_skip_ocr(): + check_ocrmypdf('graph_ocred.pdf', 'test_skip.pdf', '-s') + + +def check_ocr_timeout(renderer): + out = check_ocrmypdf('skew.pdf', 'test_timeout_%s.pdf' % renderer, + '--tesseract-timeout', '1.0') + pdfinfo = pdf_get_all_pageinfo(out) + assert pdfinfo[0]['has_text'] == False + + +def test_ocr_timeout(): + yield check_ocr_timeout, 'hocr' + yield check_ocr_timeout, 'tesseract' + + +def test_skip_big(): + out = check_ocrmypdf('enormous.pdf', 'test_enormous.pdf', + '--skip-big', '10') + pdfinfo = pdf_get_all_pageinfo(out) + assert pdfinfo[0]['has_text'] == False + + +def check_maximum_options(renderer): + check_ocrmypdf( + 'multipage.pdf', 'test_multipage%s.pdf' % renderer, + '-d', '-c', '-i', '-g', '-f', '-k', '--oversample', '300', + '--skip-big', '10', '--title', 'Too Many Weird Files', + '--author', 'py.test', '--pdf-renderer', renderer) + + +def test_maximum_options(): + yield check_maximum_options, 'hocr' + yield check_maximum_options, 'tesseract'