From 4ce249e6eddb7b40b6e4869e13b62898620ab08f Mon Sep 17 00:00:00 2001 From: fritz-hh Date: Mon, 22 Apr 2013 22:50:34 +0200 Subject: [PATCH] OCRmyPDF.sh: new debug option (-g) added --- OCRmyPDF.sh | 31 +++++++++++++++++++++++-------- 1 file changed, 23 insertions(+), 8 deletions(-) diff --git a/OCRmyPDF.sh b/OCRmyPDF.sh index 5cb14990..56c8c9e3 100644 --- a/OCRmyPDF.sh +++ b/OCRmyPDF.sh @@ -12,10 +12,14 @@ tesseract engine) Copyright: fritz from NAS4Free forum Version: $VERSION -Usage: OCRmyPDF.sh [-h] [-v] [-k] [-d] [-c] [-i] [-l language] inputfile +Usage: OCRmyPDF.sh [-h] [-v] [-g] [-k] [-d] [-c] [-i] [-l language] inputfile -h : Display this help message -v : Increase the verbosity (this option can be used more than once) +-g : Activate debug mode: + - Generates a PDF file containing each page twice (once with the image, once without the image + but with the OCRed text as well as the detected bounding boxes detected during OCR) + - Set the verbosity to the highest possible -k : Do not delete the temporary files -d : Deskew each page before performing OCR -c : Clean each page before performing OCR @@ -49,12 +53,14 @@ KEEP_TMP="0" # do not delete the temporary files (default) PREPROCESS_DESKEW="0" # 0=no, 1=yes (deskew image) PREPROCESS_CLEAN="0" # 0=no, 1=yes (clean image to improve OCR) PREPROCESS_CLEANINPDF="0" # 0=no, 1=yes (put cleaned image in final PDF) +DEBUG_MODE="0" # 0=no, 1=yes (generates each PDF page twice, with and without image) # Parse optional command line arguments -while getopts ":hvkdcil:" opt; do +while getopts ":hvgkdcil:" opt; do case $opt in h) usage ; exit 0 ;; v) VERBOSITY=$(($VERBOSITY+1)) ;; + g) VERBOSITY="10"; DEBUG_MODE="1" ;; k) KEEP_TMP="1" ;; d) PREPROCESS_DESKEW="1" ;; c) PREPROCESS_CLEAN="1" ;; @@ -136,10 +142,11 @@ while read pageSize ; do [ $VERBOSITY -ge $LOG_INFO ] && echo "Processing page $page" # create the name of the required file - curOrigImg="$tmp/${page}_Image" # original image available in the current PDF page - # (the image file may have a different orientation than in the pdf file) - curHocr="$tmp/$page.hocr" # hocr file to be generated by the OCR SW for the current page - curOCRedPDF="$tmp/${page}-ocred.pdf" # PDF file containing the image + the OCRed text for the current page + curOrigImg="$tmp/${page}_Image" # original image available in the current PDF page + # (the image file may have a different orientation than in the pdf file) + curHocr="$tmp/$page.hocr" # hocr file to be generated by the OCR SW for the current page + curOCRedPDF="$tmp/${page}-ocred.pdf" # PDF file containing the image + the OCRed text for the current page + curOCRedPDFDebug="$tmp/${page}-debug-ocred.pdf" # PDF file containing data required to find out if OCR worked correctly [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Computing embedded image resolution" # get width / height of PDF page @@ -208,15 +215,22 @@ while read pageSize ; do mv "$curHocr.html" "$curHocr" # embed text and image to new pdf file - [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Embedding text in PDF" if [ "$PREPROCESS_CLEANINPDF" -eq "1" ]; then image4finalPDF="$curImgPixmapClean" else image4finalPDF="$curImgPixmapDeskewed" fi + [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Embedding text in PDF" ! python hocrTransform.py -r $dpi -i "$image4finalPDF" "$curHocr" "$curOCRedPDF" \ && echo "Could not create PDF file from \"$curHocr\". Exiting..." && exit $EXIT_OTHER_ERROR + # if requested generate special debug PDF page with visible OCR text + if [ $DEBUG_MODE -eq "1" ] ; then + [ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Embedding text in PDF (debug page)" + ! python hocrTransform.py -b -r $dpi "$curHocr" "$curOCRedPDFDebug" \ + && echo "Could not create PDF file from \"$curHocr\". Exiting..." && exit $EXIT_OTHER_ERROR + fi + # delete temporary files created for the current page # to avoid using to much disk space in case of PDF files having many pages if [ $KEEP_TMP -eq 0 ]; then @@ -229,6 +243,7 @@ while read pageSize ; do # go to next page of the pdf cpt=$(($cpt+1)) + done < "$FILE_SIZE_PAGES" @@ -262,7 +277,7 @@ grep -i "Status.*not valid" "$FILE_VALIDATION_LOG" && pdf_valid=0 grep -i "Status.*Not well-formed" "$FILE_VALIDATION_LOG" && pdf_valid=0 [ $pdf_valid -eq 1 ] && echo "Output file: The generated PDF/A file is VALID" \ || echo "Output file: The generated PDF/A file is INVALID" - +[ $VERBOSITY -ge $LOG_DEBUG ] && cat "$FILE_VALIDATION_LOG"