Compare commits
44
Commits
v2.0-rc1
...
v2.0-stable
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1c34fd69cf | ||
|
|
4cf38404cc | ||
|
|
be830ddc31 | ||
|
|
18322b424f | ||
|
|
6901c60db4 | ||
|
|
e369ce6766 | ||
|
|
64e4e5d91e | ||
|
|
efce7de9ae | ||
|
|
38c64ac689 | ||
|
|
6d203e3eee | ||
|
|
81f461e557 | ||
|
|
988bde1387 | ||
|
|
aedbabdbe8 | ||
|
|
6ed53e53c7 | ||
|
|
a78630ce99 | ||
|
|
6653066784 | ||
|
|
e40f1fa081 | ||
|
|
a872ce751d | ||
|
|
317846fbdc | ||
|
|
f581a55544 | ||
|
|
447b291e70 | ||
|
|
01d07253e8 | ||
|
|
034a466094 | ||
|
|
c6211e2335 | ||
|
|
1d03a6417d | ||
|
|
1d62ef27a2 | ||
|
|
996048dc08 | ||
|
|
bf02ee3bdc | ||
|
|
a3c7fba02d | ||
|
|
a8cd7febf6 | ||
|
|
20c008b84f | ||
|
|
7cd73566be | ||
|
|
e56fd53d06 | ||
|
|
810b1b3b3e | ||
|
|
cb0b033fe7 | ||
|
|
46f673a3b7 | ||
|
|
455303b3d4 | ||
|
|
24a84d6380 | ||
|
|
9aa2171052 | ||
|
|
3a46ea1f36 | ||
|
|
d33779f301 | ||
|
|
d6ea0793b8 | ||
|
|
4e5e5bb925 | ||
|
|
3232ed8e38 |
+63
-36
@@ -41,11 +41,11 @@ Usage: OCRmyPDF.sh [-h] [-v] [-g] [-k] [-d] [-c] [-i] [-o dpi] [-f] [-l languag
|
|||||||
-f : Force to OCR the whole document, even if some page already contain font data
|
-f : Force to OCR the whole document, even if some page already contain font data
|
||||||
(which should not be the case for PDF files built from scnanned images)
|
(which should not be the case for PDF files built from scnanned images)
|
||||||
-l : Set the language of the PDF file in order to improve OCR results (default "eng")
|
-l : Set the language of the PDF file in order to improve OCR results (default "eng")
|
||||||
Any language supported by tesseract is supported.
|
Any language supported by tesseract is supported (Tesseract uses 3-character ISO 639-2 language codes)
|
||||||
|
Multiple languages may be specified, separated by '+' characters.
|
||||||
-C : Pass an additional configuration file to the tesseract OCR engine.
|
-C : Pass an additional configuration file to the tesseract OCR engine.
|
||||||
(this option can be used more than once)
|
(this option can be used more than once)
|
||||||
Note 1: The configuration file must be available in the "tessdata/configs" folder of your tesseract installation
|
Note 1: The configuration file must be available in the "tessdata/configs" folder of your tesseract installation
|
||||||
Note 2: The folder "./tess-cfg" contains useful tesseract configuration files
|
|
||||||
inputfile : PDF file to be OCRed
|
inputfile : PDF file to be OCRed
|
||||||
outputfile : The PDF/A file that will be generated
|
outputfile : The PDF/A file that will be generated
|
||||||
--------------------------------------------------------------------------------------
|
--------------------------------------------------------------------------------------
|
||||||
@@ -111,8 +111,7 @@ done
|
|||||||
# Remove the optional arguments parsed above.
|
# Remove the optional arguments parsed above.
|
||||||
shift $((OPTIND-1))
|
shift $((OPTIND-1))
|
||||||
|
|
||||||
# Check if the number of mandatory parameters
|
# Check if the number of mandatory parameters provided is as expected
|
||||||
# provided is as expected
|
|
||||||
if [ "$#" -ne "2" ]; then
|
if [ "$#" -ne "2" ]; then
|
||||||
echo "Exactly two mandatory argument shall be provided ($# arguments provided)"
|
echo "Exactly two mandatory argument shall be provided ($# arguments provided)"
|
||||||
usage
|
usage
|
||||||
@@ -141,20 +140,36 @@ cd "`dirname $0`"
|
|||||||
! command -v pdfimages > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
! command -v pdfimages > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
! command -v pdftoppm > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
! command -v pdftoppm > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
! command -v pdffonts > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
! command -v pdffonts > /dev/null && echo "Please install poppler-utils. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
! command -v pdftk > /dev/null && echo "Please install pdftk. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
|
||||||
[ $PREPROCESS_CLEAN -eq 1 ] && ! command -v unpaper > /dev/null && echo "Please install unpaper. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
[ $PREPROCESS_CLEAN -eq 1 ] && ! command -v unpaper > /dev/null && echo "Please install unpaper. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
! command -v tesseract > /dev/null && echo "Please install tesseract and tesseract-data. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
! command -v tesseract > /dev/null && echo "Please install tesseract and tesseract-data. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
! command -v python2 > /dev/null && echo "Please install python v2.x, and the python libraries: reportlab, lxml. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
! command -v python2 > /dev/null && echo "Please install python v2.x. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
|
! python2 -c 'import lxml' 2>/dev/null && echo "Please install the python library lxml. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
|
! python2 -c 'import reportlab' 2>/dev/null && echo "Please install the python library reportlab. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
! command -v gs > /dev/null && echo "Please install ghostcript. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
! command -v gs > /dev/null && echo "Please install ghostcript. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
! command -v java > /dev/null && echo "Please install java. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
! command -v java > /dev/null && echo "Please install java. Exiting..." && exit $EXIT_MISSING_DEPENDENCY
|
||||||
|
|
||||||
|
|
||||||
# ensure tesseract v3.02.02 or newer is installed
|
|
||||||
|
# ensure the right tesseract version is installed
|
||||||
# older versions are known to produce malformed hocr output and should not be used
|
# older versions are known to produce malformed hocr output and should not be used
|
||||||
tessversion=`tesseract -v 2>&1 | grep "tesseract"`
|
# Even 3.02.01 fails in few cases (see issue #28). I decided to allow this version anyway because
|
||||||
if [ $VERBOSITY -ge $LOG_WARN -a $((`echo ${tessversion} | sed s/[^0-9]//g`-30202)) -lt 0 ]; then
|
# 3.02.02 is not yet available for some widespread linux distributions
|
||||||
echo "Warning: Please use tesseract 3.02.02 or newer. Older versions are known to produce invalid hocr output (installed version: ${tessversion})"
|
reqtessversion="3.02.01"
|
||||||
fi
|
tessversion=`tesseract -v 2>&1 | grep "tesseract" | sed s/[^0-9.]//g`
|
||||||
|
tesstooold=$(echo "`echo $tessversion | sed s/[.]//2`-`echo $reqtessversion | sed s/[.]//2` < 0" | bc)
|
||||||
|
[ "$tesstooold" -eq "1" ] \
|
||||||
|
&& echo "Please install tesseract ${reqtessversion} or newer (currently installed version is ${tessversion})" && exit $EXIT_MISSING_DEPENDENCY
|
||||||
|
|
||||||
|
# ensure the right GNU parallel version is installed
|
||||||
|
# older version do not support -q flag (required to escape special characters)
|
||||||
|
reqparallelversion="20130222"
|
||||||
|
parallelversion=`parallel --minversion 0`
|
||||||
|
! parallel --minversion "$reqparallelversion" > /dev/null \
|
||||||
|
&& echo "Please install GNU parallel ${reqparallelversion} or newer (currently installed version is ${parallelversion})" && exit $EXIT_MISSING_DEPENDENCY
|
||||||
|
|
||||||
|
# ensure pdftoppm is provided by poppler-utils, not the older xpdf version
|
||||||
|
! pdftoppm -v 2>&1 | grep -q 'Poppler' && echo "Please remove xpdf and install poppler-utils. Exiting..." && $EXIT_MISSING_DEPENDENCY
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
# Display the version of the tools if log level is LOG_DEBUG
|
# Display the version of the tools if log level is LOG_DEBUG
|
||||||
@@ -171,9 +186,6 @@ if [ $VERBOSITY -ge $LOG_DEBUG ]; then
|
|||||||
pdftoppm -v
|
pdftoppm -v
|
||||||
pdffonts -v
|
pdffonts -v
|
||||||
echo "--------------------------------"
|
echo "--------------------------------"
|
||||||
echo "pdftk version:"
|
|
||||||
pdftk --version
|
|
||||||
echo "--------------------------------"
|
|
||||||
echo "unpaper version:"
|
echo "unpaper version:"
|
||||||
unpaper --version
|
unpaper --version
|
||||||
echo "--------------------------------"
|
echo "--------------------------------"
|
||||||
@@ -193,24 +205,44 @@ fi
|
|||||||
|
|
||||||
|
|
||||||
|
|
||||||
# Initialize path to temporary files
|
# check if the languages passed to tesseract are all supported
|
||||||
today=$(date +"%Y%m%d_%H%M")
|
for currentlan in `echo "$LAN" | sed 's/+/ /g'`; do
|
||||||
fld=$(basename "$FILE_INPUT_PDF" | sed 's/[.][^.]*//')
|
if ! tesseract --list-langs 2>&1 | grep "^$currentlan\$" > /dev/null; then
|
||||||
TMP_FLD="${TMP}/$today.filename.$fld"
|
echo "The language \"$currentlan\" is not supported by tesseract."
|
||||||
|
tesseract --list-langs 2>&1 | tr '\n' ' '; echo
|
||||||
|
echo "Exiting..."
|
||||||
|
exit $EXIT_BAD_ARGS
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
# Initialize path to temporary files using mktemp
|
||||||
|
# Goal: save tmp file in a sub-folder of the $TMPDIR environment variable (or in "/tmp" if unset)
|
||||||
|
# Unfortunately, Linux mktemp is not compatible with FreeBSD/OSX mktemp
|
||||||
|
# Linux version requires no arg
|
||||||
|
# FreeBSD requires '-t prefix' to be used so that $TMPDIR is taken into account
|
||||||
|
# But in Linux '-t template' is handled differently than in FreeBSD
|
||||||
|
# Therefore different calls must be used for Linux and for FreeBSD
|
||||||
|
prefix="$(date +"%Y%m%d_%H%M").filename.$(basename "$FILE_INPUT_PDF" | sed 's/[.][^.]*$//')" # prefix made of date, time and pdf file name without extension
|
||||||
|
TMP_FLD=`mktemp -d 2>/dev/null || mktemp -d -t "${prefix}" 2>/dev/null` # try Linux syntax first, if it fails try FreeBSD/OSX
|
||||||
|
if [ $? -ne 0 ]; then
|
||||||
|
if [ -z "$TMPDIR" ]; then
|
||||||
|
echo "Could not create folder for temporary files. Please ensure you have sufficient right and \"/tmp\" exists"
|
||||||
|
else
|
||||||
|
echo "Could not create folder for temporary files. Please ensure you have sufficient right and \"$TMPDIR\" exists"
|
||||||
|
fi
|
||||||
|
exit $EXIT_FILE_ACCESS_ERROR
|
||||||
|
fi
|
||||||
|
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Created temporary folder: \"$TMP_FLD\""
|
||||||
|
|
||||||
FILE_TMP="${TMP_FLD}/tmp.txt" # temporary file with a very short lifetime (may be used for several things)
|
FILE_TMP="${TMP_FLD}/tmp.txt" # temporary file with a very short lifetime (may be used for several things)
|
||||||
FILE_PAGES_INFO="${TMP_FLD}/pages-info.txt" # for each page: page #; width in pt; height in pt
|
FILE_PAGES_INFO="${TMP_FLD}/pages-info.txt" # for each page: page #; width in pt; height in pt
|
||||||
FILE_OUTPUT_PDF_CAT="${TMP_FLD}/ocred.pdf" # concatenated OCRed PDF files
|
|
||||||
FILE_VALIDATION_LOG="${TMP_FLD}/pdf_validation.log" # log file containing the results of the validation of the PDF/A file
|
FILE_VALIDATION_LOG="${TMP_FLD}/pdf_validation.log" # log file containing the results of the validation of the PDF/A file
|
||||||
|
|
||||||
# Create tmp folder
|
|
||||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Creating temporary folder: \"$TMP_FLD\""
|
|
||||||
rm -r -f "${TMP_FLD}"
|
|
||||||
mkdir -p "${TMP_FLD}"
|
|
||||||
|
|
||||||
|
|
||||||
|
# get the size of each pdf page (width / height) in pt (i.e. inch/72)
|
||||||
|
|
||||||
# get the size of each pdf page (width / height) in pt (inch*72)
|
|
||||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Input file: Extracting size of each page (in pt)"
|
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Input file: Extracting size of each page (in pt)"
|
||||||
! identify -format "%w %h\n" "$FILE_INPUT_PDF" > "$FILE_TMP" \
|
! identify -format "%w %h\n" "$FILE_INPUT_PDF" > "$FILE_TMP" \
|
||||||
&& echo "Could not get size of PDF pages. Exiting..." && exit $EXIT_BAD_INPUT_FILE
|
&& echo "Could not get size of PDF pages. Exiting..." && exit $EXIT_BAD_INPUT_FILE
|
||||||
@@ -219,23 +251,18 @@ sed '/^$/d' "$FILE_TMP" | awk '{printf "%04d %s\n", NR, $0}' > "$FILE_PAGES_INFO
|
|||||||
numpages=`tail -n 1 "$FILE_PAGES_INFO" | cut -f1 -d" "`
|
numpages=`tail -n 1 "$FILE_PAGES_INFO" | cut -f1 -d" "`
|
||||||
|
|
||||||
# process each page of the input pdf file
|
# process each page of the input pdf file
|
||||||
parallel -q -k --halt-on-error 1 "$OCR_PAGE" "$FILE_INPUT_PDF" "{}" "$numpages" "$TMP_FLD" \
|
parallel --gnu -q -k --halt-on-error 1 "$OCR_PAGE" "$FILE_INPUT_PDF" "{}" "$numpages" "$TMP_FLD" \
|
||||||
"$VERBOSITY" "$LAN" "$KEEP_TMP" "$PREPROCESS_DESKEW" "$PREPROCESS_CLEAN" "$PREPROCESS_CLEANTOPDF" "$OVERSAMPLING_DPI" \
|
"$VERBOSITY" "$LAN" "$KEEP_TMP" "$PREPROCESS_DESKEW" "$PREPROCESS_CLEAN" "$PREPROCESS_CLEANTOPDF" "$OVERSAMPLING_DPI" \
|
||||||
"$PDF_NOIMG" "$TESS_CFG_FILES" "$FORCE_OCR" < "$FILE_PAGES_INFO"
|
"$PDF_NOIMG" "$TESS_CFG_FILES" "$FORCE_OCR" < "$FILE_PAGES_INFO"
|
||||||
ret_code="$?"
|
ret_code="$?"
|
||||||
[ $ret_code -ne 0 ] && exit $ret_code
|
[ $ret_code -ne 0 ] && exit $ret_code
|
||||||
|
|
||||||
# concatenate all pages
|
# concatenate all pages and convert the pdf file to match PDF/A format
|
||||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Concatenating all pages"
|
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Concatenating all pages to the final PDF/A file"
|
||||||
! pdftk "${TMP_FLD}/"*ocred*.pdf cat output "$FILE_OUTPUT_PDF_CAT" \
|
|
||||||
&& echo "Could not concatenate individual PDF pages (\"${TMP_FLD}/*-ocred.pdf\") to one file. Exiting..." && exit $EXIT_OTHER_ERROR
|
|
||||||
|
|
||||||
# convert the pdf file to match PDF/A format
|
|
||||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Converting to PDF/A"
|
|
||||||
! gs -dQUIET -dPDFA -dBATCH -dNOPAUSE -dUseCIEColor \
|
! gs -dQUIET -dPDFA -dBATCH -dNOPAUSE -dUseCIEColor \
|
||||||
-sProcessColorModel=DeviceCMYK -sDEVICE=pdfwrite -sPDFACompatibilityPolicy=2 \
|
-sProcessColorModel=DeviceCMYK -sDEVICE=pdfwrite -sPDFACompatibilityPolicy=2 \
|
||||||
-sOutputFile="$FILE_OUTPUT_PDFA" "$FILE_OUTPUT_PDF_CAT" 1> /dev/null 2> /dev/null \
|
-sOutputFile="$FILE_OUTPUT_PDFA" "${TMP_FLD}/"*ocred*.pdf 1> /dev/null 2> /dev/null \
|
||||||
&& echo "Could not convert PDF file \"$FILE_OUTPUT_PDF_CAT\" to PDF/A. Exiting..." && exit $EXIT_OTHER_ERROR
|
&& echo "Could not concatenate all pages to the final PDF/A file. Exiting..." && exit $EXIT_OTHER_ERROR
|
||||||
|
|
||||||
# validate generated pdf file (compliance to PDF/A)
|
# validate generated pdf file (compliance to PDF/A)
|
||||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Checking compliance to PDF/A standard"
|
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Output file: Checking compliance to PDF/A standard"
|
||||||
|
|||||||
@@ -11,12 +11,14 @@ Main features
|
|||||||
- Generates a searchable PDF/A file from a PDF file containing only images
|
- Generates a searchable PDF/A file from a PDF file containing only images
|
||||||
- Places OCRed text accurately below the image to ease copy / paste
|
- Places OCRed text accurately below the image to ease copy / paste
|
||||||
- Keeps the exact resolution of the original embedded images
|
- Keeps the exact resolution of the original embedded images
|
||||||
- or if requested oversample the images before OCRing so as to get better results
|
- or if requested oversamples the images before OCRing so as to get better results
|
||||||
- If requested deskews and / or clean the image before performing OCR
|
- If requested deskews and / or clean the image before performing OCR
|
||||||
- Validates the generated file against the PDF/A specification using jhove
|
- Validates the generated file against the PDF/A specification using jhove
|
||||||
- Provides debug mode to enable easy verification of the OCR results
|
- Provides debug mode to enable easy verification of the OCR results
|
||||||
- Processes several pages in parallel if more than one CPU core is available
|
- Processes several pages in parallel if more than one CPU core is available
|
||||||
|
|
||||||
|
For details: please consult the release notes
|
||||||
|
|
||||||
Motivation
|
Motivation
|
||||||
----------
|
----------
|
||||||
|
|
||||||
@@ -35,13 +37,14 @@ I found many, but none of them were really satisfying.
|
|||||||
Install
|
Install
|
||||||
-------
|
-------
|
||||||
|
|
||||||
Download OCRmyPDF here: https://github.com/fritz-hh/OCRmyPDF/tags
|
Download OCRmyPDF here: https://github.com/fritz-hh/OCRmyPDF/releases
|
||||||
|
|
||||||
Copy the file in onto your linux/unix machine and extract it.
|
Copy the file in onto your linux/unix machine and extract it.
|
||||||
|
|
||||||
Run: "sh ./OCRmyPDF.sh -h" to get the script usage
|
Run: "sh ./OCRmyPDF.sh -h" to get the script usage
|
||||||
|
|
||||||
If not yet installed, the script will notify you about dependencies that need to be installed
|
If not yet installed, the script will notify you about dependencies that need to be installed.
|
||||||
|
The script requires specific versions of the dependencies. Older version than the ones mentioned in the release notes are likely not to be compatible to OCRmyPDF.
|
||||||
|
|
||||||
Support
|
Support
|
||||||
-------
|
-------
|
||||||
@@ -49,7 +52,7 @@ Support
|
|||||||
In case you detect an issue, please:
|
In case you detect an issue, please:
|
||||||
|
|
||||||
- Check if your issue is already known
|
- Check if your issue is already known
|
||||||
- if no problem report exists on github, please create one here: https://github.com/fritz-hh/OCRmyPDF/issues
|
- If no problem report exists on github, please create one here: https://github.com/fritz-hh/OCRmyPDF/issues
|
||||||
- Describe your problem thoroughly
|
- Describe your problem thoroughly
|
||||||
- Append the console output of the script when running the debug mode (-g option)
|
- Append the console output of the script when running the debug mode (-g option)
|
||||||
- If possible provide your input PDF file as well as the content of the temporary folder (using a file sharing service like www.file-upload.net)
|
- If possible provide your input PDF file as well as the content of the temporary folder (using a file sharing service like www.file-upload.net)
|
||||||
@@ -57,4 +60,9 @@ In case you detect an issue, please:
|
|||||||
Press & Media
|
Press & Media
|
||||||
-------------
|
-------------
|
||||||
|
|
||||||
- c't 1-2014, page 59: Detailed presentation of OCRmyPDF v1.0 in the leading german IT magazine (c't)
|
- c't 1-2014, page 59: Detailed presentation of OCRmyPDF v1.0 in the leading german IT magazine c't (http://www.heise.de/ct/inhalt/2014/1/58/)
|
||||||
|
|
||||||
|
Disclaimer
|
||||||
|
----------
|
||||||
|
|
||||||
|
The software is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
|||||||
+77
-1
@@ -5,6 +5,81 @@ Please always read this file before installing the package
|
|||||||
|
|
||||||
Download software here: https://github.com/fritz-hh/OCRmyPDF/tags
|
Download software here: https://github.com/fritz-hh/OCRmyPDF/tags
|
||||||
|
|
||||||
|
v2.0-stable (2014-01-25):
|
||||||
|
=======
|
||||||
|
|
||||||
|
New features
|
||||||
|
------------
|
||||||
|
|
||||||
|
- Check if the language(s) passed using the -l option is supported by tesseract (fixes #60)
|
||||||
|
|
||||||
|
Changes
|
||||||
|
-------
|
||||||
|
|
||||||
|
- Allow OCRmyPDF to be used with tesseract 3.02.01, even though OCR might fail for few PDF file (see issue #28). Rationale: For some linux distribution, no newer version than tesseract 3.02.01 is available
|
||||||
|
|
||||||
|
Fixes
|
||||||
|
-----
|
||||||
|
|
||||||
|
- More robust algorithm for checking the version of the installed tesseract package
|
||||||
|
|
||||||
|
Tested with
|
||||||
|
-----------
|
||||||
|
|
||||||
|
- Operating system: FreeBSD 9.1
|
||||||
|
- Dependencies:
|
||||||
|
- parallel 20130222
|
||||||
|
- poppler-utils 0.22.2
|
||||||
|
- ImageMagick 6.8.0-7 2013-03-30
|
||||||
|
- Unpaper 0.3
|
||||||
|
- tesseract 3.02.02
|
||||||
|
- Python 2.7.3
|
||||||
|
- ghoscript (gs): 9.06
|
||||||
|
- java: openjdk version "1.7.0_17"
|
||||||
|
|
||||||
|
v2.0-rc2 (2014-01-16):
|
||||||
|
=======
|
||||||
|
|
||||||
|
New features
|
||||||
|
------------
|
||||||
|
|
||||||
|
- None
|
||||||
|
|
||||||
|
Changes
|
||||||
|
-------
|
||||||
|
|
||||||
|
- Size reduction of final PDF file: (fixes #50)
|
||||||
|
- Support for monochrome (Black&White) images (massive size reduction in final PDF: >80%)
|
||||||
|
- Reduced size of grayscale images (by 13% on test PDF file)
|
||||||
|
- Preventing fi, fl ligatures does not require anymore to pass an additional config file to tesseract using the -C option (fixes #58)
|
||||||
|
- Location of temporary folder according to content of environment variable TMPDIR.
|
||||||
|
- Dependency to pdftk removed
|
||||||
|
- Check for compatible versions of dependencies: (fixes #51)
|
||||||
|
- parallel and tesseract
|
||||||
|
- python libraries reportlab and lxml
|
||||||
|
|
||||||
|
Fixes
|
||||||
|
-----
|
||||||
|
|
||||||
|
- Improved portability with various shells (dash, bash, tcsh) and OS (FreeBSD, MAC OSX, Linux) (fixes #59)
|
||||||
|
- Corrected bug in case the input PDF file contains a space character (fixes #48)
|
||||||
|
- Prevent spurious error message in case there is no image in a PDF page
|
||||||
|
- Prevent collision of temporary folder names (fixes #57)
|
||||||
|
|
||||||
|
Tested with
|
||||||
|
-----------
|
||||||
|
|
||||||
|
- Operating system: FreeBSD 9.1
|
||||||
|
- Dependencies:
|
||||||
|
- parallel 20130222
|
||||||
|
- poppler-utils 0.22.2
|
||||||
|
- ImageMagick 6.8.0-7 2013-03-30
|
||||||
|
- Unpaper 0.3
|
||||||
|
- tesseract 3.02.02
|
||||||
|
- Python 2.7.3
|
||||||
|
- ghoscript (gs): 9.06
|
||||||
|
- java: openjdk version "1.7.0_17"
|
||||||
|
|
||||||
v2.0-rc1 (2014-01-07):
|
v2.0-rc1 (2014-01-07):
|
||||||
====
|
====
|
||||||
|
|
||||||
@@ -23,7 +98,7 @@ Changes
|
|||||||
-------
|
-------
|
||||||
|
|
||||||
- In debug mode: The debug page is now placed after the respective "normal" page
|
- In debug mode: The debug page is now placed after the respective "normal" page
|
||||||
- Reduced disk space usage if -d (deskew) or -c (cleanup) options are not selected
|
- Reduced disk space usage in temporary folder if -d (deskew) or -c (cleanup) options are not selected
|
||||||
- New file src/config.sh containing various configuration parameters
|
- New file src/config.sh containing various configuration parameters
|
||||||
- Documentation of the tesseract config file "tess-cfg/no_ligature" improved
|
- Documentation of the tesseract config file "tess-cfg/no_ligature" improved
|
||||||
- Improved consistency of the temporary file names
|
- Improved consistency of the temporary file names
|
||||||
@@ -42,6 +117,7 @@ Tested with
|
|||||||
|
|
||||||
- Operating system: FreeBSD 9.1
|
- Operating system: FreeBSD 9.1
|
||||||
- Dependencies:
|
- Dependencies:
|
||||||
|
- parallel 20130222
|
||||||
- poppler-utils 0.22.2
|
- poppler-utils 0.22.2
|
||||||
- ImageMagick 6.8.0-7 2013-03-30
|
- ImageMagick 6.8.0-7 2013-03-30
|
||||||
- Unpaper 0.3
|
- Unpaper 0.3
|
||||||
|
|||||||
+12
-5
@@ -1,11 +1,22 @@
|
|||||||
|
#####################################################################################
|
||||||
|
# The following parameters might be changed by the user
|
||||||
|
#####################################################################################
|
||||||
|
|
||||||
|
DEFAULT_DPI=300 # dpi value used as fall back if the page dpi cannot be determined
|
||||||
|
|
||||||
|
#####################################################################################
|
||||||
|
# Do NOT change the following parameters
|
||||||
|
#####################################################################################
|
||||||
|
|
||||||
TOOLNAME="OCRmyPDF"
|
TOOLNAME="OCRmyPDF"
|
||||||
VERSION="v2.0-rc1"
|
VERSION="v2.0-stable"
|
||||||
|
|
||||||
# possible exit codes
|
# possible exit codes
|
||||||
EXIT_BAD_ARGS="1"
|
EXIT_BAD_ARGS="1"
|
||||||
EXIT_BAD_INPUT_FILE="2"
|
EXIT_BAD_INPUT_FILE="2"
|
||||||
EXIT_MISSING_DEPENDENCY="3"
|
EXIT_MISSING_DEPENDENCY="3"
|
||||||
EXIT_INVALID_OUPUT_PDFA="4"
|
EXIT_INVALID_OUPUT_PDFA="4"
|
||||||
|
EXIT_FILE_ACCESS_ERROR="5"
|
||||||
EXIT_OTHER_ERROR="15"
|
EXIT_OTHER_ERROR="15"
|
||||||
|
|
||||||
# possible log levels
|
# possible log levels
|
||||||
@@ -16,10 +27,6 @@ LOG_DEBUG="3" # debug level logging
|
|||||||
|
|
||||||
# various paths
|
# various paths
|
||||||
SRC="./src" # location of the source folder (except source of external tools like jhove)
|
SRC="./src" # location of the source folder (except source of external tools like jhove)
|
||||||
TMP="./tmp" # location of the temporary files (one sub-folder will be created per PDF file to be processed)
|
|
||||||
OCR_PAGE="$SRC/ocrPage.sh" # path to the script aimed at OCRing one page
|
OCR_PAGE="$SRC/ocrPage.sh" # path to the script aimed at OCRing one page
|
||||||
JHOVE="./jhove/bin/JhoveApp.jar" # java SW for validating the final PDF/A
|
JHOVE="./jhove/bin/JhoveApp.jar" # java SW for validating the final PDF/A
|
||||||
JHOVE_CFG="./jhove/conf/jhove.conf" # location of the jhove config file
|
JHOVE_CFG="./jhove/conf/jhove.conf" # location of the jhove config file
|
||||||
|
|
||||||
# other
|
|
||||||
DEFAULT_DPI=300 # dpi value used as fall back if the page dpi cannot be determined
|
|
||||||
Regular → Executable
+121
-21
@@ -1,4 +1,5 @@
|
|||||||
#!/usr/bin/python
|
#!/usr/local/bin/python2
|
||||||
|
# coding: utf-8
|
||||||
##############################################################################
|
##############################################################################
|
||||||
# Copyright (c) 2013-14: fritz-hh from Github (https://github.com/fritz-hh)
|
# Copyright (c) 2013-14: fritz-hh from Github (https://github.com/fritz-hh)
|
||||||
#
|
#
|
||||||
@@ -6,12 +7,95 @@
|
|||||||
# Initial version by Jonathan Brinley, jonathanbrinley@gmail.com
|
# Initial version by Jonathan Brinley, jonathanbrinley@gmail.com
|
||||||
##############################################################################
|
##############################################################################
|
||||||
from reportlab.pdfgen.canvas import Canvas
|
from reportlab.pdfgen.canvas import Canvas
|
||||||
|
from reportlab.pdfgen.pdfimages import PDFImage
|
||||||
from reportlab.lib.units import inch
|
from reportlab.lib.units import inch
|
||||||
from lxml import etree as ElementTree
|
from lxml import etree as ElementTree
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
import re, sys
|
import re, sys
|
||||||
import argparse
|
import argparse
|
||||||
|
|
||||||
|
|
||||||
|
def monkeypatch_method(cls):
|
||||||
|
'''
|
||||||
|
Override a class method at runtime.
|
||||||
|
|
||||||
|
Rationale:
|
||||||
|
https://mail.python.org/pipermail/python-dev/2008-January/076194.html
|
||||||
|
'''
|
||||||
|
def decorator(func):
|
||||||
|
setattr(cls, func.__name__, func)
|
||||||
|
return func
|
||||||
|
return decorator
|
||||||
|
|
||||||
|
|
||||||
|
@monkeypatch_method(PDFImage)
|
||||||
|
def PIL_imagedata(self):
|
||||||
|
'''
|
||||||
|
Add ability to output greyscale and 1-bit PIL images without conversion to RGB.
|
||||||
|
|
||||||
|
The upstream Python 2.7 version of reportlab converts 1-bit PIL images to RGB
|
||||||
|
instead of saving them in a lower BPP format. They have since added the following
|
||||||
|
fix to their Python 3.3 branch, but it has not been back-ported.
|
||||||
|
|
||||||
|
https://bitbucket.org/rptlab/reportlab/commits/177ddcbe4df6f9b461dac62612df9b8da3966a5d
|
||||||
|
'''
|
||||||
|
image = self.image
|
||||||
|
if image.format == 'JPEG':
|
||||||
|
fp = image.fp
|
||||||
|
fp.seek(0)
|
||||||
|
return self._jpg_imagedata(fp)
|
||||||
|
|
||||||
|
from reportlab.lib.utils import import_zlib
|
||||||
|
from reportlab import rl_config
|
||||||
|
from reportlab.pdfbase.pdfutils import _AsciiBase85Encode, _chunker
|
||||||
|
|
||||||
|
self.source = 'PIL'
|
||||||
|
zlib = import_zlib()
|
||||||
|
if not zlib:
|
||||||
|
return
|
||||||
|
|
||||||
|
bpc = 8
|
||||||
|
# Use the colorSpace in the image
|
||||||
|
if image.mode == 'CMYK':
|
||||||
|
myimage = image
|
||||||
|
colorSpace = 'DeviceCMYK'
|
||||||
|
bpp = 4
|
||||||
|
elif image.mode == '1':
|
||||||
|
myimage = image
|
||||||
|
colorSpace = 'DeviceGray'
|
||||||
|
bpp = 1
|
||||||
|
bpc = 1
|
||||||
|
elif image.mode == 'L':
|
||||||
|
myimage = image
|
||||||
|
colorSpace = 'DeviceGray'
|
||||||
|
bpp = 1
|
||||||
|
else:
|
||||||
|
myimage = image.convert('RGB')
|
||||||
|
colorSpace = 'RGB'
|
||||||
|
bpp = 3
|
||||||
|
imgwidth, imgheight = myimage.size
|
||||||
|
|
||||||
|
# this describes what is in the image itself
|
||||||
|
# *NB* according to the spec you can only use the short form in inline images
|
||||||
|
|
||||||
|
imagedata = ['BI /W %d /H %d /BPC %d /CS /%s /F [%s/Fl] ID' %
|
||||||
|
(imgwidth, imgheight, bpc, colorSpace, rl_config.useA85 and '/A85 ' or '')]
|
||||||
|
|
||||||
|
# use a flate filter and, optionally, Ascii Base 85 to compress
|
||||||
|
raw = myimage.tostring()
|
||||||
|
rowstride = (imgwidth * bpc * bpp + 7) / 8
|
||||||
|
assert len(raw) == rowstride * imgheight, "Wrong amount of data for image"
|
||||||
|
data = zlib.compress(raw) # this bit is very fast...
|
||||||
|
|
||||||
|
if rl_config.useA85:
|
||||||
|
# ...sadly this may not be
|
||||||
|
data = _AsciiBase85Encode(data)
|
||||||
|
# append in blocks of 60 characters
|
||||||
|
_chunker(data, imagedata)
|
||||||
|
imagedata.append('EI')
|
||||||
|
return (imagedata, imgwidth, imgheight)
|
||||||
|
|
||||||
|
|
||||||
class hocrTransform():
|
class hocrTransform():
|
||||||
"""
|
"""
|
||||||
A class for converting documents from the hOCR format.
|
A class for converting documents from the hOCR format.
|
||||||
@@ -24,20 +108,21 @@ class hocrTransform():
|
|||||||
|
|
||||||
self.hocr = ElementTree.ElementTree()
|
self.hocr = ElementTree.ElementTree()
|
||||||
self.hocr.parse(hocrFileName)
|
self.hocr.parse(hocrFileName)
|
||||||
|
|
||||||
# if the hOCR file has a namespace, ElementTree requires its use to find elements
|
# if the hOCR file has a namespace, ElementTree requires its use to find elements
|
||||||
matches = re.match('({.*})html', self.hocr.getroot().tag)
|
matches = re.match('({.*})html', self.hocr.getroot().tag)
|
||||||
self.xmlns = ''
|
self.xmlns = ''
|
||||||
if matches:
|
if matches:
|
||||||
self.xmlns = matches.group(1)
|
self.xmlns = matches.group(1)
|
||||||
|
|
||||||
# get dimension in pt (not pixel!!!!) of the OCRed image
|
# get dimension in pt (not pixel!!!!) of the OCRed image
|
||||||
|
self.width, self.height = None, None
|
||||||
for div in self.hocr.findall(".//%sdiv[@class='ocr_page']"%(self.xmlns)):
|
for div in self.hocr.findall(".//%sdiv[@class='ocr_page']"%(self.xmlns)):
|
||||||
coords = self.element_coordinates(div)
|
coords = self.element_coordinates(div)
|
||||||
self.width = self.px2pt(coords[2]-coords[0])
|
self.width = self.px2pt(coords[2]-coords[0])
|
||||||
self.height = self.px2pt(coords[3]-coords[1])
|
self.height = self.px2pt(coords[3]-coords[1])
|
||||||
break # there shouldn't be more than one, and if there is, we don't want it
|
break # there shouldn't be more than one, and if there is, we don't want it
|
||||||
|
|
||||||
# no width and heigh definition in the ocr_image element of the hocr file
|
# no width and heigh definition in the ocr_image element of the hocr file
|
||||||
if self.width is None:
|
if self.width is None:
|
||||||
print("No page dimension found in the hocr file")
|
print("No page dimension found in the hocr file")
|
||||||
@@ -54,7 +139,7 @@ class hocrTransform():
|
|||||||
return self._get_element_text(body).encode('utf-8') # XML gives unicode
|
return self._get_element_text(body).encode('utf-8') # XML gives unicode
|
||||||
else:
|
else:
|
||||||
return ''
|
return ''
|
||||||
|
|
||||||
def _get_element_text(self, element):
|
def _get_element_text(self, element):
|
||||||
"""
|
"""
|
||||||
Return the textual content of the element and its children
|
Return the textual content of the element and its children
|
||||||
@@ -67,7 +152,7 @@ class hocrTransform():
|
|||||||
if element.tail is not None:
|
if element.tail is not None:
|
||||||
text = text + element.tail
|
text = text + element.tail
|
||||||
return text
|
return text
|
||||||
|
|
||||||
def element_coordinates(self, element):
|
def element_coordinates(self, element):
|
||||||
"""
|
"""
|
||||||
Returns a tuple containing the coordinates of the bounding box around
|
Returns a tuple containing the coordinates of the bounding box around
|
||||||
@@ -80,13 +165,24 @@ class hocrTransform():
|
|||||||
coords = matches.group(1).split()
|
coords = matches.group(1).split()
|
||||||
out = (int(coords[0]),int(coords[1]),int(coords[2]),int(coords[3]))
|
out = (int(coords[0]),int(coords[1]),int(coords[2]),int(coords[3]))
|
||||||
return out
|
return out
|
||||||
|
|
||||||
def px2pt(self, pxl):
|
def px2pt(self, pxl):
|
||||||
"""
|
"""
|
||||||
Returns the length in pt given length in pxl
|
Returns the length in pt given length in pxl
|
||||||
"""
|
"""
|
||||||
return float(pxl)/self.dpi*inch
|
return float(pxl)/self.dpi*inch
|
||||||
|
|
||||||
|
def replace_unsupported_chars(self, str):
|
||||||
|
"""
|
||||||
|
Given an input string, returns the corresponding string that:
|
||||||
|
- is available in the helvetica facetype
|
||||||
|
- does not contain any ligature (to allow easy search in the PDF file)
|
||||||
|
"""
|
||||||
|
# The 'u' before the character to replace indicates that it is a unicode character
|
||||||
|
str=str.replace(u"fl","fl")
|
||||||
|
str=str.replace(u"fi","fi")
|
||||||
|
return str
|
||||||
|
|
||||||
def to_pdf(self, outFileName, imageFileName, showBoundingboxes, fontname="Helvetica"):
|
def to_pdf(self, outFileName, imageFileName, showBoundingboxes, fontname="Helvetica"):
|
||||||
"""
|
"""
|
||||||
Creates a PDF file with an image superimposed on top of the text.
|
Creates a PDF file with an image superimposed on top of the text.
|
||||||
@@ -97,13 +193,13 @@ class hocrTransform():
|
|||||||
"""
|
"""
|
||||||
# create the PDF file
|
# create the PDF file
|
||||||
pdf = Canvas(outFileName, pagesize=(self.width, self.height), pageCompression=1) # page size in points (1/72 in.)
|
pdf = Canvas(outFileName, pagesize=(self.width, self.height), pageCompression=1) # page size in points (1/72 in.)
|
||||||
|
|
||||||
# draw bounding box for each paragraph
|
# draw bounding box for each paragraph
|
||||||
pdf.setStrokeColorRGB(0,1,1) # light blue for bounding box of paragraph
|
pdf.setStrokeColorRGB(0,1,1) # light blue for bounding box of paragraph
|
||||||
pdf.setFillColorRGB(0,1,1) # light blue for bounding box of paragraph
|
pdf.setFillColorRGB(0,1,1) # light blue for bounding box of paragraph
|
||||||
pdf.setLineWidth(0) # no line for bounding box
|
pdf.setLineWidth(0) # no line for bounding box
|
||||||
for elem in self.hocr.findall(".//%sp[@class='%s']" % (self.xmlns, "ocr_par")):
|
for elem in self.hocr.findall(".//%sp[@class='%s']" % (self.xmlns, "ocr_par")):
|
||||||
|
|
||||||
elemtxt=self._get_element_text(elem).rstrip()
|
elemtxt=self._get_element_text(elem).rstrip()
|
||||||
if len(elemtxt) == 0:
|
if len(elemtxt) == 0:
|
||||||
continue
|
continue
|
||||||
@@ -113,12 +209,12 @@ class hocrTransform():
|
|||||||
y1=self.px2pt(coords[1])
|
y1=self.px2pt(coords[1])
|
||||||
x2=self.px2pt(coords[2])
|
x2=self.px2pt(coords[2])
|
||||||
y2=self.px2pt(coords[3])
|
y2=self.px2pt(coords[3])
|
||||||
|
|
||||||
# draw the bbox border
|
# draw the bbox border
|
||||||
if showBoundingboxes == True:
|
if showBoundingboxes == True:
|
||||||
pdf.rect(x1, self.height-y2, x2-x1, y2-y1, fill=1)
|
pdf.rect(x1, self.height-y2, x2-x1, y2-y1, fill=1)
|
||||||
|
|
||||||
|
|
||||||
# check if element with class 'ocrx_word' are available
|
# check if element with class 'ocrx_word' are available
|
||||||
# otherwise use 'ocr_line' as fallback
|
# otherwise use 'ocr_line' as fallback
|
||||||
elemclass="ocr_line"
|
elemclass="ocr_line"
|
||||||
@@ -133,6 +229,9 @@ class hocrTransform():
|
|||||||
for elem in self.hocr.findall(".//%sspan[@class='%s']" % (self.xmlns, elemclass)):
|
for elem in self.hocr.findall(".//%sspan[@class='%s']" % (self.xmlns, elemclass)):
|
||||||
|
|
||||||
elemtxt=self._get_element_text(elem).rstrip()
|
elemtxt=self._get_element_text(elem).rstrip()
|
||||||
|
|
||||||
|
elemtxt=self.replace_unsupported_chars(elemtxt)
|
||||||
|
|
||||||
if len(elemtxt) == 0:
|
if len(elemtxt) == 0:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
@@ -141,7 +240,7 @@ class hocrTransform():
|
|||||||
y1=self.px2pt(coords[1])
|
y1=self.px2pt(coords[1])
|
||||||
x2=self.px2pt(coords[2])
|
x2=self.px2pt(coords[2])
|
||||||
y2=self.px2pt(coords[3])
|
y2=self.px2pt(coords[3])
|
||||||
|
|
||||||
# draw the bbox border
|
# draw the bbox border
|
||||||
if showBoundingboxes == True:
|
if showBoundingboxes == True:
|
||||||
pdf.rect(x1, self.height-y2, x2-x1, y2-y1, fill=0)
|
pdf.rect(x1, self.height-y2, x2-x1, y2-y1, fill=0)
|
||||||
@@ -152,7 +251,7 @@ class hocrTransform():
|
|||||||
|
|
||||||
# set cursor to bottom left corner of bbox (adjust for dpi)
|
# set cursor to bottom left corner of bbox (adjust for dpi)
|
||||||
text.setTextOrigin(x1, self.height-y2)
|
text.setTextOrigin(x1, self.height-y2)
|
||||||
|
|
||||||
# scale the width of the text to fill the width of the bbox
|
# scale the width of the text to fill the width of the bbox
|
||||||
text.setHorizScale(100*(x2-x1)/pdf.stringWidth(elemtxt, fontname, fontsize))
|
text.setHorizScale(100*(x2-x1)/pdf.stringWidth(elemtxt, fontname, fontsize))
|
||||||
|
|
||||||
@@ -162,13 +261,14 @@ class hocrTransform():
|
|||||||
|
|
||||||
# put the image on the page, scaled to fill the page
|
# put the image on the page, scaled to fill the page
|
||||||
if imageFileName != None:
|
if imageFileName != None:
|
||||||
im = Image.open(imageFileName)
|
im = Image.open(imageFileName)
|
||||||
pdf.drawInlineImage(im, 0, 0, width=self.width, height=self.height)
|
pdf.drawInlineImage(im, 0, 0, width=self.width, height=self.height)
|
||||||
|
|
||||||
# finish up the page and save it
|
# finish up the page and save it
|
||||||
pdf.showPage()
|
pdf.showPage()
|
||||||
pdf.save()
|
pdf.save()
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
parser = argparse.ArgumentParser(description='Convert hocr file to PDF')
|
parser = argparse.ArgumentParser(description='Convert hocr file to PDF')
|
||||||
parser.add_argument('-b', '--boundingboxes', action="store_true", default=False, help='Show bounding boxes borders')
|
parser.add_argument('-b', '--boundingboxes', action="store_true", default=False, help='Show bounding boxes borders')
|
||||||
@@ -181,5 +281,5 @@ if __name__ == "__main__":
|
|||||||
hocr = hocrTransform(args.hocrfile, args.resolution)
|
hocr = hocrTransform(args.hocrfile, args.resolution)
|
||||||
hocr.to_pdf(args.outputfile, args.image, args.boundingboxes)
|
hocr.to_pdf(args.outputfile, args.image, args.boundingboxes)
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Regular → Executable
+26
-18
@@ -37,13 +37,14 @@ FORCE_OCR="${14}" # Force to OCR, even if the page already contains fonts
|
|||||||
# Output: A file containing the characteristics of the embedded image. File structure:
|
# Output: A file containing the characteristics of the embedded image. File structure:
|
||||||
# DPI=<dpi>
|
# DPI=<dpi>
|
||||||
# COLOR_SPACE=<colorspace>
|
# COLOR_SPACE=<colorspace>
|
||||||
|
# DEPTH=<colordepth>
|
||||||
# Returns:
|
# Returns:
|
||||||
# - 0: if no error occurs
|
# - 0: if no error occurs
|
||||||
# - 1: in case the page already contains fonts (which should be the case for PDF generated from scanned pages)
|
# - 1: in case the page already contains fonts (which should be the case for PDF generated from scanned pages)
|
||||||
# - 2: in case the page contains more than one image
|
# - 2: in case the page contains more than one image
|
||||||
##################################
|
##################################
|
||||||
getImgInfo() {
|
getImgInfo() {
|
||||||
local page widthPDF heightPDF curImgInfo nbImg curImg propCurImg widthCurImg heightCurImg colorspaceCurImg dpi
|
local page widthPDF heightPDF curImgInfo nbImg curImg propCurImg widthCurImg heightCurImg colorspaceCurImg depthCurImg dpi
|
||||||
|
|
||||||
# page number
|
# page number
|
||||||
page="$1"
|
page="$1"
|
||||||
@@ -58,25 +59,26 @@ getImgInfo() {
|
|||||||
|
|
||||||
|
|
||||||
# check if the page already contains fonts (which should not be the case for PDF based on scanned files
|
# check if the page already contains fonts (which should not be the case for PDF based on scanned files
|
||||||
[ `pdffonts -f $page -l $page ${FILE_INPUT_PDF} | wc -l` -gt 2 ] && echo "Page $page: Page already contains font data !!!" && return 1
|
[ `pdffonts -f $page -l $page "${FILE_INPUT_PDF}" | wc -l` -gt 2 ] && echo "Page $page: Page already contains font data !!!" && return 1
|
||||||
|
|
||||||
|
|
||||||
# extract raw image from pdf file to compute resolution
|
# extract raw image from pdf file to compute resolution
|
||||||
# unfortunately this image can have another orientation than in the pdf...
|
# unfortunately this image can have another orientation than in the pdf...
|
||||||
# so we will have to extract it again later using pdftoppm
|
# so we will have to extract it again later using pdftoppm
|
||||||
pdfimages -f $page -l $page -j "$FILE_INPUT_PDF" "$curOrigImg" 1>&2
|
pdfimages -f $page -l $page -j "$FILE_INPUT_PDF" "$curOrigImg" 1>&2
|
||||||
# count number of extracted images
|
# count number of extracted images
|
||||||
nbImg=`ls -1 "$curOrigImg"* | wc -l`
|
nbImg=$((`ls -1 "$curOrigImg"* 2>/dev/null | wc -l`))
|
||||||
if [ $nbImg -ne "1" ]; then
|
if [ $nbImg -ne "1" ]; then
|
||||||
[ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Expecting exactly 1 image covering the whole page (found $nbImg). Cannot compute dpi value."
|
[ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Expecting exactly 1 image covering the whole page (found $nbImg). Cannot compute dpi value."
|
||||||
return 2
|
return 2
|
||||||
fi
|
fi
|
||||||
# Get characteristics of the extracted image
|
# Get characteristics of the extracted image
|
||||||
curImg=`ls -1 "$curOrigImg"*`
|
curImg=`ls -1 "$curOrigImg"* 2>/dev/null`
|
||||||
propCurImg=`identify -format "%w %h %[colorspace]" "$curImg"`
|
propCurImg=`identify -format "%w %h %[colorspace] %[depth]" "$curImg"`
|
||||||
widthCurImg=`echo "$propCurImg" | cut -f1 -d" "`
|
widthCurImg=`echo "$propCurImg" | cut -f1 -d" "`
|
||||||
heightCurImg=`echo "$propCurImg" | cut -f2 -d" "`
|
heightCurImg=`echo "$propCurImg" | cut -f2 -d" "`
|
||||||
colorspaceCurImg=`echo "$propCurImg" | cut -f3 -d" "`
|
colorspaceCurImg=`echo "$propCurImg" | cut -f3 -d" "`
|
||||||
|
depthCurImg=`echo "$propCurImg" | cut -f4 -d" "`
|
||||||
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Size ${heightCurImg}x${widthCurImg} (in pixel)"
|
[ $VERBOSITY -ge $LOG_DEBUG ] && echo "Page $page: Size ${heightCurImg}x${widthCurImg} (in pixel)"
|
||||||
|
|
||||||
# compute the resolution of the image (making the assumption that x & y resolution are equal)
|
# compute the resolution of the image (making the assumption that x & y resolution are equal)
|
||||||
@@ -87,6 +89,7 @@ getImgInfo() {
|
|||||||
# save the image characteristics
|
# save the image characteristics
|
||||||
echo "DPI=$dpi" > "$curImgInfo"
|
echo "DPI=$dpi" > "$curImgInfo"
|
||||||
echo "COLOR_SPACE=$colorspaceCurImg" >> "$curImgInfo"
|
echo "COLOR_SPACE=$colorspaceCurImg" >> "$curImgInfo"
|
||||||
|
echo "DEPTH=$depthCurImg" >> "$curImgInfo"
|
||||||
|
|
||||||
return 0
|
return 0
|
||||||
}
|
}
|
||||||
@@ -109,24 +112,26 @@ curImgInfo="$TMP_FLD/${page}.orig-img-info.txt" # Detected characteristics of
|
|||||||
|
|
||||||
|
|
||||||
# auto-detect the characteristics of the embedded image
|
# auto-detect the characteristics of the embedded image
|
||||||
|
depthCurImg="8" # default color depth
|
||||||
|
colorspaceCurImg="sRGB" # default color space
|
||||||
|
dpi=$DEFAULT_DPI # default resolution
|
||||||
|
|
||||||
getImgInfo "$page" "$widthPDF" "$heightPDF" "$curImgInfo"
|
getImgInfo "$page" "$widthPDF" "$heightPDF" "$curImgInfo"
|
||||||
ret_code="$?"
|
ret_code="$?"
|
||||||
|
|
||||||
# in case the page contains text do not OCR, unless the FORCE_OCR flag is set
|
# in case the page contains text do not OCR, unless the FORCE_OCR flag is set
|
||||||
if [ "$ret_code" -eq "1" -a "$FORCE_OCR" -eq "0" ]; then
|
if ([ "$ret_code" -eq "1" ] && [ "$FORCE_OCR" -eq "0" ]); then
|
||||||
echo "Page $page: Exiting... (Use the -f option to force OCRing, even though fonts are available in the input file)" && exit $EXIT_BAD_INPUT_FILE
|
echo "Page $page: Exiting... (Use the -f option to force OCRing, even though fonts are available in the input file)" && exit $EXIT_BAD_INPUT_FILE
|
||||||
elif [ "$ret_code" -eq "1" -a "$FORCE_OCR" -eq "1" ]; then
|
elif ([ "$ret_code" -eq "1" ] && [ "$FORCE_OCR" -eq "1" ]); then
|
||||||
colorspaceCurImg="sRGB"
|
|
||||||
dpi=$DEFAULT_DPI
|
|
||||||
[ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: OCRing anyway, assuming a default resolution of $dpi dpi"
|
[ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: OCRing anyway, assuming a default resolution of $dpi dpi"
|
||||||
# in case the page contains more than one image, warn the user but go on with default parameters
|
# in case the page contains more than one image, warn the user but go on with default parameters
|
||||||
elif [ "$ret_code" -eq "2" ]; then
|
elif [ "$ret_code" -eq "2" ]; then
|
||||||
colorspaceCurImg="sRGB"
|
|
||||||
dpi=$DEFAULT_DPI
|
|
||||||
[ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Continuing anyway, assuming a default resolution of $dpi dpi"
|
[ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Continuing anyway, assuming a default resolution of $dpi dpi"
|
||||||
else
|
else
|
||||||
# read the image characteristics from the file
|
# read the image characteristics from the file
|
||||||
dpi=`cat "$curImgInfo" | grep "^DPI=" | cut -f2 -d"="`
|
dpi=`cat "$curImgInfo" | grep "^DPI=" | cut -f2 -d"="`
|
||||||
colorspaceCurImg=`cat "$curImgInfo" | grep "^COLOR_SPACE=" | cut -f2 -d"="`
|
colorspaceCurImg=`cat "$curImgInfo" | grep "^COLOR_SPACE=" | cut -f2 -d"="`
|
||||||
|
depthCurImg=`cat "$curImgInfo" | grep "^DEPTH=" | cut -f2 -d"="`
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# perform oversampling if the resolution is not sufficient to get good OCR results
|
# perform oversampling if the resolution is not sufficient to get good OCR results
|
||||||
@@ -137,10 +142,13 @@ elif [ "$dpi" -lt "200" ]; then
|
|||||||
[ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Low image resolution detected ($dpi dpi). If needed, please use the \"-o\" to try to get better OCR results."
|
[ $VERBOSITY -ge $LOG_WARN ] && echo "Page $page: Low image resolution detected ($dpi dpi). If needed, please use the \"-o\" to try to get better OCR results."
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# Identify if page image should be saved as ppm (color) or pgm (gray)
|
# Identify if page image should be saved as ppm (color), pgm (gray) or pbm (b&w)
|
||||||
ext="ppm"
|
ext="ppm" # by default (color image) the extension of the extracted image is ppm
|
||||||
opt=""
|
opt="" # by default (color image) no option as to be passed to pdftoppm
|
||||||
if [ "$colorspaceCurImg" = "Gray" ]; then
|
if [ "$colorspaceCurImg" = "Gray" ] && [ "$depthCurImg" = "1" ]; then # if monochrome (b&w)
|
||||||
|
ext="pbm"
|
||||||
|
opt="-mono"
|
||||||
|
elif [ "$colorspaceCurImg" = "Gray" ]; then # if gray
|
||||||
ext="pgm"
|
ext="pgm"
|
||||||
opt="-gray"
|
opt="-gray"
|
||||||
fi
|
fi
|
||||||
@@ -201,7 +209,7 @@ fi
|
|||||||
# delete temporary files created for the current page
|
# delete temporary files created for the current page
|
||||||
# to avoid using to much disk space in case of PDF files having many pages
|
# to avoid using to much disk space in case of PDF files having many pages
|
||||||
if [ $KEEP_TMP -eq 0 ]; then
|
if [ $KEEP_TMP -eq 0 ]; then
|
||||||
rm -f "$curOrigImg"*.*
|
rm -f "$curOrigImg"*
|
||||||
rm -f "$curHocr"
|
rm -f "$curHocr"
|
||||||
rm -f "$curImgPixmap"
|
rm -f "$curImgPixmap"
|
||||||
rm -f "$curImgPixmapDeskewed"
|
rm -f "$curImgPixmapDeskewed"
|
||||||
@@ -209,4 +217,4 @@ if [ $KEEP_TMP -eq 0 ]; then
|
|||||||
rm -f "$curImgInfo"
|
rm -f "$curImgInfo"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
exit 0
|
exit 0
|
||||||
|
|||||||
@@ -1,12 +0,0 @@
|
|||||||
##############################################################################
|
|
||||||
# Readme
|
|
||||||
#
|
|
||||||
# Copyright (c) 2013-14: fritz-hh from Github (https://github.com/fritz-hh)
|
|
||||||
##############################################################################
|
|
||||||
|
|
||||||
The file(s) located in this folder are tesseract configuration files.
|
|
||||||
(Tesseract configuration files enable to tune the behaviour of tesseract)
|
|
||||||
|
|
||||||
If needed, these files should be copied to the "tessdata/configs" folder of your tesseract installation.
|
|
||||||
|
|
||||||
To request OCRmyPDF.sh to use a configuration file, please use the -C option
|
|
||||||
@@ -1,9 +0,0 @@
|
|||||||
##############################################################################
|
|
||||||
# tesseract config file provided for OCRmyPDF
|
|
||||||
#
|
|
||||||
# prevents tesseract to detect ligatures, as ligatures are not displayed correctly in the final PDF file
|
|
||||||
# but are replaced by a black square
|
|
||||||
#
|
|
||||||
# Copyright (c) 2013-14: fritz-hh from Github (https://github.com/fritz-hh)
|
|
||||||
##############################################################################
|
|
||||||
tessedit_char_blacklist fifl
|
|
||||||
Reference in New Issue
Block a user