commit 03932e877a6e2a30eaa132be2cec71d6a1fd6397 Author: FPille Date: Mon Jul 29 20:56:31 2019 +0200 Initial commit diff --git a/.travis.yml b/.travis.yml new file mode 100644 index 00000000..d420e8c8 --- /dev/null +++ b/.travis.yml @@ -0,0 +1,26 @@ + +language: generic +sudo: required +dist: trusty + +before_install: + - sudo add-apt-repository ppa:heyarje/libav-11 -y + - sudo add-apt-repository ppa:alex-p/tesseract-ocr -y + - sudo apt-get update -qq + - sudo apt-get install libleptonica-dev -y # required to build jbig2enc + - sudo apt-get install zlib1g-dev -y # required to build jbig2enc + - sudo apt-get install imagemagick -y # required to convert logo to desktop icon + +script: + - export OCRMYPDF_VERSION=8.3.2 + - bash build-appimage.sh + - bash test/test-appimage.sh + - wget https://github.com/probonopd/uploadtool/raw/master/upload.sh + +after_success: + - bash upload.sh OCRmyPDF*.AppImage + +branches: + except: + - # Do not build tags that we create when we upload to GitHub Releases + - /^(?i:continuous)/ diff --git a/README.md b/README.md new file mode 100644 index 00000000..bba097e4 --- /dev/null +++ b/README.md @@ -0,0 +1,40 @@ +# OCRmyPDF-AppImage [![Build Status](https://travis-ci.com/FPille/OCRmyPDF-AppImage.svg?branch=master)](https://travis-ci.com/FPille/OCRmyPDF-AppImage) +[AppImage][APPIMAGE] for [OCRmyPDF][OCRMYPDF] + +## Usage +Download OCRmyPDF*.AppImage, make it executable and run it. +``` +wget https://github.com/FPille/OCRmyPDF-AppImage/releases/download/continuous/OCRmyPDF-8.3.2-x86_64.AppImage +chmod +x OCRmyPDF*.AppImage +./OCRmyPDF*.AppImage --help +``` + + Beside OCRmyPDF additional command line programs can be run with this AppImage like: +* img2pdf +* ghostscript +* pngquant +* python3.6 +* qpdf +* tesseract +* unpaper + +Just use the program name as first parameter plus options: +``` +./OCRmyPDF*.AppImage tesseract -v +tesseract 4.1.0 + leptonica-1.76.0 + libjpeg 8d (libjpeg-turbo 1.3.0) : libpng 1.2.50 : libtiff 4.0.3 : zlib 1.2.11 : libwebp 0.4.0 : libopenjp2 2.3.0 + Found AVX2 + Found AVX + Found SSE +``` +Or create a symlink for the corresponding program: +``` +ln -s OCRmyPDF*.AppImage tesseract +./tesseract --list-langs +``` + + +[APPIMAGE]: https://appimage.org +[OCRMYPDF]: https://github.com/jbarlow83/OCRmyPDF + diff --git a/appimage/AppRun.sh b/appimage/AppRun.sh new file mode 100755 index 00000000..b04bcd76 --- /dev/null +++ b/appimage/AppRun.sh @@ -0,0 +1,47 @@ +#! /bin/bash + +# TODO: Add "usage" ouput and additional functions like displaying of licences +# and man pages +# could be in a similar manner as the AppRun script of qpdf AppImage +# refer to: https://github.com/qpdf/qpdf/blob/master/appimage/AppRun + +HERE="$(dirname "$(readlink -f "${0}")")" + +export PATH="$HERE/usr/bin:$HERE/usr/local/bin:$HERE/usr/python/bin:$PATH" +export LD_PRELOAD="$HERE/usr/lib/liblept.so.5" +export LD_LIBRARY_PATH="$HERE/usr/lib:$HERE/usr/lib/x86_64-linux-gnu:$LD_LIBRARY_PATH" +export TESSDATA_PREFIX="$HERE/usr/share/tesseract-ocr/4.00/tessdata" +export GS_LIB="$HERE/usr/share/ghostscript/9.26/lib:$HERE/usr/share/ghostscript/9.26/Resource:$HERE/usr/share/ghostscript/9.26/Resource/Init" + +# Allow the AppImage to be symlinked to e.g., /usr/bin/commandname +# or called with ./Some*.AppImage commandname ... +# refer to https://github.com/AppImage/AppImageKit/wiki/Bundling-command-line-tools + +if [ ! -z $APPIMAGE ] ; then + BINARY_NAME=$(basename "$ARGV0") +else + BINARY_NAME=$(basename "$0") + export APPDIR="$HERE" # required for the wrapper scripts of linuxdeploy-plugin-python +fi + +if [ ! -z "$1" ] && [ -e "$HERE/bin/$1" ] ; then + MAIN="$HERE/bin/$1" ; shift +elif [ ! -z "$1" ] && [ -e "$HERE/usr/bin/$1" ] ; then + MAIN="$HERE/usr/bin/$1" ; shift +elif [ ! -z "$1" ] && [ -e "$HERE/usr/python/bin/$1" ] ; then + MAIN="$HERE/usr/python/bin/$1" ; shift +elif [ ! -z "$1" ] && [ -e "$HERE/usr/local/bin/$1" ] ; then + MAIN="$HERE/usr/local/bin/$1" ; shift +elif [ -e "$HERE/bin/$BINARY_NAME" ] ; then + MAIN="$HERE/bin/$BINARY_NAME" +elif [ -e "$HERE/usr/bin/$BINARY_NAME" ] ; then + MAIN="$HERE/usr/bin/$BINARY_NAME" +elif [ -e "$HERE/usr/python/bin/$BINARY_NAME" ] ; then + MAIN="$HERE/usr/python/bin/$BINARY_NAME" +elif [ -e "$HERE/usr/local/bin/$BINARY_NAME" ] ; then + MAIN="$HERE/usr/local/bin/$BINARY_NAME" +else + MAIN="$HERE/usr/python/bin/ocrmypdf" +fi + +exec "${MAIN}" "$@" diff --git a/appimage/ocrmypdf.png b/appimage/ocrmypdf.png new file mode 100644 index 00000000..be28f69c Binary files /dev/null and b/appimage/ocrmypdf.png differ diff --git a/build-appimage.sh b/build-appimage.sh new file mode 100644 index 00000000..72defc38 --- /dev/null +++ b/build-appimage.sh @@ -0,0 +1,116 @@ +#! /bin/bash + +set -x +set -e + +# use RAM disk if possible +if [ "$CI" == "" ] && [ -d /dev/shm ]; then + TEMP_BASE=/dev/shm +else + TEMP_BASE=/tmp +fi + +BUILD_DIR=$(mktemp -d -p "$TEMP_BASE" OCRmyPDF-AppImage-build-XXXXXX) + +cleanup () { + if [ -d "$BUILD_DIR" ]; then + rm -rf "$BUILD_DIR" + fi +} + +trap cleanup EXIT + +# store repo root as variable +REPO_ROOT=$(readlink -f "$(dirname "$(dirname "$0")")") +OLD_CWD=$(readlink -f .) + +pushd "$BUILD_DIR" + +mkdir -p AppDir +mkdir -p PackageDir +mkdir -p jbig2 + +# download linuxdeploy AppImage and linuxdeploy-plugin-python AppImage +wget https://github.com/TheAssassin/linuxdeploy/releases/download/continuous/linuxdeploy-x86_64.AppImage +# wget https://github.com/niess/linuxdeploy-plugin-python/releases/download/continuous/linuxdeploy-plugin-python-x86_64.AppImage + +# use adapted linuxdeploy-plugin-python instead of the original one (otherwise OCRmyPDF breaks) +wget https://github.com/FPille/linuxdeploy-plugin-python/releases/download/continuous/linuxdeploy-plugin-python-x86_64.AppImage + +chmod +x linuxdeploy*.AppImage + + +ARCH=$(uname -i) +export ARCH + + +# .desktop file +cat > ocrmypdf.desktop <<\EOF +[Desktop Entry] +Name=ocrmypdf +Type=Application +Exec=ocrmypdf +Icon=ocrmypdf +Terminal=true +Comment=OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched +Categories=Graphics;Scanning;OCR; +EOF + + +# download logo and convert it to desktop icon +# requires Imagemagick (convert) +wget https://raw.githubusercontent.com/jbarlow83/OCRmyPDF/master/docs/images/logo-social.png +convert logo-social.png -resize 512x512\> -size 512x512 xc:white +swap -gravity center -composite ocrmypdf.png + + +# download and intsall packages required by OCRmyPDF +pushd PackageDir +packages=(tesseract-ocr tesseract-ocr-all libavformat56 ghostscript qpdf pngquant) + +for i in "${packages[@]}" +do + apt-get -d -o dir::cache="$PWD" -o Debug::NoLocking=1 install "$i" -y +done + +wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O unpaper_6.1-1.deb + +find . -type f -name \*.deb -exec dpkg-deb -X {} "$BUILD_DIR"/AppDir \; +popd + + +# compile and install jbig2 +# requires libleptonica-dev, zlib1g-dev +wget -q https://github.com/agl/jbig2enc/archive/0.29.tar.gz -O - | \ + tar xz -C jbig2 --strip-components=1 +pushd jbig2 +./autogen.sh +./configure --prefix="$BUILD_DIR"/AppDir/usr +make && make install +popd + + +# remove unnecessary data from AppDir +pushd "$BUILD_DIR"/AppDir +[ -d etc ] && rm -rf ./etc +[ -d var ] && rm -rf ./var +popd + + +# export LD_LIBRARY_PATH so that dependencies of shared libraries can be deployed by linuxdeploy-x86_64.AppImage +export LD_LIBRARY_PATH="$BUILD_DIR/AppDir/usr/lib:$BUILD_DIR/AppDir/usr/lib/x86_64-linux-gnu:$LD_LIBRARY_PATH" + +#OCRMYPDF_VERSION=8.3.2 # exported in .travis.yml file +export PIP_REQUIREMENTS="ocrmypdf==$OCRMYPDF_VERSION" +export VERSION="$OCRMYPDF_VERSION" +export OUTPUT=OCRmyPDF-"$VERSION"-"$ARCH".AppImage +export PYTHON_SOURCE=https://www.python.org/ftp/python/3.6.8/Python-3.6.8.tgz + +./linuxdeploy-x86_64.AppImage --appdir AppDir --plugin python \ + -d ocrmypdf.desktop -i ocrmypdf.png \ + --custom-apprun "$REPO_ROOT"/appimage/AppRun.sh --output appimage + + +# move AppImage back to old CWD +mv "$OUTPUT" "$OLD_CWD"/ + +popd diff --git a/test/test-appimage.sh b/test/test-appimage.sh new file mode 100644 index 00000000..17d7a92d --- /dev/null +++ b/test/test-appimage.sh @@ -0,0 +1,60 @@ +#! /bin/bash + +set -x +set -e + +chmod +x OCRmyPDF*.AppImage + +# run OCRmyPDF to test if the AppImage can ocr a test file +run_appimage() +{ + ./OCRmyPDF*.AppImage -l deu -s -d --jbig2-lossy --optimize 1 "$TRAVIS_BUILD_DIR"/test/test.pdf output.pdf +} + + +# check AppImage for common issues +run_appimagelint() +{ + wget https://github.com/TheAssassin/appimagelint/releases/download/continuous/appimagelint-x86_64.AppImage + chmod +x appimagelint-x86_64.AppImage + ./appimagelint-x86_64.AppImage OCRmyPDF*.AppImage +} + + +# extract the OCRmyPDF AppImage, install pytest & test requirements and run pytest +run_pytest() +{ + git clone --depth=1 --branch v$OCRMYPDF_VERSION https://github.com/jbarlow83/OCRmyPDF.git + ./OCRmyPDF*.AppImage --appimage-extract + + pushd squashfs-root + ./AppRun python3 -m pip install pytest + ./AppRun python3 -m pip install -r ../OCRmyPDF/requirements/test.txt + ./AppRun python3 -m pytest ../OCRmyPDF -n auto + popd +} + + +run_appimage + +run_appimagelint + +# run_pytest # commented out in order to get the AppImage uploaded to github release + +# 'test_flate_to_jbig2' fails +# > assert pim.filters[0] == '/JBIG2Decode' +# E AssertionError: assert '/FlateDecode' == '/JBIG2Decode' +# E - /FlateDecode +# E + /JBIG2Decode + +# ../OCRmyPDF/tests/test_optimize.py:131: AssertionError +# ----------------------------- Captured stderr call ----------------------------- +# INFO - Input file is not a PDF, checking if it is an image... +# INFO - Input file is an image +# INFO - Image seems valid. Try converting to PDF... +# INFO - Successfully converted to PDF, processing... +# pngquant: unrecognized option '--skip-if-larger' +# INFO - Optimize ratio: 1.02 savings: 1.7% +# INFO - Output file is a PDF/A-2B (as expected) + + diff --git a/test/test.pdf b/test/test.pdf new file mode 100644 index 00000000..5d247764 Binary files /dev/null and b/test/test.pdf differ