Initial commit

This commit is contained in:
FPille
2019-07-29 20:56:31 +02:00
commit 03932e877a
7 changed files with 289 additions and 0 deletions
+26
View File
@@ -0,0 +1,26 @@
language: generic
sudo: required
dist: trusty
before_install:
- sudo add-apt-repository ppa:heyarje/libav-11 -y
- sudo add-apt-repository ppa:alex-p/tesseract-ocr -y
- sudo apt-get update -qq
- sudo apt-get install libleptonica-dev -y # required to build jbig2enc
- sudo apt-get install zlib1g-dev -y # required to build jbig2enc
- sudo apt-get install imagemagick -y # required to convert logo to desktop icon
script:
- export OCRMYPDF_VERSION=8.3.2
- bash build-appimage.sh
- bash test/test-appimage.sh
- wget https://github.com/probonopd/uploadtool/raw/master/upload.sh
after_success:
- bash upload.sh OCRmyPDF*.AppImage
branches:
except:
- # Do not build tags that we create when we upload to GitHub Releases
- /^(?i:continuous)/
+40
View File
@@ -0,0 +1,40 @@
# OCRmyPDF-AppImage [![Build Status](https://travis-ci.com/FPille/OCRmyPDF-AppImage.svg?branch=master)](https://travis-ci.com/FPille/OCRmyPDF-AppImage)
[AppImage][APPIMAGE] for [OCRmyPDF][OCRMYPDF]
## Usage
Download OCRmyPDF*.AppImage, make it executable and run it.
```
wget https://github.com/FPille/OCRmyPDF-AppImage/releases/download/continuous/OCRmyPDF-8.3.2-x86_64.AppImage
chmod +x OCRmyPDF*.AppImage
./OCRmyPDF*.AppImage --help
```
Beside OCRmyPDF additional command line programs can be run with this AppImage like:
* img2pdf
* ghostscript
* pngquant
* python3.6
* qpdf
* tesseract
* unpaper
Just use the program name as first parameter plus options:
```
./OCRmyPDF*.AppImage tesseract -v
tesseract 4.1.0
leptonica-1.76.0
libjpeg 8d (libjpeg-turbo 1.3.0) : libpng 1.2.50 : libtiff 4.0.3 : zlib 1.2.11 : libwebp 0.4.0 : libopenjp2 2.3.0
Found AVX2
Found AVX
Found SSE
```
Or create a symlink for the corresponding program:
```
ln -s OCRmyPDF*.AppImage tesseract
./tesseract --list-langs
```
[APPIMAGE]: https://appimage.org
[OCRMYPDF]: https://github.com/jbarlow83/OCRmyPDF
+47
View File
@@ -0,0 +1,47 @@
#! /bin/bash
# TODO: Add "usage" ouput and additional functions like displaying of licences
# and man pages
# could be in a similar manner as the AppRun script of qpdf AppImage
# refer to: https://github.com/qpdf/qpdf/blob/master/appimage/AppRun
HERE="$(dirname "$(readlink -f "${0}")")"
export PATH="$HERE/usr/bin:$HERE/usr/local/bin:$HERE/usr/python/bin:$PATH"
export LD_PRELOAD="$HERE/usr/lib/liblept.so.5"
export LD_LIBRARY_PATH="$HERE/usr/lib:$HERE/usr/lib/x86_64-linux-gnu:$LD_LIBRARY_PATH"
export TESSDATA_PREFIX="$HERE/usr/share/tesseract-ocr/4.00/tessdata"
export GS_LIB="$HERE/usr/share/ghostscript/9.26/lib:$HERE/usr/share/ghostscript/9.26/Resource:$HERE/usr/share/ghostscript/9.26/Resource/Init"
# Allow the AppImage to be symlinked to e.g., /usr/bin/commandname
# or called with ./Some*.AppImage commandname ...
# refer to https://github.com/AppImage/AppImageKit/wiki/Bundling-command-line-tools
if [ ! -z $APPIMAGE ] ; then
BINARY_NAME=$(basename "$ARGV0")
else
BINARY_NAME=$(basename "$0")
export APPDIR="$HERE" # required for the wrapper scripts of linuxdeploy-plugin-python
fi
if [ ! -z "$1" ] && [ -e "$HERE/bin/$1" ] ; then
MAIN="$HERE/bin/$1" ; shift
elif [ ! -z "$1" ] && [ -e "$HERE/usr/bin/$1" ] ; then
MAIN="$HERE/usr/bin/$1" ; shift
elif [ ! -z "$1" ] && [ -e "$HERE/usr/python/bin/$1" ] ; then
MAIN="$HERE/usr/python/bin/$1" ; shift
elif [ ! -z "$1" ] && [ -e "$HERE/usr/local/bin/$1" ] ; then
MAIN="$HERE/usr/local/bin/$1" ; shift
elif [ -e "$HERE/bin/$BINARY_NAME" ] ; then
MAIN="$HERE/bin/$BINARY_NAME"
elif [ -e "$HERE/usr/bin/$BINARY_NAME" ] ; then
MAIN="$HERE/usr/bin/$BINARY_NAME"
elif [ -e "$HERE/usr/python/bin/$BINARY_NAME" ] ; then
MAIN="$HERE/usr/python/bin/$BINARY_NAME"
elif [ -e "$HERE/usr/local/bin/$BINARY_NAME" ] ; then
MAIN="$HERE/usr/local/bin/$BINARY_NAME"
else
MAIN="$HERE/usr/python/bin/ocrmypdf"
fi
exec "${MAIN}" "$@"
Binary file not shown.

After

Width:  |  Height:  |  Size: 15 KiB

+116
View File
@@ -0,0 +1,116 @@
#! /bin/bash
set -x
set -e
# use RAM disk if possible
if [ "$CI" == "" ] && [ -d /dev/shm ]; then
TEMP_BASE=/dev/shm
else
TEMP_BASE=/tmp
fi
BUILD_DIR=$(mktemp -d -p "$TEMP_BASE" OCRmyPDF-AppImage-build-XXXXXX)
cleanup () {
if [ -d "$BUILD_DIR" ]; then
rm -rf "$BUILD_DIR"
fi
}
trap cleanup EXIT
# store repo root as variable
REPO_ROOT=$(readlink -f "$(dirname "$(dirname "$0")")")
OLD_CWD=$(readlink -f .)
pushd "$BUILD_DIR"
mkdir -p AppDir
mkdir -p PackageDir
mkdir -p jbig2
# download linuxdeploy AppImage and linuxdeploy-plugin-python AppImage
wget https://github.com/TheAssassin/linuxdeploy/releases/download/continuous/linuxdeploy-x86_64.AppImage
# wget https://github.com/niess/linuxdeploy-plugin-python/releases/download/continuous/linuxdeploy-plugin-python-x86_64.AppImage
# use adapted linuxdeploy-plugin-python instead of the original one (otherwise OCRmyPDF breaks)
wget https://github.com/FPille/linuxdeploy-plugin-python/releases/download/continuous/linuxdeploy-plugin-python-x86_64.AppImage
chmod +x linuxdeploy*.AppImage
ARCH=$(uname -i)
export ARCH
# .desktop file
cat > ocrmypdf.desktop <<\EOF
[Desktop Entry]
Name=ocrmypdf
Type=Application
Exec=ocrmypdf
Icon=ocrmypdf
Terminal=true
Comment=OCRmyPDF adds an OCR text layer to scanned PDF files, allowing them to be searched
Categories=Graphics;Scanning;OCR;
EOF
# download logo and convert it to desktop icon
# requires Imagemagick (convert)
wget https://raw.githubusercontent.com/jbarlow83/OCRmyPDF/master/docs/images/logo-social.png
convert logo-social.png -resize 512x512\> -size 512x512 xc:white +swap -gravity center -composite ocrmypdf.png
# download and intsall packages required by OCRmyPDF
pushd PackageDir
packages=(tesseract-ocr tesseract-ocr-all libavformat56 ghostscript qpdf pngquant)
for i in "${packages[@]}"
do
apt-get -d -o dir::cache="$PWD" -o Debug::NoLocking=1 install "$i" -y
done
wget -q 'https://www.dropbox.com/s/vaq0kbwi6e6au80/unpaper_6.1-1.deb?raw=1' -O unpaper_6.1-1.deb
find . -type f -name \*.deb -exec dpkg-deb -X {} "$BUILD_DIR"/AppDir \;
popd
# compile and install jbig2
# requires libleptonica-dev, zlib1g-dev
wget -q https://github.com/agl/jbig2enc/archive/0.29.tar.gz -O - | \
tar xz -C jbig2 --strip-components=1
pushd jbig2
./autogen.sh
./configure --prefix="$BUILD_DIR"/AppDir/usr
make && make install
popd
# remove unnecessary data from AppDir
pushd "$BUILD_DIR"/AppDir
[ -d etc ] && rm -rf ./etc
[ -d var ] && rm -rf ./var
popd
# export LD_LIBRARY_PATH so that dependencies of shared libraries can be deployed by linuxdeploy-x86_64.AppImage
export LD_LIBRARY_PATH="$BUILD_DIR/AppDir/usr/lib:$BUILD_DIR/AppDir/usr/lib/x86_64-linux-gnu:$LD_LIBRARY_PATH"
#OCRMYPDF_VERSION=8.3.2 # exported in .travis.yml file
export PIP_REQUIREMENTS="ocrmypdf==$OCRMYPDF_VERSION"
export VERSION="$OCRMYPDF_VERSION"
export OUTPUT=OCRmyPDF-"$VERSION"-"$ARCH".AppImage
export PYTHON_SOURCE=https://www.python.org/ftp/python/3.6.8/Python-3.6.8.tgz
./linuxdeploy-x86_64.AppImage --appdir AppDir --plugin python \
-d ocrmypdf.desktop -i ocrmypdf.png \
--custom-apprun "$REPO_ROOT"/appimage/AppRun.sh --output appimage
# move AppImage back to old CWD
mv "$OUTPUT" "$OLD_CWD"/
popd
+60
View File
@@ -0,0 +1,60 @@
#! /bin/bash
set -x
set -e
chmod +x OCRmyPDF*.AppImage
# run OCRmyPDF to test if the AppImage can ocr a test file
run_appimage()
{
./OCRmyPDF*.AppImage -l deu -s -d --jbig2-lossy --optimize 1 "$TRAVIS_BUILD_DIR"/test/test.pdf output.pdf
}
# check AppImage for common issues
run_appimagelint()
{
wget https://github.com/TheAssassin/appimagelint/releases/download/continuous/appimagelint-x86_64.AppImage
chmod +x appimagelint-x86_64.AppImage
./appimagelint-x86_64.AppImage OCRmyPDF*.AppImage
}
# extract the OCRmyPDF AppImage, install pytest & test requirements and run pytest
run_pytest()
{
git clone --depth=1 --branch v$OCRMYPDF_VERSION https://github.com/jbarlow83/OCRmyPDF.git
./OCRmyPDF*.AppImage --appimage-extract
pushd squashfs-root
./AppRun python3 -m pip install pytest
./AppRun python3 -m pip install -r ../OCRmyPDF/requirements/test.txt
./AppRun python3 -m pytest ../OCRmyPDF -n auto
popd
}
run_appimage
run_appimagelint
# run_pytest # commented out in order to get the AppImage uploaded to github release
# 'test_flate_to_jbig2' fails
# > assert pim.filters[0] == '/JBIG2Decode'
# E AssertionError: assert '/FlateDecode' == '/JBIG2Decode'
# E - /FlateDecode
# E + /JBIG2Decode
# ../OCRmyPDF/tests/test_optimize.py:131: AssertionError
# ----------------------------- Captured stderr call -----------------------------
# INFO - Input file is not a PDF, checking if it is an image...
# INFO - Input file is an image
# INFO - Image seems valid. Try converting to PDF...
# INFO - Successfully converted to PDF, processing...
# pngquant: unrecognized option '--skip-if-larger'
# INFO - Optimize ratio: 1.02 savings: 1.7%
# INFO - Output file is a PDF/A-2B (as expected)
BIN
View File
Binary file not shown.