diff --git a/debian/copyright b/debian/copyright index 8c190b4d..62a52737 100644 --- a/debian/copyright +++ b/debian/copyright @@ -79,7 +79,7 @@ Copyright: held by the contributors to the Wikipedia article "Optical character (epson.pdf generated from Wikipedia article as of 2016-09-14) License: CC-BY-SA-3.0 -Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf +Files: tests/resources/typewriter.png tests/resources/2400dpi.pdf tests/resources/3small.pdf Copyright: (C) 2005 Ellywa License: GFDL-1.2+ or CC-BY-SA-1.0 or CC-BY-SA-2.0 or CC-BY-SA-2.5 or CC-BY-SA-3.0 @@ -87,7 +87,7 @@ Files: tests/resources/overlay.pdf Copyright: (C) 2017 Max Anderson License: Expat -Files: tests/resources/baiona*.png +Files: tests/resources/baiona*.png tests/resources/3small.pdf Copyright: (C) 2014 Euskaldunaa License: CC-BY-SA-4.0 diff --git a/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 00000000..bb37acb2 Binary files /dev/null and b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/pdf.bin differ diff --git a/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin new file mode 100644 index 00000000..d4784957 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.1.0 with Leptonica diff --git a/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..151bced2 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,30 @@ +Tarnose + + + + + + +Bokale oa + + + +Lehuntze + + + + + +Mugerre + + + + +Milafranga Komunikabideak + +BAIONA i zeettnansise — + +1 Trenbideak -- ~~~ + +t\ Basusarri — spmsans20141004 se: . a ~ + \ No newline at end of file diff --git a/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 00000000..2e6ae52c Binary files /dev/null and b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/pdf.bin differ diff --git a/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin new file mode 100644 index 00000000..d4784957 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.1.0 with Leptonica diff --git a/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..2df091b8 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,2 @@ +Covfefe is a perfectly cromulent word. + \ No newline at end of file diff --git a/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin new file mode 100644 index 00000000..54fb87cb Binary files /dev/null and b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/pdf.bin differ diff --git a/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin new file mode 100644 index 00000000..d4784957 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stderr.bin @@ -0,0 +1 @@ +Tesseract Open Source OCR Engine v4.1.0 with Leptonica diff --git a/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/stdout.bin new file mode 100644 index 00000000..e69de29b diff --git a/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin new file mode 100644 index 00000000..522e4174 --- /dev/null +++ b/tests/cache/3small/__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt/txt.bin @@ -0,0 +1,27 @@ +Linzensoep a la Waterman + + + +4 ons linzen + +3 liter water + +3 uien + +bloem, boter + +2 kopjes melk + +laurier, kruidnagel, kerrie, zout + +De linzgen wassen en in -l liter kokend wa- +ter 1 dag laten weken, 2 liter water bij +de linzen voegen, zonder het water waarin +ze geweekt zijn af te gieten, De helft van +de uien bakken met laurier en Kruicdnagel. +Alle uien, kerrie en gout bij de linzen +voegen, Alles aan de kook brengen,. Van de +bloem met boter en melk een papje maken en +verder afmaken met de soep, Als de linzen +gaar Zijn is de soep klaar. + \ No newline at end of file diff --git a/tests/cache/manifest.jsonl b/tests/cache/manifest.jsonl index 73520981..05a0e86c 100644 --- a/tests/cache/manifest.jsonl +++ b/tests/cache/manifest.jsonl @@ -66,3 +66,6 @@ {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/cmyk.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.2 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.3", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/lichtenstein.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} {"tesseract_version": "tesseract 4.0.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.0.10 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.6.0-x86_64-i386-64bit", "python": "3.7.4", "argv_slug": "__-l__deu__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/graph_ocred.pdf", "args": ["-l", "deu", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000002_ocr.png__000002_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000002_ocr.png", "$TMPDIR/000002_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000001_ocr.png__000001_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000001_ocr.png", "$TMPDIR/000001_ocr_tess", "pdf", "txt"]} +{"tesseract_version": "tesseract 4.1.0 leptonica-1.78.0 libgif 5.1.4 : libjpeg 9c : libpng 1.6.37 : libtiff 4.1.0 : zlib 1.2.11 : libwebp 1.0.3 : libopenjp2 2.3.1 Found AVX2 Found AVX Found SSE ", "platform": "Darwin-18.7.0-x86_64-i386-64bit", "python": "3.7.5", "argv_slug": "__-l__eng__000003_ocr.png__000003_ocr_tess__pdf__txt", "sourcefile": "resources/3small.pdf", "args": ["-l", "eng", "-c", "textonly_pdf=1", "$TMPDIR/000003_ocr.png", "$TMPDIR/000003_ocr_tess", "pdf", "txt"]} diff --git a/tests/resources/3small.pdf b/tests/resources/3small.pdf new file mode 100644 index 00000000..7d282496 Binary files /dev/null and b/tests/resources/3small.pdf differ diff --git a/tests/test_main.py b/tests/test_main.py index a5af5b10..a08a0dde 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -21,6 +21,7 @@ import shutil from math import isclose from pathlib import Path from subprocess import PIPE, run +from unittest.mock import patch import PIL import pytest @@ -572,7 +573,12 @@ language_model_penalty_non_freq_dict_word 0 ) check_ocrmypdf( - resources / 'ccitt.pdf', outdir / 'out.pdf', '--tesseract-config', cfg_file + resources / '3small.pdf', + outdir / 'out.pdf', + '--tesseract-config', + cfg_file, + '--pages', + '1', ) @@ -634,7 +640,7 @@ def test_pagesize_consistency(renderer, resources, outpdf): first_page_dimensions = pytest.helpers.first_page_dimensions - infile = resources / 'linn.pdf' + infile = resources / '3small.pdf' before_dims = first_page_dimensions(infile) @@ -647,12 +653,14 @@ def test_pagesize_consistency(renderer, resources, outpdf): '--deskew', '--remove-background', '--clean-final' if pytest.helpers.have_unpaper() else None, + '--pages', + '1', ) after_dims = first_page_dimensions(outpdf) - assert isclose(before_dims[0], after_dims[0]) - assert isclose(before_dims[1], after_dims[1]) + assert isclose(before_dims[0], after_dims[0], rel_tol=1e-4) + assert isclose(before_dims[1], after_dims[1], rel_tol=1e-4) def test_skip_big_with_no_images(spoof_tesseract_noop, resources, outpdf): @@ -794,7 +802,7 @@ def test_compression_changed( def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): sidecar = outpdf.with_suffix('.txt') check_ocrmypdf( - resources / 'multipage.pdf', + resources / '3small.pdf', outpdf, '--skip-text', '--sidecar', @@ -802,7 +810,7 @@ def test_sidecar_pagecount(spoof_tesseract_cache, resources, outpdf): env=spoof_tesseract_cache, ) - pdfinfo = PdfInfo(resources / 'multipage.pdf') + pdfinfo = PdfInfo(resources / '3small.pdf') num_pages = len(pdfinfo) with open(sidecar, 'r', encoding='utf-8') as f: @@ -858,17 +866,18 @@ def test_decompression_bomb(resources, outpdf): def test_text_curves(spoof_tesseract_noop, resources, outpdf): - check_ocrmypdf(resources / 'vector.pdf', outpdf, env=spoof_tesseract_noop) + with patch('ocrmypdf._pipeline.VECTOR_PAGE_DPI', 100): + check_ocrmypdf(resources / 'vector.pdf', outpdf, env=spoof_tesseract_noop) - info = PdfInfo(outpdf) - assert len(info.pages[0].images) == 0, "added images to the vector PDF" + info = PdfInfo(outpdf) + assert len(info.pages[0].images) == 0, "added images to the vector PDF" - check_ocrmypdf( - resources / 'vector.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop - ) + check_ocrmypdf( + resources / 'vector.pdf', outpdf, '--force-ocr', env=spoof_tesseract_noop + ) - info = PdfInfo(outpdf) - assert len(info.pages[0].images) != 0, "force did not rasterize" + info = PdfInfo(outpdf) + assert len(info.pages[0].images) != 0, "force did not rasterize" def test_output_is_dir(spoof_tesseract_noop, resources, outdir): diff --git a/tests/test_tess4.py b/tests/test_tess4.py index bb9cc49a..0b835ea7 100644 --- a/tests/test_tess4.py +++ b/tests/test_tess4.py @@ -37,30 +37,6 @@ def test_tesseract_v4(): assert tesseract.v4() -def test_pagesize_consistency_tess4(resources, outpdf): - from math import isclose - - infile = resources / 'linn.pdf' - - before_dims = pytest.helpers.first_page_dimensions(infile) - - check_ocrmypdf( - infile, - outpdf, - '--pdf-renderer', - 'sandwich', - '--clean' if pytest.helpers.have_unpaper() else None, - '--deskew', - '--remove-background', - '--clean-final' if pytest.helpers.have_unpaper() else None, - ) - - after_dims = pytest.helpers.first_page_dimensions(outpdf) - - assert isclose(before_dims[0], after_dims[0]) - assert isclose(before_dims[1], after_dims[1]) - - @pytest.mark.parametrize('basename', ['graph_ocred.pdf', 'cardinal.pdf']) def test_skip_pages_does_not_replicate(resources, basename, outdir): infile = resources / basename