Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
66337813e6 | ||
|
|
eb5a211e72 | ||
|
|
5142933120 | ||
|
|
06ab114aa8 | ||
|
|
1257419465 | ||
|
|
30404f53f0 | ||
|
|
1ce8edbdfe | ||
|
|
d4b704a0ae | ||
|
|
2d64e1536d | ||
|
|
c8b581ac31 | ||
|
|
ad8dead7df | ||
|
|
c9bd87254e | ||
|
|
f4cb424451 | ||
|
|
fef14778d5 | ||
|
|
86ec63f215 | ||
|
|
5b10ec9d39 | ||
|
|
800c75c4e5 | ||
|
|
24d64b04c3 | ||
|
|
48e2750551 | ||
|
|
e182c5f63e | ||
|
|
06d52326db | ||
|
|
ebfe4f0d29 | ||
|
|
ad22977c84 | ||
|
|
6ac50646f0 | ||
|
|
24b6a4ad50 | ||
|
|
e802896d4d | ||
|
|
0b5a20e593 | ||
|
|
642998ead6 | ||
|
|
698aab4f75 | ||
|
|
34231ac667 | ||
|
|
ddedf7cd2e | ||
|
|
9d127d354c | ||
|
|
2d2a4894ab | ||
|
|
862861e3ca | ||
|
|
892db88f0e | ||
|
|
eeb44f78cc | ||
|
|
863835f660 | ||
|
|
393c5a9ea4 | ||
|
|
c6b9a49cbb | ||
|
|
17a4831745 | ||
|
|
7caf1e85ff | ||
|
|
f59a757e8b | ||
|
|
872bafad4b | ||
|
|
8599400445 | ||
|
|
b6eebadf05 | ||
|
|
a4e88eb8f0 | ||
|
|
f6257c2183 | ||
|
|
64891c2fc3 | ||
|
|
fe156db41d | ||
|
|
0f942fb714 | ||
|
|
be8ca589d4 | ||
|
|
3b6f6782f0 | ||
|
|
21c0e045cb | ||
|
|
ebbf68bd08 | ||
|
|
2059e916da | ||
|
|
c22f245606 | ||
|
|
7b9025f397 | ||
|
|
b109445215 | ||
|
|
fd1cd8e50a | ||
|
|
c6c70c2171 | ||
|
|
a9a473f2e5 | ||
|
|
6268e2faff | ||
|
|
ec3f506500 | ||
|
|
00daa51a73 | ||
|
|
e60f4d3f43 | ||
|
|
7460745f80 | ||
|
|
5e14d5b0dd | ||
|
|
d118132fa6 | ||
|
|
5f47aac36f | ||
|
|
c6b2fa8851 | ||
|
|
1b92f447c3 | ||
|
|
82e7eb91d2 | ||
|
|
4f4ad0fb76 | ||
|
|
1d0b8641a0 | ||
|
|
daca919775 | ||
|
|
1598f2f0e5 | ||
|
|
2b23f7ec73 | ||
|
|
6528234608 | ||
|
|
642ebc6098 | ||
|
|
74fdfeea3f | ||
|
|
3754185f56 | ||
|
|
df9f5157bd | ||
|
|
aa060db5bc | ||
|
|
d43212d30b | ||
|
|
a0f9ca3a30 | ||
|
|
0cefe886ec | ||
|
|
f656c00f41 | ||
|
|
03da34ee24 | ||
|
|
9bccff4f88 | ||
|
|
2bd586e093 | ||
|
|
9af94ac9b7 | ||
|
|
8174089c8b | ||
|
|
41eb54cc0a | ||
|
|
12a2f78c4d | ||
|
|
d372f1f7fa | ||
|
|
6f5b75bcd0 | ||
|
|
a2d3e0b53e | ||
|
|
7f67556995 | ||
|
|
db8c37e58c | ||
|
|
a87c81a64f | ||
|
|
4b986a5943 | ||
|
|
2fae9b655e | ||
|
|
2541f6cf89 | ||
|
|
33b68454f3 | ||
|
|
977665d2b6 | ||
|
|
fd7497f00d | ||
|
|
790ff58f67 | ||
|
|
4b98ce391b | ||
|
|
417dbd43f6 | ||
|
|
7a12908db9 | ||
|
|
9462f0a28f | ||
|
|
e760622a5c | ||
|
|
1b086f60a9 | ||
|
|
85cbf94a6e | ||
|
|
6f4286e1b1 | ||
|
|
39888ae8c9 | ||
|
|
dd361ecd05 | ||
|
|
32759c9025 | ||
|
|
59440448ee | ||
|
|
51b54893ce | ||
|
|
1f3665f614 | ||
|
|
75c34b873a | ||
|
|
fe4296c53b | ||
|
|
c85278b31d | ||
|
|
5dbc080fa0 | ||
|
|
e02f6c1e97 | ||
|
|
8c9a8fc85c | ||
|
|
23d558ad8c | ||
|
|
be107b4fed | ||
|
|
8d2535e327 | ||
|
|
5eb4fe0052 | ||
|
|
d8ff4485f8 | ||
|
|
82bce463ae | ||
|
|
016dfd420c | ||
|
|
8f5c95f0f4 | ||
|
|
168fc60774 | ||
|
|
c84d0f606d | ||
|
|
8b54ce338f | ||
|
|
18c4aa10bf | ||
|
|
991db17fde | ||
|
|
2c07515907 | ||
|
|
27a3b80376 | ||
|
|
8c381a0227 | ||
|
|
86145a8c76 | ||
|
|
7513f5425c | ||
|
|
af3c3c6466 | ||
|
|
db3e75e33e | ||
|
|
ce49fc26dd | ||
|
|
d0d0a98dca | ||
|
|
94c52a6fa3 | ||
|
|
57771f06a3 | ||
|
|
4581027246 | ||
|
|
31b5f63f85 | ||
|
|
957fb1494e | ||
|
|
9e3e4f2687 | ||
|
|
2155bcacb4 | ||
|
|
346da95899 | ||
|
|
f4f7946a0c | ||
|
|
c2919f2e1c | ||
|
|
a63d624052 | ||
|
|
af91489376 | ||
|
|
d146d2b65c | ||
|
|
4ff4ed24a8 |
+1
-2
@@ -11,8 +11,6 @@ concurrency =
|
|||||||
multiprocessing
|
multiprocessing
|
||||||
source =
|
source =
|
||||||
src/ocrmypdf
|
src/ocrmypdf
|
||||||
omit =
|
|
||||||
tests/spoof/*
|
|
||||||
|
|
||||||
[report]
|
[report]
|
||||||
exclude_lines =
|
exclude_lines =
|
||||||
@@ -23,3 +21,4 @@ exclude_lines =
|
|||||||
if 0:
|
if 0:
|
||||||
if False:
|
if False:
|
||||||
if __name__ == .__main__.:
|
if __name__ == .__main__.:
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
|||||||
+2
-2
@@ -1,6 +1,6 @@
|
|||||||
# OCRmyPDF
|
# OCRmyPDF
|
||||||
#
|
#
|
||||||
FROM ubuntu:19.10 as base
|
FROM ubuntu:20.04 as base
|
||||||
|
|
||||||
FROM base as builder
|
FROM base as builder
|
||||||
|
|
||||||
@@ -24,7 +24,7 @@ RUN \
|
|||||||
# Needs libleptonica-dev, zlib1g-dev
|
# Needs libleptonica-dev, zlib1g-dev
|
||||||
RUN \
|
RUN \
|
||||||
mkdir jbig2 \
|
mkdir jbig2 \
|
||||||
&& curl -L https://github.com/agl/jbig2enc/archive/0.29.tar.gz | \
|
&& curl -L https://github.com/agl/jbig2enc/archive/ea6a40a.tar.gz | \
|
||||||
tar xz -C jbig2 --strip-components=1 \
|
tar xz -C jbig2 --strip-components=1 \
|
||||||
&& cd jbig2 \
|
&& cd jbig2 \
|
||||||
&& ./autogen.sh && ./configure && make && make install \
|
&& ./autogen.sh && ./configure && make && make install \
|
||||||
|
|||||||
+38
-19
@@ -1,25 +1,44 @@
|
|||||||
|
# dotfiles
|
||||||
|
.*
|
||||||
|
!.coveragerc
|
||||||
|
!.dockerignore
|
||||||
|
!.git_archival.txt
|
||||||
|
!.gitattributes
|
||||||
|
!.gitignore
|
||||||
|
!.pre-commit-config.yaml
|
||||||
|
!.readthedocs.yml
|
||||||
|
|
||||||
|
# Dev scratch
|
||||||
*.ipynb
|
*.ipynb
|
||||||
*.pdf
|
|
||||||
*.pyc
|
|
||||||
*.rst
|
|
||||||
*.sublime*
|
|
||||||
**/*.pyc
|
**/*.pyc
|
||||||
.*/
|
/*.pdf
|
||||||
!.git/
|
/*.qdf
|
||||||
.ruffus_history.sqlite
|
/*.png
|
||||||
bin/
|
/scratch.py
|
||||||
build/
|
IDEAS
|
||||||
docs/
|
log/
|
||||||
dist/
|
|
||||||
htmlcov/
|
|
||||||
include/
|
|
||||||
lib/
|
|
||||||
MANIFEST.in
|
|
||||||
ocrmypdf.egg-info/
|
|
||||||
staging/
|
|
||||||
tests/cache/
|
|
||||||
tests/output/
|
|
||||||
tests/resources/private/
|
tests/resources/private/
|
||||||
tmp/
|
tmp/
|
||||||
venv*/
|
venv*/
|
||||||
|
/debug_tests.py
|
||||||
|
*.traineddata
|
||||||
|
/private
|
||||||
|
|
||||||
|
# Package building
|
||||||
|
*.egg-info/
|
||||||
|
build/
|
||||||
|
dist/
|
||||||
wheelhouse/
|
wheelhouse/
|
||||||
|
pip-wheel-metadata/
|
||||||
|
|
||||||
|
# Code coverage
|
||||||
|
htmlcov/
|
||||||
|
|
||||||
|
# Docker specific
|
||||||
|
bin/
|
||||||
|
docs/
|
||||||
|
include/
|
||||||
|
lib/
|
||||||
|
|
||||||
|
# Docker include .git/
|
||||||
|
!.git/
|
||||||
|
|||||||
@@ -20,9 +20,13 @@ ocrmypdf ...arguments... input.pdf output.pdf
|
|||||||
Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful.
|
Run with verbosity or higher `-v1` to see more detailed logging. This information may be helpful.
|
||||||
|
|
||||||
**Example file**
|
**Example file**
|
||||||
Please include an example *input* PDF (or image). The input file is more helpful.
|
Include an input PDF or image that demonstrates your issue.
|
||||||
|
|
||||||
If possible, use an input file with no personal or confidential information. At your option you may GPG-encrypt the file for OCRmyPDF's author only.
|
Please provide an input file with no personal or confidential information. At your option you may `GPG-encrypt the file <https://github.com/jbarlow83/OCRmyPDF/wiki>` for OCRmyPDF's author only.
|
||||||
|
|
||||||
|
Links to files hosted elsewhere are perfectly acceptable. You could also look in ``tests/resources`` and see if any of those files reproduce your issue.
|
||||||
|
|
||||||
|
(Exceptions: Issues with installation, command line argument parsing, test suite failures.Issues without example files usually cannot be resolved.)
|
||||||
|
|
||||||
**Expected behavior**
|
**Expected behavior**
|
||||||
A clear and concise description of what you expected to happen.
|
A clear and concise description of what you expected to happen.
|
||||||
|
|||||||
+28
-33
@@ -1,47 +1,42 @@
|
|||||||
# Development environment
|
# dotfiles
|
||||||
.bash_history
|
.*
|
||||||
.pylintrc
|
!.coveragerc
|
||||||
.pytest_cache/
|
!.dockerignore
|
||||||
.ruffus_history.sqlite
|
!.git_archival.txt
|
||||||
.venv*/
|
!.gitattributes
|
||||||
*.pyc
|
!.gitignore
|
||||||
*.sublime-*
|
!.pre-commit-config.yaml
|
||||||
*.DS_Store
|
!.readthedocs.yml
|
||||||
.mypy_cache/
|
|
||||||
|
# Dev scratch
|
||||||
|
*.ipynb
|
||||||
|
**/*.pyc
|
||||||
|
/*.pdf
|
||||||
|
/*.qdf
|
||||||
|
/*.png
|
||||||
|
/scratch.py
|
||||||
|
IDEAS
|
||||||
|
log/
|
||||||
|
tests/resources/private/
|
||||||
|
tmp/
|
||||||
|
venv*/
|
||||||
|
/debug_tests.py
|
||||||
|
*.traineddata
|
||||||
|
/private
|
||||||
|
|
||||||
# Package building
|
# Package building
|
||||||
.eggs/
|
|
||||||
*.egg-info/
|
*.egg-info/
|
||||||
build/
|
build/
|
||||||
dist/
|
dist/
|
||||||
wheelhouse/
|
wheelhouse/
|
||||||
pip-wheel-metadata/
|
pip-wheel-metadata/
|
||||||
|
|
||||||
|
# Code coverage
|
||||||
|
htmlcov/
|
||||||
|
|
||||||
# Automatically generated files
|
# Automatically generated files
|
||||||
docs/_build/
|
docs/_build/
|
||||||
docs/_static/
|
docs/_static/
|
||||||
docs/_templates/
|
docs/_templates/
|
||||||
docs/Makefile
|
docs/Makefile
|
||||||
ocrmypdf/lib/_*.py
|
ocrmypdf/lib/_*.py
|
||||||
|
|
||||||
# Code coverage
|
|
||||||
.coverage*
|
|
||||||
htmlcov/
|
|
||||||
|
|
||||||
# Testing
|
|
||||||
.ipynb_checkpoints/
|
|
||||||
.vscode/
|
|
||||||
*.ipynb
|
|
||||||
*.profile
|
|
||||||
/*.pdf
|
|
||||||
/*.qdf
|
|
||||||
/*.png
|
|
||||||
/scratch.py
|
|
||||||
IDEAS
|
|
||||||
log/
|
|
||||||
tests/output/
|
|
||||||
tests/resources/private/
|
|
||||||
tmp/
|
|
||||||
/debug_tests.py
|
|
||||||
*.traineddata
|
|
||||||
/private
|
|
||||||
|
|||||||
+2
-1
@@ -148,7 +148,8 @@ In addition to tesseract, OCRmyPDF uses the following external binaries:
|
|||||||
|
|
||||||
- ``gs`` (Ghostscript)
|
- ``gs`` (Ghostscript)
|
||||||
- ``unpaper``
|
- ``unpaper``
|
||||||
- ``qpdf``
|
- ``pngquant``
|
||||||
|
- ``jbig2``
|
||||||
|
|
||||||
In each case OCRmyPDF will search the ``PATH`` environment variable to
|
In each case OCRmyPDF will search the ``PATH`` environment variable to
|
||||||
locate the binaries.
|
locate the binaries.
|
||||||
|
|||||||
+5
-5
@@ -56,8 +56,8 @@ OCRmyPDF does not.
|
|||||||
On Windows, the script that calls ``ocrmypdf.ocr()`` must be protected
|
On Windows, the script that calls ``ocrmypdf.ocr()`` must be protected
|
||||||
by an "ifmain" guard (``if __name__ == '__main__'``) or you must use
|
by an "ifmain" guard (``if __name__ == '__main__'``) or you must use
|
||||||
``ocrmypdf.ocr(...use_threads=True)``. If you do not take at least one
|
``ocrmypdf.ocr(...use_threads=True)``. If you do not take at least one
|
||||||
of these steps, Windows fork semantics will prevent OCRmyPDF from working
|
of these steps, Windows process semantics will prevent OCRmyPDF from working
|
||||||
correct.
|
correctly.
|
||||||
|
|
||||||
Logging
|
Logging
|
||||||
-------
|
-------
|
||||||
@@ -105,8 +105,8 @@ Reference
|
|||||||
:members:
|
:members:
|
||||||
:undoc-members:
|
:undoc-members:
|
||||||
|
|
||||||
.. autoclass:: ocrmypdf.ExitCode
|
.. autofunction:: ocrmypdf.configure_logging
|
||||||
|
|
||||||
|
.. automodule:: ocrmypdf.exceptions
|
||||||
:members:
|
:members:
|
||||||
:undoc-members:
|
:undoc-members:
|
||||||
|
|
||||||
.. autofunction:: ocrmypdf.configure_logging
|
|
||||||
|
|||||||
+5
-1
@@ -56,7 +56,11 @@ portrait pages.
|
|||||||
ocrmypdf --rotate-pages myfile.pdf myfile.pdf
|
ocrmypdf --rotate-pages myfile.pdf myfile.pdf
|
||||||
|
|
||||||
You can increase (decrease) the parameter ``--rotate-pages-threshold``
|
You can increase (decrease) the parameter ``--rotate-pages-threshold``
|
||||||
to make page rotation more (less) aggressive.
|
to make page rotation more (less) aggressive. The threshold number is the ratio
|
||||||
|
of how confidence the OCR engine is that the document image should be changed,
|
||||||
|
compared to kept the same. A value of ``15.0`` is the default, and is fairly
|
||||||
|
conservative. A value of ``2.0`` will produce more rotations, and more false
|
||||||
|
positives.
|
||||||
|
|
||||||
If the page is "just a little off horizontal", like a crooked picture,
|
If the page is "just a little off horizontal", like a crooked picture,
|
||||||
then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal
|
then you want ``--deskew``. ``--rotate-pages`` is for when the cardinal
|
||||||
|
|||||||
+10
-4
@@ -22,14 +22,20 @@ As the error message suggests, your options are:
|
|||||||
- ``ocrmypdf --skip-text`` to skip OCR and other processing on any
|
- ``ocrmypdf --skip-text`` to skip OCR and other processing on any
|
||||||
pages that contain text. Text pages will be copied into the output
|
pages that contain text. Text pages will be copied into the output
|
||||||
PDF without modification.
|
PDF without modification.
|
||||||
|
- ``ocrmypdf --redo-ocr`` to scan the file for any existing OCR
|
||||||
|
(non-printing text), remove it, and do OCR again. This is one way
|
||||||
|
to take advantage of improvements in OCR accuracy. Printable vector
|
||||||
|
text is excluded from OCR, so this can be used on files that contain
|
||||||
|
a mix of digital and scanned files.
|
||||||
|
|
||||||
|
|
||||||
Input file 'filename' is not a valid PDF
|
Input file 'filename' is not a valid PDF
|
||||||
========================================
|
========================================
|
||||||
|
|
||||||
OCRmyPDF passes files through qpdf, a program that fixes errors in PDFs,
|
OCRmyPDF checks files with pikepdf, a library that in turn uses libqpdf to fixes
|
||||||
before it tries to work on them. In most cases this happens because the
|
errors in PDFs, before it tries to work on them. In most cases this happens
|
||||||
PDF is corrupt and truncated (incomplete file copying) and not much can
|
because the PDF is corrupt and truncated (incomplete file copying) and not much
|
||||||
be done.
|
can be done.
|
||||||
|
|
||||||
You can try rewriting the file with Ghostscript:
|
You can try rewriting the file with Ghostscript:
|
||||||
|
|
||||||
|
|||||||
+3
-1
@@ -26,7 +26,8 @@ image processing and OCR to existing PDFs.
|
|||||||
docker
|
docker
|
||||||
advanced
|
advanced
|
||||||
batch
|
batch
|
||||||
security
|
performance
|
||||||
|
pdfsecurity
|
||||||
errors
|
errors
|
||||||
|
|
||||||
.. toctree::
|
.. toctree::
|
||||||
@@ -34,6 +35,7 @@ image processing and OCR to existing PDFs.
|
|||||||
:maxdepth: 2
|
:maxdepth: 2
|
||||||
|
|
||||||
api
|
api
|
||||||
|
plugins
|
||||||
contributing
|
contributing
|
||||||
|
|
||||||
Indices and tables
|
Indices and tables
|
||||||
|
|||||||
+96
-38
@@ -45,14 +45,8 @@ Debian and Ubuntu 18.04 or newer
|
|||||||
.. |ubu-1804| image:: https://repology.org/badge/version-for-repo/ubuntu_18_04/ocrmypdf.svg
|
.. |ubu-1804| image:: https://repology.org/badge/version-for-repo/ubuntu_18_04/ocrmypdf.svg
|
||||||
:alt: Ubuntu 18.04 LTS
|
:alt: Ubuntu 18.04 LTS
|
||||||
|
|
||||||
.. |ubu-1810| image:: https://repology.org/badge/version-for-repo/ubuntu_18_10/ocrmypdf.svg
|
.. |ubu-2004| image:: https://repology.org/badge/version-for-repo/ubuntu_20_04/ocrmypdf.svg
|
||||||
:alt: Ubuntu 18.10
|
:alt: Ubuntu 20.04 LTS
|
||||||
|
|
||||||
.. |ubu-1904| image:: https://repology.org/badge/version-for-repo/ubuntu_19_04/ocrmypdf.svg
|
|
||||||
:alt: Ubuntu 19.04
|
|
||||||
|
|
||||||
.. |ubu-1910| image:: https://repology.org/badge/version-for-repo/ubuntu_19_10/ocrmypdf.svg
|
|
||||||
:alt: Ubuntu 19.10
|
|
||||||
|
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| **OCRmyPDF versions in Debian & Ubuntu** |
|
| **OCRmyPDF versions in Debian & Ubuntu** |
|
||||||
@@ -61,7 +55,7 @@ Debian and Ubuntu 18.04 or newer
|
|||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |deb-stable| |deb-testing| |deb-unstable| |
|
| |deb-stable| |deb-testing| |deb-unstable| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
| |ubu-1804| |ubu-1810| |ubu-1904| |ubu-1910| |
|
| |ubu-1804| |ubu-2004| |
|
||||||
+-----------------------------------------------+
|
+-----------------------------------------------+
|
||||||
|
|
||||||
Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later, including users
|
Users of Debian 9 ("stretch") or later or Ubuntu 18.04 or later, including users
|
||||||
@@ -133,7 +127,41 @@ from sources <#installing-head-revision-from-sources>`__.
|
|||||||
|
|
||||||
.. _ubuntu-lts-latest:
|
.. _ubuntu-lts-latest:
|
||||||
|
|
||||||
Installing the latest version on Ubuntu 18.04 LTS
|
Installing the latest version on Ubuntu 20.04 LTS
|
||||||
|
-------------------------------------------------
|
||||||
|
|
||||||
|
Ubuntu 20.04 includes ocrmypdf 9.6.0 - you can install that with ``apt``. To
|
||||||
|
install a more recent version, uninstall the system-provided version of
|
||||||
|
ocrmypdf, and install the following dependencies:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
sudo apt-get -y remove ocrmypdf # remove system ocrmypdf, if installed
|
||||||
|
sudo apt-get -y update
|
||||||
|
sudo apt-get -y install \
|
||||||
|
ghostscript \
|
||||||
|
icc-profiles-free \
|
||||||
|
liblept5 \
|
||||||
|
libxml2 \
|
||||||
|
pngquant \
|
||||||
|
python3-pip \
|
||||||
|
tesseract-ocr \
|
||||||
|
zlib1g
|
||||||
|
|
||||||
|
To install ocrmypdf for the system:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
sudo pip3 install ocrmypdf
|
||||||
|
|
||||||
|
To install for the current user only:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
export PATH=$HOME/.local/bin:$PATH
|
||||||
|
pip3 install --user ocrmypdf
|
||||||
|
|
||||||
|
Ubuntu 18.04 LTS
|
||||||
-------------------------------------------------
|
-------------------------------------------------
|
||||||
|
|
||||||
Ubuntu 18.04 includes ocrmypdf 6.1.2 - you can install that with ``apt``, but
|
Ubuntu 18.04 includes ocrmypdf 6.1.2 - you can install that with ``apt``, but
|
||||||
@@ -327,20 +355,7 @@ standard tooling needed to build packages, such as a compiler and binary tools.
|
|||||||
|
|
||||||
sudo pacman -S base-devel
|
sudo pacman -S base-devel
|
||||||
|
|
||||||
The OCRmyPDF package depends on `the python-pdfminer.six AUR package
|
Now you are ready to install the OCRmyPDF package.
|
||||||
<https://aur.archlinux.org/packages/python-pdfminer.six/>`__. Dependencies on
|
|
||||||
AUR packages are not automatically resolved, so this package must be manually
|
|
||||||
installed first.
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
curl -O https://aur.archlinux.org/cgit/aur.git/snapshot/python-pdfminer.six.tar.gz
|
|
||||||
tar xvzf python-pdfminer.six.tar.gz
|
|
||||||
cd python-pdfminer.six
|
|
||||||
makepkg -sri
|
|
||||||
|
|
||||||
With that complete you can then repeat the same series of steps for the
|
|
||||||
OCRmyPDF package.
|
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -374,11 +389,10 @@ page.
|
|||||||
fine without it but will produce larger output files. The encoder is
|
fine without it but will produce larger output files. The encoder is
|
||||||
available from `the jbig2enc-git AUR package
|
available from `the jbig2enc-git AUR package
|
||||||
<https://aur.archlinux.org/packages/jbig2enc-git/>`__ and may be installed
|
<https://aur.archlinux.org/packages/jbig2enc-git/>`__ and may be installed
|
||||||
using the same series of steps as for the installation of the pdfminer.six
|
using the same series of steps as for the installation OCRmyPDF AUR
|
||||||
and OCRmyPDF AUR packages. Alternatively, it may be built manually from
|
package. Alternatively, it may be built manually from source following the
|
||||||
source following the instructions in `Installing the JBIG2 encoder
|
instructions in `Installing the JBIG2 encoder <jbig2>`__. If JBIG2 is
|
||||||
<jbig2>`__. If JBIG2 is installed, OCRmyPDF 7.0.0 and later will
|
installed, OCRmyPDF 7.0.0 and later will automatically detect it.
|
||||||
automatically detect it.
|
|
||||||
|
|
||||||
Alpine Linux
|
Alpine Linux
|
||||||
------------
|
------------
|
||||||
@@ -443,7 +457,10 @@ languages you can optionally install them all:
|
|||||||
Manual installation on macOS
|
Manual installation on macOS
|
||||||
----------------------------
|
----------------------------
|
||||||
|
|
||||||
These instructions probably work on all macOS supported by Homebrew.
|
These instructions probably work on all macOS supported by Homebrew, and are
|
||||||
|
for installing a more current version of OCRmyPDF than is available from
|
||||||
|
Homebrew. Note that the Homebrew versions usually track the release versions
|
||||||
|
fairly closely.
|
||||||
|
|
||||||
If it's not already present, `install Homebrew <http://brew.sh/>`__.
|
If it's not already present, `install Homebrew <http://brew.sh/>`__.
|
||||||
|
|
||||||
@@ -454,14 +471,8 @@ Update Homebrew:
|
|||||||
brew update
|
brew update
|
||||||
|
|
||||||
Install or upgrade the required Homebrew packages, if any are missing.
|
Install or upgrade the required Homebrew packages, if any are missing.
|
||||||
To do this, download the ``Brewfile`` that lists all of the dependencies
|
To do this, use ``brew edit ocrmypdf`` to obtain a recent list of Homebrew
|
||||||
to the current directory, and run ``brew bundle`` to process them
|
dependencies. You could also check the ``azure-pipelines.yml``.
|
||||||
(installing or upgrading as needed). ``Brewfile`` is a plain text file.
|
|
||||||
|
|
||||||
.. code-block:: bash
|
|
||||||
|
|
||||||
wget https://github.com/jbarlow83/OCRmyPDF/raw/master/.travis/Brewfile
|
|
||||||
brew bundle
|
|
||||||
|
|
||||||
This will include the English, French, German and Spanish language
|
This will include the English, French, German and Spanish language
|
||||||
packs. If you need other languages you can optionally install them all:
|
packs. If you need other languages you can optionally install them all:
|
||||||
@@ -592,6 +603,53 @@ Docker
|
|||||||
You can also :ref:`Install the Docker <docker-install>` container on Windows. Ensure that
|
You can also :ref:`Install the Docker <docker-install>` container on Windows. Ensure that
|
||||||
your command prompt can run the docker "hello world" container.
|
your command prompt can run the docker "hello world" container.
|
||||||
|
|
||||||
|
Installing on Cygwin64 under Windows
|
||||||
|
====================================
|
||||||
|
|
||||||
|
First install the the following prerequisite Cygwin packages using ``setup-x86_64.exe``::
|
||||||
|
|
||||||
|
python36 (or later)
|
||||||
|
python3?-devel
|
||||||
|
python3?-pip
|
||||||
|
python3?-lxml
|
||||||
|
python3?-imaging
|
||||||
|
|
||||||
|
(where 3? means match the version of python3 you installed)
|
||||||
|
|
||||||
|
gcc-g++
|
||||||
|
ghostscript (<=9.50 or >=9.52-2 see note below)
|
||||||
|
libexempi3
|
||||||
|
libexempi-devel
|
||||||
|
libffi6
|
||||||
|
libffi-devel
|
||||||
|
pngquant
|
||||||
|
qpdf
|
||||||
|
libqpdf-devel
|
||||||
|
tesseract-ocr
|
||||||
|
tesseract-ocr-devel
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
|
||||||
|
The Cygwin package for Ghostscript in versions 9.52 and
|
||||||
|
9.52-1 contained a bug that caused an exception to occur when
|
||||||
|
ocrmypdf invoked gs. Make sure you have either 9.50 (or earlier)
|
||||||
|
or 9.52-2 (or later).
|
||||||
|
|
||||||
|
Then open a Cygwin terminal (i.e. ``mintty``), run the following commands. Note
|
||||||
|
that if you are using the version of ``pip`` that was installed with the Cygwin
|
||||||
|
Python package, the command name will be ``pip3``. If you have since updated
|
||||||
|
``pip`` (with, for instance ``pip3 install --upgrade pip``) the the command is
|
||||||
|
likely just ``pip`` instead of ``pip3``:
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
pip3 install wheel
|
||||||
|
pip3 install ocrmypdf
|
||||||
|
|
||||||
|
The optional dependency "unpaper" that is currently not available under Cygwin.
|
||||||
|
Without it, certain options such as ``--clean`` will produce an error message.
|
||||||
|
However, the OCR-to-text-layer functionality is available.
|
||||||
|
|
||||||
Installing with Python pip
|
Installing with Python pip
|
||||||
==========================
|
==========================
|
||||||
|
|
||||||
|
|||||||
@@ -68,7 +68,7 @@ license, OCRmyPDF's GPL license, and any other licenses.
|
|||||||
|
|
||||||
Setting aside these concerns, a side effect of OCRmyPDF is it may
|
Setting aside these concerns, a side effect of OCRmyPDF is it may
|
||||||
incidentally sanitize PDFs that contain certain types of malware. It
|
incidentally sanitize PDFs that contain certain types of malware. It
|
||||||
runs ``qpdf`` to repair the PDF, which could correct malformed PDF
|
repairs the PDF with pikepdf/libqpdf, which could correct malformed PDF
|
||||||
structures that are part of an attack. When PDF/A output is selected
|
structures that are part of an attack. When PDF/A output is selected
|
||||||
(the default), the input PDF is partially reconstructed by Ghostscript.
|
(the default), the input PDF is partially reconstructed by Ghostscript.
|
||||||
When ``--force-ocr`` is used, all pages are rasterized and reconverted
|
When ``--force-ocr`` is used, all pages are rasterized and reconverted
|
||||||
@@ -144,10 +144,9 @@ set, the document cannot be viewed without the password.
|
|||||||
Either way, OCRmyPDF does not remove passwords from PDFs and exits with
|
Either way, OCRmyPDF does not remove passwords from PDFs and exits with
|
||||||
an error on encountering them.
|
an error on encountering them.
|
||||||
|
|
||||||
``qpdf``, one of OCRmyPDF's dependencies, can remove passwords. If the
|
``qpdf`` can remove passwords. If the owner and user password are set, a
|
||||||
owner and user password are set, a password is required for ``qpdf``. If
|
password is required for ``qpdf``. If only the owner password is set, then the
|
||||||
only the owner password is set, then the password can be stripped, even
|
password can be stripped, even if one does not have the owner password.
|
||||||
if one does not have the owner password.
|
|
||||||
|
|
||||||
After OCR is applied, password protection is not permitted on PDF/A
|
After OCR is applied, password protection is not permitted on PDF/A
|
||||||
documents but the file can be converted to regular PDF.
|
documents but the file can be converted to regular PDF.
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
===========
|
||||||
|
Performance
|
||||||
|
===========
|
||||||
|
|
||||||
|
Some users have noticed that current versions of OCRmyPDF do not run as quickly
|
||||||
|
as some older versions (specifically 6.x and older). This is because OCRmyPDF
|
||||||
|
added image optimization as a postprocessing step, and it is enabled by default.
|
||||||
|
|
||||||
|
Speed
|
||||||
|
=====
|
||||||
|
|
||||||
|
If running OCRmyPDF quickly is your main goal, you can use settings such as:
|
||||||
|
|
||||||
|
* ``--optimize 0`` to disable file size optimization
|
||||||
|
* ``--output-type pdf`` to disable PDF/A generation
|
||||||
|
* ``--fast-web-view 0`` to disable fast web view optimization
|
||||||
|
* ``--skip-big`` to skip large images, if some pages have large images
|
||||||
|
|
||||||
|
You can also avoid:
|
||||||
|
|
||||||
|
* ``--force-ocr``
|
||||||
|
* Image preprocessing
|
||||||
+68
-12
@@ -2,23 +2,79 @@
|
|||||||
Plugins
|
Plugins
|
||||||
=======
|
=======
|
||||||
|
|
||||||
You can use plugins to customize the behavior of OCRmyPDF at certain
|
You can use plugins to customize the behavior of OCRmyPDF at certain points of
|
||||||
points of interest.
|
interest.
|
||||||
|
|
||||||
Currently, it is possible to: - override the decision for whether or not
|
Currently, it is possible to:
|
||||||
to perform OCR on a particular file - modify the image is about to be
|
|
||||||
sent for OCR
|
- add new command line arguments
|
||||||
|
- override the decision for whether or not to perform OCR on a particular file
|
||||||
|
- modify the image is about to be sent for OCR
|
||||||
|
- modify the page image before it is converted to PDF
|
||||||
|
|
||||||
|
OCRmyPDF plugins are based on the Python ``pluggy`` package and conform to its
|
||||||
|
conventions. Note that: plugins installed with as setuptools entrypoints are
|
||||||
|
not checked currently, because OCRmyPDF assumes you may not want to enable
|
||||||
|
plugins for all files. Also, plugins must be functions, not classes.
|
||||||
|
|
||||||
How plugins are imported
|
How plugins are imported
|
||||||
========================
|
========================
|
||||||
|
|
||||||
Plugins are imported on demand, by the OCRmyPDF worker process that
|
Plugins are imported on demand, by the OCRmyPDF worker process that needs to use
|
||||||
needs to use them. As such, plugins cannot share state with each other,
|
them. As such, plugins cannot share state with other plugins, cannot rely on
|
||||||
and will be imported many times, once for each worker process.
|
their module's or the interpreter's global state, and should expect asynchronous
|
||||||
|
copies of themselves to be running. Plugins can write intermediate files to the
|
||||||
|
folder specified in ``options.work_folder``.
|
||||||
|
|
||||||
Plugins currently cannot override the same hook.
|
Plugins should work whether executed in threads or processes.
|
||||||
|
|
||||||
How plugins are invoked
|
Script plugins
|
||||||
=======================
|
==============
|
||||||
|
|
||||||
Plugins may be called from the command line:
|
Script plugins may be called from the command line, by specifying the name of a file.
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ocrmypdf --plugin example_plugin.py input.pdf output.pdf
|
||||||
|
|
||||||
|
Multiple plugins may be called by issuing the ``--plugin`` argument multiple times.
|
||||||
|
|
||||||
|
Packaged plugins
|
||||||
|
================
|
||||||
|
|
||||||
|
Installed plugins may be installed into the same virtual environment as OCRmyPDF
|
||||||
|
is installed into. They may be invoked using Python standard module naming.
|
||||||
|
|
||||||
|
.. code-block:: bash
|
||||||
|
|
||||||
|
ocrmypdf --plugin ocrmypdf_fancypants.pockets.contents input.pdf output.pdf
|
||||||
|
|
||||||
|
OCRmyPDF does not automatically import plugins, because the assumption is that
|
||||||
|
plugins affect different files differently and you may not want them activated
|
||||||
|
all the time. The command line or ``ocrmypdf.ocr(plugin='...')`` must call
|
||||||
|
for them.
|
||||||
|
|
||||||
|
Third parties that wish to distribute packages for ocrmypdf should package them
|
||||||
|
as packaged plugins, and these modules should begin with the name ``ocrmypdf_``
|
||||||
|
similar to ``pytest`` packages such as ``pytest-cov`` (the package) and
|
||||||
|
``pytest_cov`` (the module).
|
||||||
|
|
||||||
|
Plugin hooks
|
||||||
|
============
|
||||||
|
|
||||||
|
A plugin may provide the following hooks. Hooks should be decorated with
|
||||||
|
``ocrmypdf.hookimpl``, for example:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
from ocrmpydf import hookimpl
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def add_options(parser):
|
||||||
|
pass
|
||||||
|
|
||||||
|
The following is a complete list of hooks that may be installed and when
|
||||||
|
they are called.
|
||||||
|
|
||||||
|
.. automodule:: ocrmypdf.pluginspec
|
||||||
|
:members:
|
||||||
|
|||||||
+107
-3
@@ -5,14 +5,118 @@ Release notes
|
|||||||
OCRmyPDF uses `semantic versioning <http://semver.org/>`__ for its
|
OCRmyPDF uses `semantic versioning <http://semver.org/>`__ for its
|
||||||
command line interface and its public API.
|
command line interface and its public API.
|
||||||
|
|
||||||
The ``ocrmypdf`` package may now be imported. The public API may be
|
OCRmyPDF's output messages are not considered part of the stable interface -
|
||||||
useful in scripts that launch OCRmyPDF processes or that wish to use
|
that is, output messages may be improved at any release level, so parsing them
|
||||||
some of its features for working with PDFs.
|
may be unreliable. Use the API to depend on precise behavior.
|
||||||
|
|
||||||
|
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||||
|
wish to use some of its features for working with PDFs.
|
||||||
|
|
||||||
Note that it is licensed under GPLv3, so scripts that
|
Note that it is licensed under GPLv3, so scripts that
|
||||||
``import ocrmypdf`` and are released publicly should probably also be
|
``import ocrmypdf`` and are released publicly should probably also be
|
||||||
licensed under GPLv3.
|
licensed under GPLv3.
|
||||||
|
|
||||||
|
v10.2.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Update Docker image to use Ubuntu 20.04.
|
||||||
|
- Fixed issue PDF/A acquires title "Untitled" after conversion. (#582)
|
||||||
|
- Fixed a problem where, when using ``--pdf-renderer hocr``, some text would
|
||||||
|
be missing from the output when using a more recent version of Tesseract.
|
||||||
|
Tesseract began adding more detailed markup about the semantics of text
|
||||||
|
that our HOCR transform did not recognize, so it ignored them. This option is
|
||||||
|
not the default. If necessary ``--redo-ocr`` also redoing OCR to fix such issues.
|
||||||
|
- Fixed an error in Python 3.9 beta, due to removal of deprecated
|
||||||
|
``Element.getchildren()``. (#584)
|
||||||
|
- Implemented support using the API with ``BytesIO`` and other file stream objects.
|
||||||
|
(#545)
|
||||||
|
|
||||||
|
v10.1.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed ``OMP_THREAD_LIMIT`` set to invalid value error messages on some input
|
||||||
|
files. (The error was harmless, apart from less than optimal performance in
|
||||||
|
some cases.)
|
||||||
|
|
||||||
|
v10.1.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Previously, we ``--clean-final`` would cause an unpaper-cleaned page image to
|
||||||
|
be produced twice, which was necessary in some cases but not in general. We
|
||||||
|
now take this optimization opportunity and reuse the image if possible.
|
||||||
|
- We now provide PNG files as input to unpaper, since it accepts them, instead
|
||||||
|
of generating PPM files which can be very large. This can improve performance
|
||||||
|
and temporary disk usage.
|
||||||
|
- Documentation updated for plugins.
|
||||||
|
|
||||||
|
v10.0.1
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Fixed regression when ``-l lang1+lang2`` is used from command line.
|
||||||
|
|
||||||
|
v10.0.0
|
||||||
|
=======
|
||||||
|
|
||||||
|
**Breaking changes**
|
||||||
|
|
||||||
|
- Support for pdfminer.six version 20181108 has been dropped, along with a
|
||||||
|
monkeypatch that made this version work.
|
||||||
|
- Output messages are now displayed in color (when supported by the terminal)
|
||||||
|
and prefixes describing the severity of the message are removed. As such
|
||||||
|
programs that parse OCRmyPDF's log message will need to be revised. (Please
|
||||||
|
consider using OCRmyPDF as a library instead.)
|
||||||
|
- The minimum version for certain dependencies has increased.
|
||||||
|
- Many API changes; see developer changes.
|
||||||
|
- The Python libraries pluggy and coloredlogs are now required.
|
||||||
|
|
||||||
|
**New features and improvements**
|
||||||
|
|
||||||
|
- PDF page scanning is now parallelized across CPUs, speeding up this phase
|
||||||
|
dramatically for files with a high page counts.
|
||||||
|
- PDF page scanning is optimized, addressing some performance regressions.
|
||||||
|
- PDF page scanning is no longer run on pages that are not selected when the
|
||||||
|
``--pages`` argument is used.
|
||||||
|
- PDF page scanning is now independent of Ghostscript, ending our past reliance
|
||||||
|
on this occasionally unstable feature in Ghostscript.
|
||||||
|
- A plugin architecture has been added, currently allowing one to more easily
|
||||||
|
use a different OCR engine or PDF renderer from Tesseract and Ghostscript,
|
||||||
|
respectively. A plugin can also override some decisions, such changing
|
||||||
|
the OCR settings after initial scanning.
|
||||||
|
- Colored log messages.
|
||||||
|
|
||||||
|
**Developer changes**
|
||||||
|
|
||||||
|
- The test spoofing mechanism, used to test correct handling of failures in
|
||||||
|
Tesseract and Ghostscript, has been removed in favor of using plugins for
|
||||||
|
testing. The spoofing mechanism was fairly complex and required many special
|
||||||
|
hacks for Windows.
|
||||||
|
- Code describing the resolution in DPI of images was refactored into a
|
||||||
|
``ocrmypdf.helpers.Resolution`` class.
|
||||||
|
- The module ``ocrmypdf._exec`` is now private to OCRmyPDF.
|
||||||
|
- The ``ocrmypdf.hocrtransform`` module has been updated to follow PEP8 naming
|
||||||
|
conventions.
|
||||||
|
- Ghostscript is no longer used for finding the location of text in PDFs, and
|
||||||
|
APIs related to this feature have been removed.
|
||||||
|
- Lots of internal reorganization to support plugins.
|
||||||
|
|
||||||
|
v9.8.2
|
||||||
|
======
|
||||||
|
|
||||||
|
- Fixed an issue where OCRmyPDF would ignore text inside Form XObject when
|
||||||
|
making certain decisions about whether a document already had text.
|
||||||
|
- Fixed file size increase warning to take overhead of small files into account.
|
||||||
|
- Added instructions for installing on Cygwin.
|
||||||
|
|
||||||
|
v9.8.1
|
||||||
|
======
|
||||||
|
|
||||||
|
- Fixed an issue where unexpected files in the ``%PROGRAMFILES%\gs`` directory
|
||||||
|
(Windows) caused an exception.
|
||||||
|
- Mark pdfminer.six 20200517 as supported.
|
||||||
|
- If jbig2enc is missing and optimization is requested, a warning is issued
|
||||||
|
instead of an error, which was the intended behavior.
|
||||||
|
- Documentation updates.
|
||||||
|
|
||||||
v9.8.0
|
v9.8.0
|
||||||
======
|
======
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,53 @@
|
|||||||
|
# © 2020 James R Barlow: https://github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This program is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# This program is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
from ocrmypdf import hookimpl
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def add_options(parser):
|
||||||
|
parser.add_argument('--grayscale-ocr', action='store_true')
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def prepare(options):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def validate(pdfinfo, options):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def filter_ocr_image(page, image):
|
||||||
|
if page.options.grayscale_ocr:
|
||||||
|
log.info("graying")
|
||||||
|
return image.convert('L')
|
||||||
|
return image
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def filter_page_image(page, image_filename):
|
||||||
|
output = image_filename.with_suffix('.jpg')
|
||||||
|
with Image.open(image_filename) as im:
|
||||||
|
im.save(output)
|
||||||
|
return output
|
||||||
+5
-5
@@ -33,12 +33,12 @@ import ocrmypdf
|
|||||||
|
|
||||||
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
|
INPUT_DIRECTORY = os.getenv('OCR_INPUT_DIRECTORY', '/input')
|
||||||
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
|
OUTPUT_DIRECTORY = os.getenv('OCR_OUTPUT_DIRECTORY', '/output')
|
||||||
OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', False))
|
OUTPUT_DIRECTORY_YEAR_MONTH = bool(os.getenv('OCR_OUTPUT_DIRECTORY_YEAR_MONTH', ''))
|
||||||
ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', False))
|
ON_SUCCESS_DELETE = bool(os.getenv('OCR_ON_SUCCESS_DELETE', ''))
|
||||||
DESKEW = bool(os.getenv('OCR_DESKEW', False))
|
DESKEW = bool(os.getenv('OCR_DESKEW', ''))
|
||||||
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
OCR_JSON_SETTINGS = json.loads(os.getenv('OCR_JSON_SETTINGS', '{}'))
|
||||||
POLL_NEW_FILE_SECONDS = os.getenv('OCR_POLL_NEW_FILE_SECONDS', 1)
|
POLL_NEW_FILE_SECONDS = int(os.getenv('OCR_POLL_NEW_FILE_SECONDS', '1'))
|
||||||
USE_POLLING = bool(os.getenv('OCR_USE_POLLING', False))
|
USE_POLLING = bool(os.getenv('OCR_USE_POLLING', ''))
|
||||||
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper()
|
LOGLEVEL = os.getenv('OCR_LOGLEVEL', 'INFO').upper()
|
||||||
PATTERNS = ['*.pdf']
|
PATTERNS = ['*.pdf']
|
||||||
|
|
||||||
|
|||||||
@@ -2,9 +2,11 @@
|
|||||||
# setup.py lists a separate set of requirements that are looser to simplify
|
# setup.py lists a separate set of requirements that are looser to simplify
|
||||||
# installation
|
# installation
|
||||||
cffi == 1.14.0
|
cffi == 1.14.0
|
||||||
img2pdf == 0.3.4
|
coloredlogs == 14.0 # technically optional
|
||||||
pdfminer.six == 20200402
|
img2pdf == 0.3.6
|
||||||
pikepdf == 1.11.1
|
pdfminer.six == 20200517
|
||||||
Pillow == 7.1.1
|
pikepdf == 1.15.1
|
||||||
reportlab == 3.5.34
|
pluggy == 0.13.1
|
||||||
tqdm == 4.45.0
|
Pillow == 7.1.2
|
||||||
|
reportlab == 3.5.42
|
||||||
|
tqdm == 4.46.1
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
pytest >= 5.0.0
|
pytest >= 5.0.0
|
||||||
pytest-helpers-namespace >= 2019.1.8
|
pytest-helpers-namespace >= 2019.1.8
|
||||||
pytest-xdist >= 1.31.0
|
pytest-xdist >= 1.31.0
|
||||||
pytest-cov >= 2.8.0
|
pytest-cov >= 2.10.0
|
||||||
python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3
|
python-xmp-toolkit == 2.0.1 # requires apt-get install libexempi3
|
||||||
# or brew install exempi
|
# or brew install exempi
|
||||||
#PyMuPDF == 1.13.4 # optional
|
#PyMuPDF == 1.13.4 # optional
|
||||||
|
|||||||
@@ -1 +1 @@
|
|||||||
watchdog >= 0.8.2, < 1.0
|
watchdog == 0.10.2
|
||||||
|
|||||||
@@ -23,7 +23,7 @@ force_grid_wrap=0
|
|||||||
use_parentheses=True
|
use_parentheses=True
|
||||||
line_length=88
|
line_length=88
|
||||||
known_first_party = ocrmypdf
|
known_first_party = ocrmypdf
|
||||||
known_third_party = PIL,_cffi_backend,cffi,flask,gs,img2pdf,pdfminer,pikepdf,pkg_resources,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug
|
known_third_party = PIL,_cffi_backend,cffi,flask,img2pdf,pdfminer,pikepdf,pkg_resources,pluggy,pytest,reportlab,setuptools,sphinx_rtd_theme,tqdm,watchdog,werkzeug
|
||||||
|
|
||||||
[metadata]
|
[metadata]
|
||||||
license_file = LICENSE
|
license_file = LICENSE
|
||||||
|
|||||||
@@ -27,22 +27,6 @@ if sys.version_info < (3, 6):
|
|||||||
print("Python 3.6 or newer is required", file=sys.stderr)
|
print("Python 3.6 or newer is required", file=sys.stderr)
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
|
|
||||||
# pylint: disable=w0613
|
|
||||||
|
|
||||||
|
|
||||||
command = next((arg for arg in sys.argv[1:] if not arg.startswith('-')), '')
|
|
||||||
if command.startswith('install') or command in [
|
|
||||||
'check',
|
|
||||||
'test',
|
|
||||||
'nosetests',
|
|
||||||
'easy_install',
|
|
||||||
]:
|
|
||||||
forced = '--force' in sys.argv
|
|
||||||
if forced:
|
|
||||||
print("The argument --force is deprecated. Please discontinue use.")
|
|
||||||
|
|
||||||
|
|
||||||
if 'upload' in sys.argv[1:]:
|
if 'upload' in sys.argv[1:]:
|
||||||
print('Use twine to upload the package - setup.py upload is insecure')
|
print('Use twine to upload the package - setup.py upload is insecure')
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
@@ -62,7 +46,7 @@ setup(
|
|||||||
long_description_content_type='text/markdown',
|
long_description_content_type='text/markdown',
|
||||||
url='https://github.com/jbarlow83/OCRmyPDF',
|
url='https://github.com/jbarlow83/OCRmyPDF',
|
||||||
author='James R. Barlow',
|
author='James R. Barlow',
|
||||||
author_email='jim@purplerock.ca',
|
author_email='james@purplerock.ca',
|
||||||
packages=find_packages('src', exclude=["tests", "tests.*"]),
|
packages=find_packages('src', exclude=["tests", "tests.*"]),
|
||||||
package_dir={'': 'src'},
|
package_dir={'': 'src'},
|
||||||
keywords=['PDF', 'OCR', 'optical character recognition', 'PDF/A', 'scanning'],
|
keywords=['PDF', 'OCR', 'optical character recognition', 'PDF/A', 'scanning'],
|
||||||
@@ -95,12 +79,13 @@ setup(
|
|||||||
use_scm_version={'version_scheme': 'post-release'},
|
use_scm_version={'version_scheme': 'post-release'},
|
||||||
cffi_modules=['src/ocrmypdf/lib/compile_leptonica.py:ffibuilder'],
|
cffi_modules=['src/ocrmypdf/lib/compile_leptonica.py:ffibuilder'],
|
||||||
install_requires=[
|
install_requires=[
|
||||||
'chardet >= 3.0.4, < 4', # unlisted requirement of pdfminer.six 20181108
|
|
||||||
'cffi >= 1.9.1', # must be a setup and install requirement
|
'cffi >= 1.9.1', # must be a setup and install requirement
|
||||||
|
'coloredlogs >= 14.0', # strictly optional
|
||||||
'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely
|
'img2pdf >= 0.3.0, < 0.4', # pure Python, so track HEAD closely
|
||||||
'pdfminer.six >= 20181108, <= 20200402',
|
'pdfminer.six >= 20191110, <= 20200517',
|
||||||
'pikepdf >= 1.8.1, < 2',
|
'pikepdf >= 1.14.0, < 2',
|
||||||
'Pillow >= 6.2.0',
|
'Pillow >= 7.0.0',
|
||||||
|
'pluggy >= 0.13.0',
|
||||||
'reportlab >= 3.3.0', # oldest released version with sane image handling
|
'reportlab >= 3.3.0', # oldest released version with sane image handling
|
||||||
'tqdm >= 4',
|
'tqdm >= 4',
|
||||||
],
|
],
|
||||||
|
|||||||
@@ -15,10 +15,13 @@
|
|||||||
# You should have received a copy of the GNU General Public License
|
# You should have received a copy of the GNU General Public License
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
from . import helpers, hocrtransform, leptonica, pdfa, pdfinfo
|
|
||||||
from ._version import PROGRAM_NAME, __version__
|
from pluggy import HookimplMarker as _HookimplMarker
|
||||||
from .api import Verbosity, configure_logging, ocr
|
|
||||||
from .exceptions import (
|
from ocrmypdf import helpers, hocrtransform, leptonica, pdfa, pdfinfo
|
||||||
|
from ocrmypdf._version import PROGRAM_NAME, __version__
|
||||||
|
from ocrmypdf.api import Verbosity, configure_logging, ocr
|
||||||
|
from ocrmypdf.exceptions import (
|
||||||
BadArgsError,
|
BadArgsError,
|
||||||
DpiError,
|
DpiError,
|
||||||
EncryptedPdfError,
|
EncryptedPdfError,
|
||||||
@@ -33,3 +36,6 @@ from .exceptions import (
|
|||||||
TesseractConfigError,
|
TesseractConfigError,
|
||||||
UnsupportedImageFormatError,
|
UnsupportedImageFormatError,
|
||||||
)
|
)
|
||||||
|
from ocrmypdf.pluginspec import OcrEngine, OrientationConfidence
|
||||||
|
|
||||||
|
hookimpl = _HookimplMarker('ocrmypdf')
|
||||||
|
|||||||
+15
-12
@@ -19,18 +19,20 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
from multiprocessing import set_start_method
|
||||||
|
|
||||||
from . import __version__
|
from ocrmypdf import __version__
|
||||||
from ._jobcontext import make_logger
|
from ocrmypdf._plugin_manager import get_parser_options_plugins
|
||||||
from ._sync import run_pipeline
|
from ocrmypdf._sync import run_pipeline
|
||||||
from ._validation import check_closed_streams, check_options
|
from ocrmypdf._validation import check_closed_streams, check_options
|
||||||
from .api import Verbosity, configure_logging
|
from ocrmypdf.api import Verbosity, configure_logging
|
||||||
from .cli import parser
|
from ocrmypdf.exceptions import BadArgsError, ExitCode, MissingDependencyError
|
||||||
from .exceptions import BadArgsError, ExitCode, MissingDependencyError
|
|
||||||
|
log = logging.getLogger('ocrmypdf')
|
||||||
|
|
||||||
|
|
||||||
def run(args=None):
|
def run(args=None):
|
||||||
options = parser.parse_args(args=args)
|
_parser, options, plugin_manager = get_parser_options_plugins(args=args)
|
||||||
|
|
||||||
if not check_closed_streams(options):
|
if not check_closed_streams(options):
|
||||||
return ExitCode.bad_args
|
return ExitCode.bad_args
|
||||||
@@ -47,10 +49,9 @@ def run(args=None):
|
|||||||
configure_logging(
|
configure_logging(
|
||||||
verbosity, progress_bar_friendly=options.progress_bar, manage_root_logger=True
|
verbosity, progress_bar_friendly=options.progress_bar, manage_root_logger=True
|
||||||
)
|
)
|
||||||
log = make_logger('ocrmypdf')
|
log.debug('ocrmypdf %s', __version__)
|
||||||
log.debug('ocrmypdf ' + __version__)
|
|
||||||
try:
|
try:
|
||||||
check_options(options)
|
check_options(options, plugin_manager)
|
||||||
except ValueError as e:
|
except ValueError as e:
|
||||||
log.error(e)
|
log.error(e)
|
||||||
return ExitCode.bad_args
|
return ExitCode.bad_args
|
||||||
@@ -61,9 +62,11 @@ def run(args=None):
|
|||||||
log.error(e)
|
log.error(e)
|
||||||
return ExitCode.missing_dependency
|
return ExitCode.missing_dependency
|
||||||
|
|
||||||
result = run_pipeline(options=options)
|
result = run_pipeline(options=options, plugin_manager=plugin_manager)
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
if __name__ == '__main__':
|
if __name__ == '__main__':
|
||||||
|
if sys.platform == 'darwin' and sys.version_info < (3, 8):
|
||||||
|
set_start_method('spawn') # see python bpo-33725
|
||||||
sys.exit(run())
|
sys.exit(run())
|
||||||
|
|||||||
@@ -0,0 +1,135 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This file is part of OCRmyPDF.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import logging.handlers
|
||||||
|
import multiprocessing
|
||||||
|
import os
|
||||||
|
import signal
|
||||||
|
import sys
|
||||||
|
import threading
|
||||||
|
from multiprocessing import Pool as ProcessPool
|
||||||
|
from multiprocessing.dummy import Pool as ThreadPool
|
||||||
|
from typing import Callable, Iterable, Optional
|
||||||
|
|
||||||
|
from tqdm import tqdm
|
||||||
|
|
||||||
|
|
||||||
|
def log_listener(queue):
|
||||||
|
"""Listen to the worker processes and forward the messages to logging
|
||||||
|
|
||||||
|
For simplicity this is a thread rather than a process. Only one process
|
||||||
|
should actually write to sys.stderr or whatever we're using, so if this is
|
||||||
|
made into a process the main application needs to be directed to it.
|
||||||
|
|
||||||
|
See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes
|
||||||
|
"""
|
||||||
|
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
record = queue.get()
|
||||||
|
if record is None:
|
||||||
|
break
|
||||||
|
logger = logging.getLogger(record.name)
|
||||||
|
logger.handle(record)
|
||||||
|
except Exception: # pylint: disable=broad-except
|
||||||
|
import traceback # pylint: disable=import-outside-toplevel
|
||||||
|
|
||||||
|
print("Logging problem", file=sys.stderr)
|
||||||
|
traceback.print_exc(file=sys.stderr)
|
||||||
|
|
||||||
|
|
||||||
|
def process_init(queue, user_init):
|
||||||
|
"""Initialize a process pool worker"""
|
||||||
|
|
||||||
|
# Ignore SIGINT (our parent process will kill us gracefully)
|
||||||
|
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
||||||
|
|
||||||
|
# Reconfigure the root logger for this process to send all messages to a queue
|
||||||
|
h = logging.handlers.QueueHandler(queue)
|
||||||
|
root = logging.getLogger()
|
||||||
|
root.handlers = []
|
||||||
|
root.addHandler(h)
|
||||||
|
|
||||||
|
if user_init:
|
||||||
|
user_init()
|
||||||
|
|
||||||
|
|
||||||
|
def thread_init(_queue, user_init):
|
||||||
|
if user_init:
|
||||||
|
user_init()
|
||||||
|
|
||||||
|
|
||||||
|
def exec_progress_pool(
|
||||||
|
*,
|
||||||
|
use_threads: bool,
|
||||||
|
max_workers: int,
|
||||||
|
tqdm_kwargs: dict,
|
||||||
|
task_initializer: Optional[Callable] = None,
|
||||||
|
task: Optional[Callable] = None,
|
||||||
|
task_arguments: Optional[Iterable] = None,
|
||||||
|
task_finished: Optional[Callable] = None,
|
||||||
|
):
|
||||||
|
log_queue = multiprocessing.Queue(-1)
|
||||||
|
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
||||||
|
|
||||||
|
if use_threads:
|
||||||
|
pool_class = ThreadPool
|
||||||
|
initializer = thread_init
|
||||||
|
else:
|
||||||
|
pool_class = ProcessPool
|
||||||
|
initializer = process_init
|
||||||
|
listener.start()
|
||||||
|
|
||||||
|
with tqdm(**tqdm_kwargs) as pbar:
|
||||||
|
pool = pool_class(
|
||||||
|
processes=max_workers,
|
||||||
|
initializer=initializer,
|
||||||
|
initargs=(log_queue, task_initializer),
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
results = pool.imap_unordered(task, task_arguments)
|
||||||
|
while True:
|
||||||
|
try:
|
||||||
|
result = results.next()
|
||||||
|
if task_finished:
|
||||||
|
task_finished(result, pbar)
|
||||||
|
else:
|
||||||
|
pbar.update()
|
||||||
|
except StopIteration:
|
||||||
|
break
|
||||||
|
except KeyboardInterrupt:
|
||||||
|
# Terminate pool so we exit instantly
|
||||||
|
pool.terminate()
|
||||||
|
# Don't try listener.join() here, will deadlock
|
||||||
|
raise
|
||||||
|
except Exception:
|
||||||
|
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
||||||
|
# Unless inside pytest, exit immediately because no one wants
|
||||||
|
# to wait for child processes to finalize results that will be
|
||||||
|
# thrown away. Inside pytest, we want child processes to exit
|
||||||
|
# cleanly so that they output an error messages or coverage data
|
||||||
|
# we need from them.
|
||||||
|
pool.terminate()
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
# Terminate log listener
|
||||||
|
log_queue.put_nowait(None)
|
||||||
|
pool.close()
|
||||||
|
pool.join()
|
||||||
|
|
||||||
|
listener.join()
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This file is part of OCRmyPDF.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
"""Manage third party executables"""
|
||||||
@@ -20,9 +20,6 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import warnings
|
|
||||||
from contextlib import suppress
|
|
||||||
from functools import lru_cache
|
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
@@ -31,10 +28,11 @@ from subprocess import PIPE, CalledProcessError
|
|||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ..exceptions import MissingDependencyError, SubprocessOutputError
|
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
||||||
from . import get_version, run
|
from ocrmypdf.helpers import Resolution
|
||||||
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
gslog = logging.getLogger()
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
GS = 'gs'
|
GS = 'gs'
|
||||||
if os.name == 'nt':
|
if os.name == 'nt':
|
||||||
@@ -57,12 +55,11 @@ if os.name == 'nt':
|
|||||||
GS = Path(GS).stem
|
GS = Path(GS).stem
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=1)
|
|
||||||
def version():
|
def version():
|
||||||
return get_version(GS)
|
return get_version(GS)
|
||||||
|
|
||||||
|
|
||||||
def jpeg_passthrough_available():
|
def jpeg_passthrough_available() -> bool:
|
||||||
"""Returns True if the installed version of Ghostscript supports JPEG passthru
|
"""Returns True if the installed version of Ghostscript supports JPEG passthru
|
||||||
|
|
||||||
Prior to 9.23, Ghostscript decode and re-encoded JPEGs internally. In 9.23
|
Prior to 9.23, Ghostscript decode and re-encoded JPEGs internally. In 9.23
|
||||||
@@ -79,94 +76,25 @@ def jpeg_passthrough_available():
|
|||||||
return version() >= '9.24'
|
return version() >= '9.24'
|
||||||
|
|
||||||
|
|
||||||
def _gs_error_reported(stream):
|
def _gs_error_reported(stream) -> bool:
|
||||||
return re.search(r'error', stream, flags=re.IGNORECASE)
|
return re.search(r'error', stream, flags=re.IGNORECASE)
|
||||||
|
|
||||||
|
|
||||||
def extract_text(input_file, pageno=1):
|
|
||||||
"""Use the txtwrite device to get text layout information out
|
|
||||||
|
|
||||||
For details on options of -dTextFormat see
|
|
||||||
https://www.ghostscript.com/doc/current/VectorDevices.htm#TXT
|
|
||||||
|
|
||||||
Format is like
|
|
||||||
<page>
|
|
||||||
<line>
|
|
||||||
<span bbox="left top right bottom" font="..." size="...">
|
|
||||||
<char bbox="...." c="X"/>
|
|
||||||
|
|
||||||
:param pageno: number of page to extract, or all pages if None
|
|
||||||
:return: XML-ish text representation in bytes
|
|
||||||
"""
|
|
||||||
|
|
||||||
if pageno is not None:
|
|
||||||
pages = ['-dFirstPage=%i' % pageno, '-dLastPage=%i' % pageno]
|
|
||||||
else:
|
|
||||||
pages = []
|
|
||||||
|
|
||||||
# Note due to bug https://bugs.ghostscript.com/show_bug.cgi?id=701971
|
|
||||||
# Ghostscript <= 9.50 will truncate output unless we write to stdout, so
|
|
||||||
# don't write to a file.
|
|
||||||
args_gs = (
|
|
||||||
[
|
|
||||||
GS,
|
|
||||||
'-dQUIET',
|
|
||||||
'-dSAFER',
|
|
||||||
'-dBATCH',
|
|
||||||
'-dNOPAUSE',
|
|
||||||
'-sDEVICE=txtwrite',
|
|
||||||
'-dTextFormat=0',
|
|
||||||
]
|
|
||||||
+ pages
|
|
||||||
+ ['-o', '-', fspath(input_file), "-sstdout=%stderr"]
|
|
||||||
)
|
|
||||||
|
|
||||||
try:
|
|
||||||
p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True)
|
|
||||||
except CalledProcessError as e:
|
|
||||||
raise SubprocessOutputError(
|
|
||||||
'Ghostscript text extraction failed\n%s\n%s'
|
|
||||||
% (input_file, e.stderr.decode(errors='replace'))
|
|
||||||
)
|
|
||||||
|
|
||||||
return p.stdout
|
|
||||||
|
|
||||||
|
|
||||||
def rasterize_pdf(
|
def rasterize_pdf(
|
||||||
input_file,
|
input_file: os.PathLike,
|
||||||
output_file,
|
output_file: os.PathLike,
|
||||||
xres,
|
*,
|
||||||
yres,
|
raster_device: str,
|
||||||
raster_device,
|
raster_dpi: Resolution,
|
||||||
log,
|
pageno: int = 1,
|
||||||
pageno=1,
|
page_dpi: Resolution = None,
|
||||||
page_dpi=None,
|
rotation: int = None,
|
||||||
rotation=None,
|
filter_vector: bool = False,
|
||||||
filter_vector=False,
|
|
||||||
):
|
):
|
||||||
"""Rasterize one page of a PDF at resolution (xres, yres) in canvas units.
|
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units."""
|
||||||
|
raster_dpi = raster_dpi.round(6)
|
||||||
The image is sized to match the integer pixels dimensions implied by
|
|
||||||
(xres, yres) even if those numbers are noninteger. The image's DPI will
|
|
||||||
be overridden with the values in page_dpi.
|
|
||||||
|
|
||||||
:param input_file: pathlike
|
|
||||||
:param output_file: pathlike
|
|
||||||
:param xres: resolution at which to rasterize page
|
|
||||||
:param yres:
|
|
||||||
:param raster_device:
|
|
||||||
:param log:
|
|
||||||
:param pageno: page number to rasterize (beginning at page 1)
|
|
||||||
:param page_dpi: resolution tuple (x, y) overriding output image DPI
|
|
||||||
:param rotation: 0, 90, 180, 270: clockwise angle to rotate page
|
|
||||||
:param filter_vector: if True, remove vector graphics objects
|
|
||||||
:return:
|
|
||||||
"""
|
|
||||||
res = round(xres, 6), round(yres, 6)
|
|
||||||
if not page_dpi:
|
if not page_dpi:
|
||||||
page_dpi = res
|
page_dpi = raster_dpi
|
||||||
if not log:
|
|
||||||
log = gslog
|
|
||||||
|
|
||||||
args_gs = (
|
args_gs = (
|
||||||
[
|
[
|
||||||
@@ -178,7 +106,7 @@ def rasterize_pdf(
|
|||||||
f'-sDEVICE={raster_device}',
|
f'-sDEVICE={raster_device}',
|
||||||
f'-dFirstPage={pageno}',
|
f'-dFirstPage={pageno}',
|
||||||
f'-dLastPage={pageno}',
|
f'-dLastPage={pageno}',
|
||||||
f'-r{res[0]:f}x{res[1]:f}',
|
f'-r{raster_dpi.x:f}x{raster_dpi.y:f}',
|
||||||
]
|
]
|
||||||
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
+ (['-dFILTERVECTOR'] if filter_vector else [])
|
||||||
+ [
|
+ [
|
||||||
@@ -191,7 +119,6 @@ def rasterize_pdf(
|
|||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
|
||||||
log.debug(args_gs)
|
|
||||||
try:
|
try:
|
||||||
p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True)
|
p = run(args_gs, stdout=PIPE, stderr=PIPE, check=True)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
@@ -216,43 +143,17 @@ def rasterize_pdf(
|
|||||||
elif rotation == 270:
|
elif rotation == 270:
|
||||||
im = im.transpose(Image.ROTATE_270)
|
im = im.transpose(Image.ROTATE_270)
|
||||||
if rotation % 180 == 90:
|
if rotation % 180 == 90:
|
||||||
page_dpi = page_dpi[1], page_dpi[0]
|
page_dpi = page_dpi.flip_axis()
|
||||||
im.save(fspath(output_file), dpi=page_dpi)
|
im.save(fspath(output_file), dpi=page_dpi)
|
||||||
|
|
||||||
|
|
||||||
def generate_pdfa(
|
def generate_pdfa(
|
||||||
pdf_pages,
|
pdf_pages,
|
||||||
output_file,
|
output_file: os.PathLike,
|
||||||
compression,
|
compression: str,
|
||||||
log,
|
pdf_version: str = '1.5',
|
||||||
threads=None, # deprecated parameter
|
pdfa_part: str = '2',
|
||||||
pdf_version='1.5',
|
|
||||||
pdfa_part='2',
|
|
||||||
):
|
):
|
||||||
"""Generate a PDF/A.
|
|
||||||
|
|
||||||
The pdf_pages, a list files, will be merged into output_file. One or more
|
|
||||||
PDF files may be merged. One of the files in this list must be a pdfmark
|
|
||||||
file that provides Ghostscript with details on how to perform the PDF/A
|
|
||||||
conversion. By default with we pick PDF/A-2b, but this works for 1 or 3.
|
|
||||||
|
|
||||||
compression can be 'jpeg', 'lossless', or an empty string. In 'jpeg',
|
|
||||||
Ghostscript is instructed to convert color and grayscale images to DCT
|
|
||||||
(JPEG encoding). In 'lossless' Ghostscript is told to convert images to
|
|
||||||
Flate (lossless/PNG). If the parameter is omitted Ghostscript is left to
|
|
||||||
make its own decisions about how to encode images; it appears to use a
|
|
||||||
heuristic to decide how to encode images. As of Ghostscript 9.25, we
|
|
||||||
support passthrough JPEG which allows Ghostscript to avoid transcoding
|
|
||||||
images entirely. (The feature was added in 9.23 but broken, and the 9.24
|
|
||||||
release of Ghostscript had regressions, so we don't support it until 9.25.)
|
|
||||||
"""
|
|
||||||
if not log:
|
|
||||||
log = gslog
|
|
||||||
if threads is not None:
|
|
||||||
warnings.warn(
|
|
||||||
"use of deprecated parameter 'threads'", category=DeprecationWarning
|
|
||||||
)
|
|
||||||
|
|
||||||
compression_args = []
|
compression_args = []
|
||||||
if compression == 'jpeg':
|
if compression == 'jpeg':
|
||||||
compression_args = [
|
compression_args = [
|
||||||
@@ -17,14 +17,12 @@
|
|||||||
|
|
||||||
"""Interface to jbig2 executable"""
|
"""Interface to jbig2 executable"""
|
||||||
|
|
||||||
from functools import lru_cache
|
|
||||||
from subprocess import PIPE
|
from subprocess import PIPE
|
||||||
|
|
||||||
from ..exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
from . import get_version, run
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=1)
|
|
||||||
def version():
|
def version():
|
||||||
return get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
return get_version('jbig2', regex=r'jbig2enc (\d+(\.\d+)*).*')
|
||||||
|
|
||||||
@@ -17,17 +17,14 @@
|
|||||||
|
|
||||||
"""Interface to pngquant executable"""
|
"""Interface to pngquant executable"""
|
||||||
|
|
||||||
from functools import lru_cache
|
|
||||||
from subprocess import run
|
|
||||||
from tempfile import NamedTemporaryFile
|
from tempfile import NamedTemporaryFile
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ..exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
from . import get_version
|
from ocrmypdf.subprocess import get_version, run
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=1)
|
|
||||||
def version():
|
def version():
|
||||||
return get_version('pngquant', regex=r'(\d+(\.\d+)*).*')
|
return get_version('pngquant', regex=r'(\d+(\.\d+)*).*')
|
||||||
|
|
||||||
@@ -21,17 +21,21 @@ import logging
|
|||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
from collections import namedtuple
|
from collections import namedtuple
|
||||||
from contextlib import suppress
|
|
||||||
from os import fspath
|
from os import fspath
|
||||||
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
from subprocess import PIPE, STDOUT, CalledProcessError, TimeoutExpired
|
||||||
|
from typing import List
|
||||||
|
|
||||||
from ..exceptions import (
|
from PIL import Image
|
||||||
|
|
||||||
|
from ocrmypdf.exceptions import (
|
||||||
MissingDependencyError,
|
MissingDependencyError,
|
||||||
SubprocessOutputError,
|
SubprocessOutputError,
|
||||||
TesseractConfigError,
|
TesseractConfigError,
|
||||||
)
|
)
|
||||||
from ..helpers import page_number, safe_symlink
|
from ocrmypdf.subprocess import get_version, run
|
||||||
from . import get_version, run
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence'))
|
OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence'))
|
||||||
|
|
||||||
@@ -59,16 +63,11 @@ class TesseractLoggerAdapter(logging.LoggerAdapter):
|
|||||||
return '[tesseract] %s' % (msg), kwargs
|
return '[tesseract] %s' % (msg), kwargs
|
||||||
|
|
||||||
|
|
||||||
def version(tesseract_env=None):
|
def version():
|
||||||
return get_version('tesseract', regex=r'tesseract\s(.+)', env=tesseract_env)
|
return get_version('tesseract', regex=r'tesseract\s(.+)')
|
||||||
|
|
||||||
|
|
||||||
def v4(tesseract_env=None):
|
def has_textonly_pdf(langs=None):
|
||||||
"Is this Tesseract v4.0?"
|
|
||||||
return version(tesseract_env) >= '4'
|
|
||||||
|
|
||||||
|
|
||||||
def has_textonly_pdf(tesseract_env=None, langs=None):
|
|
||||||
"""Does Tesseract have textonly_pdf capability?
|
"""Does Tesseract have textonly_pdf capability?
|
||||||
|
|
||||||
Available in v4.00.00alpha since January 2017. Best to
|
Available in v4.00.00alpha since January 2017. Best to
|
||||||
@@ -77,28 +76,28 @@ def has_textonly_pdf(tesseract_env=None, langs=None):
|
|||||||
args_tess = tess_base_args(langs, engine_mode=None) + ['--print-parameters', 'pdf']
|
args_tess = tess_base_args(langs, engine_mode=None) + ['--print-parameters', 'pdf']
|
||||||
params = ''
|
params = ''
|
||||||
try:
|
try:
|
||||||
# print-parameters can return non-UTF8 if the parameters are so initialized
|
proc = run(args_tess, check=True, stdout=PIPE, stderr=STDOUT)
|
||||||
proc = run(args_tess, check=True, stdout=PIPE, stderr=STDOUT, env=tesseract_env)
|
|
||||||
params = proc.stdout
|
params = proc.stdout
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
"Could not --print-parameters from tesseract"
|
"Could not --print-parameters from tesseract. This can happen if the "
|
||||||
|
"TESSDATA_PREFIX environment is not set to a valid tessdata folder. "
|
||||||
) from e
|
) from e
|
||||||
if b'textonly_pdf' in params:
|
if b'textonly_pdf' in params:
|
||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
def has_user_words(tesseract_env=None):
|
def has_user_words():
|
||||||
"""Does Tesseract have --user-words capability?
|
"""Does Tesseract have --user-words capability?
|
||||||
|
|
||||||
Not available in 4.0, but available in 4.1. Also available in 3.x, but
|
Not available in 4.0, but available in 4.1. Also available in 3.x, but
|
||||||
we no longer support 3.x.
|
we no longer support 3.x.
|
||||||
"""
|
"""
|
||||||
return version(tesseract_env) >= '4.1'
|
return version() >= '4.1'
|
||||||
|
|
||||||
|
|
||||||
def languages(tesseract_env=None):
|
def get_languages():
|
||||||
def lang_error(output):
|
def lang_error(output):
|
||||||
msg = (
|
msg = (
|
||||||
"Tesseract failed to report available languages.\n"
|
"Tesseract failed to report available languages.\n"
|
||||||
@@ -111,12 +110,7 @@ def languages(tesseract_env=None):
|
|||||||
args_tess = ['tesseract', '--list-langs']
|
args_tess = ['tesseract', '--list-langs']
|
||||||
try:
|
try:
|
||||||
proc = run(
|
proc = run(
|
||||||
args_tess,
|
args_tess, universal_newlines=True, stdout=PIPE, stderr=STDOUT, check=True
|
||||||
universal_newlines=True,
|
|
||||||
stdout=PIPE,
|
|
||||||
stderr=STDOUT,
|
|
||||||
check=True,
|
|
||||||
env=tesseract_env,
|
|
||||||
)
|
)
|
||||||
output = proc.stdout
|
output = proc.stdout
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
@@ -125,11 +119,11 @@ def languages(tesseract_env=None):
|
|||||||
for line in output.splitlines():
|
for line in output.splitlines():
|
||||||
if line.startswith('Error'):
|
if line.startswith('Error'):
|
||||||
raise MissingDependencyError(lang_error(output))
|
raise MissingDependencyError(lang_error(output))
|
||||||
header, *rest = output.splitlines()
|
_header, *rest = output.splitlines()
|
||||||
return set(lang.strip() for lang in rest)
|
return set(lang.strip() for lang in rest)
|
||||||
|
|
||||||
|
|
||||||
def tess_base_args(langs, engine_mode):
|
def tess_base_args(langs: List[str], engine_mode) -> List[str]:
|
||||||
args = ['tesseract']
|
args = ['tesseract']
|
||||||
if langs:
|
if langs:
|
||||||
args.extend(['-l', '+'.join(langs)])
|
args.extend(['-l', '+'.join(langs)])
|
||||||
@@ -138,7 +132,7 @@ def tess_base_args(langs, engine_mode):
|
|||||||
return args
|
return args
|
||||||
|
|
||||||
|
|
||||||
def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env=None):
|
def get_orientation(input_file: Path, engine_mode, timeout: float):
|
||||||
args_tesseract = tess_base_args(['osd'], engine_mode) + [
|
args_tesseract = tess_base_args(['osd'], engine_mode) + [
|
||||||
'--psm',
|
'--psm',
|
||||||
'0',
|
'0',
|
||||||
@@ -147,19 +141,13 @@ def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env=
|
|||||||
]
|
]
|
||||||
|
|
||||||
try:
|
try:
|
||||||
p = run(
|
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||||
args_tesseract,
|
|
||||||
stdout=PIPE,
|
|
||||||
stderr=STDOUT,
|
|
||||||
timeout=timeout,
|
|
||||||
check=True,
|
|
||||||
env=tesseract_env,
|
|
||||||
)
|
|
||||||
stdout = p.stdout
|
stdout = p.stdout
|
||||||
except TimeoutExpired:
|
except TimeoutExpired:
|
||||||
return OrientationConfidence(angle=0, confidence=0.0)
|
return OrientationConfidence(angle=0, confidence=0.0)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
tesseract_log_output(log, e.output, input_file)
|
tesseract_log_output(e.stdout)
|
||||||
|
tesseract_log_output(e.stderr)
|
||||||
if (
|
if (
|
||||||
b'Too few characters. Skipping this page' in e.output
|
b'Too few characters. Skipping this page' in e.output
|
||||||
or b'Image too large' in e.output
|
or b'Image too large' in e.output
|
||||||
@@ -181,15 +169,17 @@ def get_orientation(input_file, engine_mode, timeout: float, log, tesseract_env=
|
|||||||
return oc
|
return oc
|
||||||
|
|
||||||
|
|
||||||
def tesseract_log_output(mainlog, stdout, input_file):
|
def tesseract_log_output(stream):
|
||||||
log = TesseractLoggerAdapter(
|
tlog = TesseractLoggerAdapter(
|
||||||
mainlog, extra=mainlog.extra if hasattr(mainlog, 'extra') else None
|
log, extra=log.extra if hasattr(log, 'extra') else None
|
||||||
)
|
)
|
||||||
|
|
||||||
|
if not stream:
|
||||||
|
return
|
||||||
try:
|
try:
|
||||||
text = stdout.decode()
|
text = stream.decode()
|
||||||
except UnicodeDecodeError:
|
except UnicodeDecodeError:
|
||||||
text = stdout.decode('utf-8', 'ignore')
|
text = stream.decode('utf-8', 'ignore')
|
||||||
|
|
||||||
lines = text.splitlines()
|
lines = text.splitlines()
|
||||||
for line in lines:
|
for line in lines:
|
||||||
@@ -198,67 +188,58 @@ def tesseract_log_output(mainlog, stdout, input_file):
|
|||||||
elif line.startswith("Warning in pixReadMem"):
|
elif line.startswith("Warning in pixReadMem"):
|
||||||
continue
|
continue
|
||||||
elif 'diacritics' in line:
|
elif 'diacritics' in line:
|
||||||
log.warning("lots of diacritics - possibly poor OCR")
|
tlog.warning("lots of diacritics - possibly poor OCR")
|
||||||
elif line.startswith('OSD: Weak margin'):
|
elif line.startswith('OSD: Weak margin'):
|
||||||
log.warning("unsure about page orientation")
|
tlog.warning("unsure about page orientation")
|
||||||
elif 'Error in pixScanForForeground' in line:
|
elif 'Error in pixScanForForeground' in line:
|
||||||
pass # Appears to be spurious/problem with nonwhite borders
|
pass # Appears to be spurious/problem with nonwhite borders
|
||||||
elif 'Error in boxClipToRectangle' in line:
|
elif 'Error in boxClipToRectangle' in line:
|
||||||
pass # Always appears with pixScanForForeground message
|
pass # Always appears with pixScanForForeground message
|
||||||
elif 'parameter not found: ' in line.lower():
|
elif 'parameter not found: ' in line.lower():
|
||||||
log.error(line.strip())
|
tlog.error(line.strip())
|
||||||
problem = line.split('found: ')[1]
|
problem = line.split('found: ')[1]
|
||||||
raise TesseractConfigError(problem)
|
raise TesseractConfigError(problem)
|
||||||
elif 'error' in line.lower() or 'exception' in line.lower():
|
elif 'error' in line.lower() or 'exception' in line.lower():
|
||||||
log.error(line.strip())
|
tlog.error(line.strip())
|
||||||
elif 'warning' in line.lower():
|
elif 'warning' in line.lower():
|
||||||
log.warning(line.strip())
|
tlog.warning(line.strip())
|
||||||
elif 'read_params_file' in line.lower():
|
elif 'read_params_file' in line.lower():
|
||||||
log.error(line.strip())
|
tlog.error(line.strip())
|
||||||
else:
|
else:
|
||||||
log.info(line.strip())
|
tlog.info(line.strip())
|
||||||
|
|
||||||
|
|
||||||
def page_timedout(log, input_file, timeout):
|
def page_timedout(timeout):
|
||||||
if timeout == 0:
|
if timeout == 0:
|
||||||
return
|
return
|
||||||
prefix = f"{(page_number(input_file)):4d}: [tesseract] "
|
log.warning("[tesseract] took too long to OCR - skipping")
|
||||||
log.warning(prefix + " took too long to OCR - skipping")
|
|
||||||
|
|
||||||
|
|
||||||
def _generate_null_hocr(output_hocr, output_sidecar, image):
|
def _generate_null_hocr(output_hocr, output_text, image):
|
||||||
"""Produce a .hocr file that reports no text detected on a page that is
|
"""Produce a .hocr file that reports no text detected on a page that is
|
||||||
the same size as the input image."""
|
the same size as the input image."""
|
||||||
from PIL import Image
|
|
||||||
|
|
||||||
with Image.open(image) as im:
|
with Image.open(image) as im:
|
||||||
w, h = im.size
|
w, h = im.size
|
||||||
|
|
||||||
with open(output_hocr, 'w', encoding="utf-8") as f:
|
output_hocr.write_text(HOCR_TEMPLATE.format(w, h), encoding='utf-8')
|
||||||
f.write(HOCR_TEMPLATE.format(w, h))
|
output_text.write_text('[skipped page]', encoding='utf-8')
|
||||||
with open(output_sidecar, 'w', encoding='utf-8') as f:
|
|
||||||
f.write('[skipped page]')
|
|
||||||
|
|
||||||
|
|
||||||
def generate_hocr(
|
def generate_hocr(
|
||||||
input_file,
|
input_file: Path,
|
||||||
output_files,
|
output_hocr: Path,
|
||||||
language: list,
|
output_text: Path,
|
||||||
|
languages: list,
|
||||||
engine_mode,
|
engine_mode,
|
||||||
tessconfig: list,
|
tessconfig: list,
|
||||||
timeout: float,
|
timeout: float,
|
||||||
pagesegmode: int,
|
pagesegmode: int,
|
||||||
user_words,
|
user_words,
|
||||||
user_patterns,
|
user_patterns,
|
||||||
tesseract_env,
|
|
||||||
log,
|
|
||||||
):
|
):
|
||||||
|
prefix = output_hocr.with_suffix('')
|
||||||
|
|
||||||
output_hocr = next(o for o in output_files if fspath(o).endswith('.hocr'))
|
args_tesseract = tess_base_args(languages, engine_mode)
|
||||||
output_sidecar = next(o for o in output_files if fspath(o).endswith('.txt'))
|
|
||||||
prefix = os.path.splitext(output_hocr)[0]
|
|
||||||
|
|
||||||
args_tesseract = tess_base_args(language, engine_mode)
|
|
||||||
|
|
||||||
if pagesegmode is not None:
|
if pagesegmode is not None:
|
||||||
args_tesseract.extend(['--psm', str(pagesegmode)])
|
args_tesseract.extend(['--psm', str(pagesegmode)])
|
||||||
@@ -269,94 +250,70 @@ def generate_hocr(
|
|||||||
if user_patterns:
|
if user_patterns:
|
||||||
args_tesseract.extend(['--user-patterns', user_patterns])
|
args_tesseract.extend(['--user-patterns', user_patterns])
|
||||||
|
|
||||||
# Reminder: test suite tesseract spoofers will break after any changes
|
# Reminder: test suite tesseract test plugins will break after any changes
|
||||||
# to the number of order parameters here
|
# to the number of order parameters here
|
||||||
args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig)
|
args_tesseract.extend([input_file, prefix, 'hocr', 'txt'] + tessconfig)
|
||||||
try:
|
try:
|
||||||
p = run(
|
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||||
args_tesseract,
|
|
||||||
stdout=PIPE,
|
|
||||||
stderr=STDOUT,
|
|
||||||
timeout=timeout,
|
|
||||||
check=True,
|
|
||||||
env=tesseract_env,
|
|
||||||
)
|
|
||||||
stdout = p.stdout
|
stdout = p.stdout
|
||||||
except TimeoutExpired:
|
except TimeoutExpired:
|
||||||
# Generate a HOCR file with no recognized text if tesseract times out
|
# Generate a HOCR file with no recognized text if tesseract times out
|
||||||
# Temporary workaround to hocrTransform not being able to function if
|
# Temporary workaround to hocrTransform not being able to function if
|
||||||
# it does not have a valid hOCR file.
|
# it does not have a valid hOCR file.
|
||||||
page_timedout(log, input_file, timeout)
|
page_timedout(timeout)
|
||||||
_generate_null_hocr(output_hocr, output_sidecar, input_file)
|
_generate_null_hocr(output_hocr, output_text, input_file)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
tesseract_log_output(log, e.output, input_file)
|
tesseract_log_output(e.output)
|
||||||
if b'Image too large' in e.output:
|
if b'Image too large' in e.output:
|
||||||
_generate_null_hocr(output_hocr, output_sidecar, input_file)
|
_generate_null_hocr(output_hocr, output_text, input_file)
|
||||||
return
|
return
|
||||||
|
|
||||||
raise SubprocessOutputError() from e
|
raise SubprocessOutputError() from e
|
||||||
else:
|
else:
|
||||||
tesseract_log_output(log, stdout, input_file)
|
tesseract_log_output(stdout)
|
||||||
# The sidecar text file will get the suffix .txt; rename it to
|
# The sidecar text file will get the suffix .txt; rename it to
|
||||||
# whatever caller wants it named
|
# whatever caller wants it named
|
||||||
if os.path.exists(prefix + '.txt'):
|
if prefix.with_suffix('.txt').exists():
|
||||||
shutil.move(prefix + '.txt', output_sidecar)
|
shutil.move(prefix.with_suffix('.txt'), output_text)
|
||||||
|
|
||||||
|
|
||||||
def use_skip_page(text_only, skip_pdf, output_pdf, output_text):
|
def use_skip_page(output_pdf, output_text):
|
||||||
with open(output_text, 'w') as f:
|
output_text.write_text('[skipped page]', encoding='utf-8')
|
||||||
f.write('[skipped page]')
|
|
||||||
|
|
||||||
if skip_pdf and not text_only:
|
# A 0 byte file to the output to indicate a skip
|
||||||
# Substitute a "skipped page"
|
output_pdf.write_bytes(b'')
|
||||||
with suppress(FileNotFoundError):
|
|
||||||
os.remove(output_pdf) # In case it was partially created
|
|
||||||
safe_symlink(skip_pdf, output_pdf)
|
|
||||||
return
|
|
||||||
|
|
||||||
# Or normally, just write a 0 byte file to the output to indicate a skip
|
|
||||||
with open(output_pdf, 'wb') as out:
|
|
||||||
out.write(b'')
|
|
||||||
|
|
||||||
|
|
||||||
def generate_pdf(
|
def generate_pdf(
|
||||||
*,
|
*,
|
||||||
input_image,
|
input_file: Path,
|
||||||
skip_pdf=None,
|
output_pdf: Path,
|
||||||
output_pdf,
|
output_text: Path,
|
||||||
output_text,
|
languages: List[str],
|
||||||
language: list,
|
|
||||||
engine_mode,
|
engine_mode,
|
||||||
text_only: bool,
|
tessconfig: List[str],
|
||||||
tessconfig: list,
|
|
||||||
timeout: float,
|
timeout: float,
|
||||||
pagesegmode: int,
|
pagesegmode: int,
|
||||||
user_words,
|
user_words,
|
||||||
user_patterns,
|
user_patterns,
|
||||||
tesseract_env,
|
|
||||||
log,
|
|
||||||
):
|
):
|
||||||
"""Use Tesseract to render a PDF.
|
"""Use Tesseract to render a PDF.
|
||||||
|
|
||||||
input_image -- image to analyze
|
input_file -- image to analyze
|
||||||
skip_pdf -- if we time out, use this file as output
|
|
||||||
output_pdf -- file to generate
|
output_pdf -- file to generate
|
||||||
output_text -- OCR text file
|
output_text -- OCR text file
|
||||||
language -- list of languages to consider
|
languages -- list of languages to consider
|
||||||
engine_mode -- engine mode argument for tess v4
|
engine_mode -- engine mode argument for tess v4
|
||||||
text_only -- enable tesseract text only mode?
|
|
||||||
tessconfig -- tesseract configuration
|
tessconfig -- tesseract configuration
|
||||||
timeout -- timeout (seconds)
|
timeout -- timeout (seconds)
|
||||||
log -- logger object
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
args_tesseract = tess_base_args(language, engine_mode)
|
args_tesseract = tess_base_args(languages, engine_mode)
|
||||||
|
|
||||||
if pagesegmode is not None:
|
if pagesegmode is not None:
|
||||||
args_tesseract.extend(['--psm', str(pagesegmode)])
|
args_tesseract.extend(['--psm', str(pagesegmode)])
|
||||||
|
|
||||||
if text_only and has_textonly_pdf(tesseract_env, language):
|
args_tesseract.extend(['-c', 'textonly_pdf=1'])
|
||||||
args_tesseract.extend(['-c', 'textonly_pdf=1'])
|
|
||||||
|
|
||||||
if user_words:
|
if user_words:
|
||||||
args_tesseract.extend(['--user-words', user_words])
|
args_tesseract.extend(['--user-words', user_words])
|
||||||
@@ -366,30 +323,23 @@ def generate_pdf(
|
|||||||
|
|
||||||
prefix = os.path.splitext(output_pdf)[0] # Tesseract appends suffixes
|
prefix = os.path.splitext(output_pdf)[0] # Tesseract appends suffixes
|
||||||
|
|
||||||
# Reminder: test suite tesseract spoofers might break after any changes
|
# Reminder: test suite tesseract test plugins might break after any changes
|
||||||
# to the number of order parameters here
|
# to the number of order parameters here
|
||||||
|
|
||||||
args_tesseract.extend([input_image, prefix, 'pdf', 'txt'] + tessconfig)
|
args_tesseract.extend([input_file, prefix, 'pdf', 'txt'] + tessconfig)
|
||||||
try:
|
try:
|
||||||
p = run(
|
p = run(args_tesseract, stdout=PIPE, stderr=STDOUT, timeout=timeout, check=True)
|
||||||
args_tesseract,
|
|
||||||
stdout=PIPE,
|
|
||||||
stderr=STDOUT,
|
|
||||||
timeout=timeout,
|
|
||||||
check=True,
|
|
||||||
env=tesseract_env,
|
|
||||||
)
|
|
||||||
stdout = p.stdout
|
stdout = p.stdout
|
||||||
if os.path.exists(prefix + '.txt'):
|
if os.path.exists(prefix + '.txt'):
|
||||||
shutil.move(prefix + '.txt', output_text)
|
shutil.move(prefix + '.txt', output_text)
|
||||||
except TimeoutExpired:
|
except TimeoutExpired:
|
||||||
page_timedout(log, input_image, timeout)
|
page_timedout(timeout)
|
||||||
use_skip_page(text_only, skip_pdf, output_pdf, output_text)
|
use_skip_page(output_pdf, output_text)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
tesseract_log_output(log, e.output, input_image)
|
tesseract_log_output(e.output)
|
||||||
if b'Image too large' in e.output:
|
if b'Image too large' in e.output:
|
||||||
use_skip_page(text_only, skip_pdf, output_pdf, output_text)
|
use_skip_page(output_pdf, output_text)
|
||||||
return
|
return
|
||||||
raise SubprocessOutputError() from e
|
raise SubprocessOutputError() from e
|
||||||
else:
|
else:
|
||||||
tesseract_log_output(log, stdout, input_image)
|
tesseract_log_output(stdout)
|
||||||
@@ -20,31 +20,32 @@
|
|||||||
|
|
||||||
"""Interface to unpaper executable"""
|
"""Interface to unpaper executable"""
|
||||||
|
|
||||||
|
import logging
|
||||||
import os
|
import os
|
||||||
import shlex
|
import shlex
|
||||||
from functools import lru_cache
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
from subprocess import PIPE, STDOUT, CalledProcessError
|
||||||
from tempfile import TemporaryDirectory
|
from tempfile import TemporaryDirectory
|
||||||
|
from typing import Tuple
|
||||||
|
|
||||||
from PIL import Image
|
from PIL import Image
|
||||||
|
|
||||||
from ..exceptions import MissingDependencyError, SubprocessOutputError
|
from ocrmypdf.exceptions import MissingDependencyError, SubprocessOutputError
|
||||||
from . import get_version
|
from ocrmypdf.subprocess import get_version
|
||||||
from . import run as external_run
|
from ocrmypdf.subprocess import run as external_run
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
@lru_cache(maxsize=1)
|
|
||||||
def version():
|
def version():
|
||||||
return get_version('unpaper')
|
return get_version('unpaper')
|
||||||
|
|
||||||
|
|
||||||
def run(input_file, output_file, dpi, log, mode_args):
|
def _setup_unpaper_io(tmpdir: Path, input_file: Path) -> Tuple[Path, Path]:
|
||||||
args_unpaper = ['unpaper', '-v', '--dpi', str(dpi)] + mode_args
|
|
||||||
|
|
||||||
SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'}
|
SUFFIXES = {'1': '.pbm', 'L': '.pgm', 'RGB': '.ppm'}
|
||||||
|
with Image.open(input_file) as im:
|
||||||
with TemporaryDirectory() as tmpdir, Image.open(input_file) as im:
|
im_modified = False
|
||||||
if im.mode not in SUFFIXES.keys():
|
if im.mode not in SUFFIXES:
|
||||||
log.info("Converting image to other colorspace")
|
log.info("Converting image to other colorspace")
|
||||||
try:
|
try:
|
||||||
if im.mode == 'P' and len(im.getcolors()) == 2:
|
if im.mode == 'P' and len(im.getcolors()) == 2:
|
||||||
@@ -52,11 +53,11 @@ def run(input_file, output_file, dpi, log, mode_args):
|
|||||||
else:
|
else:
|
||||||
im = im.convert(mode='RGB')
|
im = im.convert(mode='RGB')
|
||||||
except IOError as e:
|
except IOError as e:
|
||||||
im.close()
|
|
||||||
raise MissingDependencyError(
|
raise MissingDependencyError(
|
||||||
"Could not convert image with type " + im.mode
|
"Could not convert image with type " + im.mode
|
||||||
) from e
|
) from e
|
||||||
|
else:
|
||||||
|
im_modified = True
|
||||||
try:
|
try:
|
||||||
suffix = SUFFIXES[im.mode]
|
suffix = SUFFIXES[im.mode]
|
||||||
except KeyError:
|
except KeyError:
|
||||||
@@ -64,9 +65,21 @@ def run(input_file, output_file, dpi, log, mode_args):
|
|||||||
"Failed to convert image to a supported format."
|
"Failed to convert image to a supported format."
|
||||||
) from e
|
) from e
|
||||||
|
|
||||||
input_pnm = os.path.join(tmpdir, f'input{suffix}')
|
if im_modified or input_file.suffix != '.png':
|
||||||
output_pnm = os.path.join(tmpdir, f'output{suffix}')
|
input_png = tmpdir / 'input.png'
|
||||||
im.save(input_pnm, format='PPM')
|
im.save(input_png, format='PNG', compress_level=1)
|
||||||
|
else:
|
||||||
|
# No changes, PNG input, just use the file we already have
|
||||||
|
input_png = input_file
|
||||||
|
output_pnm = tmpdir / f'output{suffix}'
|
||||||
|
return input_png, output_pnm
|
||||||
|
|
||||||
|
|
||||||
|
def run(input_file, output_file, dpi, mode_args):
|
||||||
|
args_unpaper = ['unpaper', '-v', '--dpi', str(dpi)] + mode_args
|
||||||
|
|
||||||
|
with TemporaryDirectory() as tmpdir:
|
||||||
|
input_png, output_pnm = _setup_unpaper_io(Path(tmpdir), input_file)
|
||||||
|
|
||||||
# To prevent any shenanigans from accepting arbitrary parameters in
|
# To prevent any shenanigans from accepting arbitrary parameters in
|
||||||
# --unpaper-args, we:
|
# --unpaper-args, we:
|
||||||
@@ -75,23 +88,22 @@ def run(input_file, output_file, dpi, log, mode_args):
|
|||||||
# 3) append absolute paths for the input and output file
|
# 3) append absolute paths for the input and output file
|
||||||
# This should ensure that a user cannot clobber some other file with
|
# This should ensure that a user cannot clobber some other file with
|
||||||
# their unpaper arguments (whether intentionally or otherwise)
|
# their unpaper arguments (whether intentionally or otherwise)
|
||||||
args_unpaper.extend([input_pnm, output_pnm])
|
args_unpaper.extend([os.fspath(input_png), os.fspath(output_pnm)])
|
||||||
try:
|
try:
|
||||||
proc = external_run(
|
proc = external_run(
|
||||||
args_unpaper,
|
args_unpaper,
|
||||||
check=True,
|
check=True,
|
||||||
close_fds=True,
|
close_fds=True,
|
||||||
universal_newlines=True,
|
universal_newlines=True,
|
||||||
stderr=STDOUT,
|
stderr=STDOUT, # unpaper writes logging output to stdout and stderr
|
||||||
cwd=tmpdir,
|
cwd=tmpdir, # and cannot send file output to stdout
|
||||||
stdout=PIPE,
|
stdout=PIPE,
|
||||||
)
|
)
|
||||||
except CalledProcessError as e:
|
except CalledProcessError as e:
|
||||||
log.debug(e.output)
|
log.debug(e.stderr)
|
||||||
raise e from e
|
raise e from e
|
||||||
else:
|
else:
|
||||||
log.debug(proc.stdout)
|
log.debug(proc.stderr)
|
||||||
# unpaper sets dpi to 72; fix this
|
|
||||||
try:
|
try:
|
||||||
with Image.open(output_pnm) as imout:
|
with Image.open(output_pnm) as imout:
|
||||||
imout.save(output_file, dpi=(dpi, dpi))
|
imout.save(output_file, dpi=(dpi, dpi))
|
||||||
@@ -110,7 +122,7 @@ def validate_custom_args(args: str):
|
|||||||
return unpaper_args
|
return unpaper_args
|
||||||
|
|
||||||
|
|
||||||
def clean(input_file, output_file, dpi, log, unpaper_args=None):
|
def clean(input_file, output_file, dpi, unpaper_args=None):
|
||||||
default_args = [
|
default_args = [
|
||||||
'--layout',
|
'--layout',
|
||||||
'none',
|
'none',
|
||||||
@@ -124,4 +136,4 @@ def clean(input_file, output_file, dpi, log, unpaper_args=None):
|
|||||||
]
|
]
|
||||||
if not unpaper_args:
|
if not unpaper_args:
|
||||||
unpaper_args = default_args
|
unpaper_args = default_args
|
||||||
run(input_file, output_file, dpi, log, unpaper_args)
|
run(input_file, output_file, dpi, unpaper_args)
|
||||||
+126
-109
@@ -15,12 +15,14 @@
|
|||||||
# You should have received a copy of the GNU General Public License
|
# You should have received a copy of the GNU General Public License
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
import os
|
import logging
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from typing import Optional
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
MAX_REPLACE_PAGES = 100
|
MAX_REPLACE_PAGES = 100
|
||||||
|
|
||||||
|
|
||||||
@@ -88,99 +90,10 @@ def strip_invisible_text(pdf, page):
|
|||||||
page.Contents = pikepdf.Stream(pdf, content_stream)
|
page.Contents = pikepdf.Stream(pdf, content_stream)
|
||||||
|
|
||||||
|
|
||||||
def _graft_text_layer(
|
|
||||||
*, pdf_base, page_num, text, font, font_key, procset, rotation, strip_old_text, log
|
|
||||||
):
|
|
||||||
"""Insert the text layer from text page 0 on to pdf_base at page_num"""
|
|
||||||
|
|
||||||
log.debug("Grafting")
|
|
||||||
if Path(text).stat().st_size == 0:
|
|
||||||
return
|
|
||||||
|
|
||||||
# This is a pointer indicating a specific page in the base file
|
|
||||||
pdf_text = pikepdf.open(text)
|
|
||||||
pdf_text_contents = pdf_text.pages[0].Contents.read_bytes()
|
|
||||||
|
|
||||||
base_page = pdf_base.pages.p(page_num)
|
|
||||||
|
|
||||||
# The text page always will be oriented up by this stage but the original
|
|
||||||
# content may have a rotation applied. Wrap the text stream with a rotation
|
|
||||||
# so it will be oriented the same way as the rest of the page content.
|
|
||||||
# (Previous versions OCRmyPDF rotated the content layer to match the text.)
|
|
||||||
mediabox = [float(pdf_text.pages[0].MediaBox[v]) for v in range(4)]
|
|
||||||
wt, ht = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
|
||||||
|
|
||||||
mediabox = [float(base_page.MediaBox[v]) for v in range(4)]
|
|
||||||
wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
|
||||||
|
|
||||||
translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2)
|
|
||||||
untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2)
|
|
||||||
corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1])
|
|
||||||
# -rotation because the input is a clockwise angle and this formula
|
|
||||||
# uses CCW
|
|
||||||
rotation = -rotation % 360
|
|
||||||
rotate = pikepdf.PdfMatrix().rotated(rotation)
|
|
||||||
|
|
||||||
# Because of rounding of DPI, we might get a text layer that is not
|
|
||||||
# identically sized to the target page. Scale to adjust. Normally this
|
|
||||||
# is within 0.998.
|
|
||||||
if rotation in (90, 270):
|
|
||||||
wt, ht = ht, wt
|
|
||||||
scale_x = wp / wt
|
|
||||||
scale_y = hp / ht
|
|
||||||
|
|
||||||
# log.debug('%r', scale_x, scale_y)
|
|
||||||
scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y)
|
|
||||||
|
|
||||||
# Translate the text so it is centered at (0, 0), rotate it there, adjust
|
|
||||||
# for a size different between initial and text PDF, then untranslate, and
|
|
||||||
# finally move the lower left corner to match the mediabox
|
|
||||||
ctm = translate @ rotate @ scale @ untranslate @ corner
|
|
||||||
|
|
||||||
pdf_text_contents = b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n'
|
|
||||||
|
|
||||||
new_text_layer = pikepdf.Stream(pdf_base, pdf_text_contents)
|
|
||||||
|
|
||||||
if strip_old_text:
|
|
||||||
strip_invisible_text(pdf_base, base_page)
|
|
||||||
|
|
||||||
base_page.page_contents_add(new_text_layer, prepend=True)
|
|
||||||
|
|
||||||
_update_page_resources(
|
|
||||||
page=base_page, font=font, font_key=font_key, procset=procset
|
|
||||||
)
|
|
||||||
pdf_text.close()
|
|
||||||
|
|
||||||
|
|
||||||
def _find_font(text, pdf_base):
|
|
||||||
"""Copy a font from the filename text into pdf_base"""
|
|
||||||
|
|
||||||
font, font_key = None, None
|
|
||||||
possible_font_names = ('/f-0-0', '/F1')
|
|
||||||
try:
|
|
||||||
with pikepdf.open(text) as pdf_text:
|
|
||||||
try:
|
|
||||||
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
|
||||||
except (AttributeError, IndexError, KeyError):
|
|
||||||
return None, None
|
|
||||||
for f in possible_font_names:
|
|
||||||
pdf_text_font = pdf_text_fonts.get(f, None)
|
|
||||||
if pdf_text_font is not None:
|
|
||||||
font_key = f
|
|
||||||
break
|
|
||||||
if pdf_text_font:
|
|
||||||
font = pdf_base.copy_foreign(pdf_text_font)
|
|
||||||
return font, font_key
|
|
||||||
except (FileNotFoundError, pikepdf.PdfError):
|
|
||||||
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
|
||||||
return None, None
|
|
||||||
|
|
||||||
|
|
||||||
class OcrGrafter:
|
class OcrGrafter:
|
||||||
def __init__(self, context):
|
def __init__(self, context):
|
||||||
self.context = context
|
self.context = context
|
||||||
self.log = context.log
|
self.path_base = context.origin
|
||||||
self.path_base = Path(context.origin).resolve()
|
|
||||||
|
|
||||||
self.pdf_base = pikepdf.open(self.path_base)
|
self.pdf_base = pikepdf.open(self.path_base)
|
||||||
self.font, self.font_key = None, None
|
self.font, self.font_key = None, None
|
||||||
@@ -195,10 +108,16 @@ class OcrGrafter:
|
|||||||
self.emplacements = 1
|
self.emplacements = 1
|
||||||
self.interim_count = 0
|
self.interim_count = 0
|
||||||
|
|
||||||
def graft_page(self, page_result):
|
def graft_page(
|
||||||
pageno, image, text, _sidecar, autorotate_correction = page_result
|
self,
|
||||||
if text and not self.font:
|
*,
|
||||||
self.font, self.font_key = _find_font(text, self.pdf_base)
|
pageno: int,
|
||||||
|
image: Optional[Path],
|
||||||
|
textpdf: Optional[Path],
|
||||||
|
autorotate_correction: int,
|
||||||
|
):
|
||||||
|
if textpdf and not self.font:
|
||||||
|
self.font, self.font_key = self._find_font(textpdf)
|
||||||
|
|
||||||
emplaced_page = False
|
emplaced_page = False
|
||||||
content_rotation = self.pdfinfo[pageno].rotation
|
content_rotation = self.pdfinfo[pageno].rotation
|
||||||
@@ -206,7 +125,7 @@ class OcrGrafter:
|
|||||||
if path_image is not None and path_image != self.path_base:
|
if path_image is not None and path_image != self.path_base:
|
||||||
# We are updating the old page with a rasterized PDF of the new
|
# We are updating the old page with a rasterized PDF of the new
|
||||||
# page (without changing objgen, to preserve references)
|
# page (without changing objgen, to preserve references)
|
||||||
self.log.debug("Emplacement update")
|
log.debug("Emplacement update")
|
||||||
with pikepdf.open(image) as pdf_image:
|
with pikepdf.open(image) as pdf_image:
|
||||||
self.emplacements += 1
|
self.emplacements += 1
|
||||||
foreign_image_page = pdf_image.pages[0]
|
foreign_image_page = pdf_image.pages[0]
|
||||||
@@ -220,25 +139,23 @@ class OcrGrafter:
|
|||||||
content_rotation = autorotate_correction
|
content_rotation = autorotate_correction
|
||||||
text_rotation = autorotate_correction
|
text_rotation = autorotate_correction
|
||||||
text_misaligned = (text_rotation - content_rotation) % 360
|
text_misaligned = (text_rotation - content_rotation) % 360
|
||||||
self.log.debug(
|
log.debug(
|
||||||
f"Rotations for page {pageno}: [text, auto, misalign, content] = "
|
f"Rotations for page {pageno}: [text, auto, misalign, content] = "
|
||||||
f"{text_rotation}, {autorotate_correction}, "
|
f"{text_rotation}, {autorotate_correction}, "
|
||||||
f"{text_misaligned}, {content_rotation}"
|
f"{text_misaligned}, {content_rotation}"
|
||||||
)
|
)
|
||||||
|
|
||||||
if text and self.font:
|
if textpdf and self.font:
|
||||||
# Graft the text layer onto this page, whether new or old
|
# Graft the text layer onto this page, whether new or old
|
||||||
strip_old = self.context.options.redo_ocr
|
strip_old = self.context.options.redo_ocr
|
||||||
_graft_text_layer(
|
self._graft_text_layer(
|
||||||
pdf_base=self.pdf_base,
|
|
||||||
page_num=pageno + 1,
|
page_num=pageno + 1,
|
||||||
text=text,
|
textpdf=textpdf,
|
||||||
font=self.font,
|
font=self.font,
|
||||||
font_key=self.font_key,
|
font_key=self.font_key,
|
||||||
rotation=text_misaligned,
|
rotation=text_misaligned,
|
||||||
procset=self.procset,
|
procset=self.procset,
|
||||||
strip_old_text=strip_old,
|
strip_old_text=strip_old,
|
||||||
log=self.log,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Correct the rotation if applicable
|
# Correct the rotation if applicable
|
||||||
@@ -250,10 +167,13 @@ class OcrGrafter:
|
|||||||
self.save_and_reload()
|
self.save_and_reload()
|
||||||
|
|
||||||
def save_and_reload(self):
|
def save_and_reload(self):
|
||||||
# Periodically save and reload the Pdf object. This will keep a
|
"""Save and reload the Pdf.
|
||||||
# lid on our memory usage for very large files. Attach the font to
|
|
||||||
# page 1 even if page 1 doesn't use it, so we have a way to get it
|
This will keep a lid on our memory usage for very large files. Attach
|
||||||
# back.
|
the font to page 1 even if page 1 doesn't use it, so we have a way to get it
|
||||||
|
back.
|
||||||
|
"""
|
||||||
|
|
||||||
page0 = self.pdf_base.pages[0]
|
page0 = self.pdf_base.pages[0]
|
||||||
_update_page_resources(
|
_update_page_resources(
|
||||||
page=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
page=page0, font=self.font, font_key=self.font_key, procset=self.procset
|
||||||
@@ -264,12 +184,14 @@ class OcrGrafter:
|
|||||||
# {interim_count} is the opened file we were updateing
|
# {interim_count} is the opened file we were updateing
|
||||||
# {interim_count - 1} can be deleted
|
# {interim_count - 1} can be deleted
|
||||||
# {interim_count + 1} is the new file will produce and open
|
# {interim_count + 1} is the new file will produce and open
|
||||||
old_file = self.output_file + f'_working{self.interim_count - 1}.pdf'
|
old_file = self.output_file.with_suffix(f'.working{self.interim_count - 1}.pdf')
|
||||||
if not self.context.options.keep_temporary_files:
|
if not self.context.options.keep_temporary_files:
|
||||||
with suppress(FileNotFoundError):
|
with suppress(FileNotFoundError):
|
||||||
os.unlink(old_file)
|
old_file.unlink()
|
||||||
|
|
||||||
next_file = self.output_file + f'_working{self.interim_count + 1}.pdf'
|
next_file = self.output_file.with_suffix(
|
||||||
|
f'.working{self.interim_count + 1}.pdf'
|
||||||
|
)
|
||||||
self.pdf_base.save(next_file)
|
self.pdf_base.save(next_file)
|
||||||
self.pdf_base.close()
|
self.pdf_base.close()
|
||||||
|
|
||||||
@@ -282,3 +204,98 @@ class OcrGrafter:
|
|||||||
self.pdf_base.save(self.output_file)
|
self.pdf_base.save(self.output_file)
|
||||||
self.pdf_base.close()
|
self.pdf_base.close()
|
||||||
return self.output_file
|
return self.output_file
|
||||||
|
|
||||||
|
def _find_font(self, text):
|
||||||
|
"""Copy a font from the filename text into pdf_base"""
|
||||||
|
|
||||||
|
font, font_key = None, None
|
||||||
|
possible_font_names = ('/f-0-0', '/F1')
|
||||||
|
try:
|
||||||
|
with pikepdf.open(text) as pdf_text:
|
||||||
|
try:
|
||||||
|
pdf_text_fonts = pdf_text.pages[0].Resources.get('/Font', {})
|
||||||
|
except (AttributeError, IndexError, KeyError):
|
||||||
|
return None, None
|
||||||
|
for f in possible_font_names:
|
||||||
|
pdf_text_font = pdf_text_fonts.get(f, None)
|
||||||
|
if pdf_text_font is not None:
|
||||||
|
font_key = f
|
||||||
|
break
|
||||||
|
if pdf_text_font:
|
||||||
|
font = self.pdf_base.copy_foreign(pdf_text_font)
|
||||||
|
return font, font_key
|
||||||
|
except (FileNotFoundError, pikepdf.PdfError):
|
||||||
|
# PdfError occurs if a 0-length file is written e.g. due to OCR timeout
|
||||||
|
return None, None
|
||||||
|
|
||||||
|
def _graft_text_layer(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
page_num: int,
|
||||||
|
textpdf: Path,
|
||||||
|
font: pikepdf.Object,
|
||||||
|
font_key: pikepdf.Object,
|
||||||
|
procset: pikepdf.Object,
|
||||||
|
rotation: int,
|
||||||
|
strip_old_text: bool,
|
||||||
|
):
|
||||||
|
"""Insert the text layer from text page 0 on to pdf_base at page_num"""
|
||||||
|
|
||||||
|
log.debug("Grafting")
|
||||||
|
if Path(textpdf).stat().st_size == 0:
|
||||||
|
return
|
||||||
|
|
||||||
|
# This is a pointer indicating a specific page in the base file
|
||||||
|
with pikepdf.open(textpdf) as pdf_text:
|
||||||
|
pdf_text_contents = pdf_text.pages[0].Contents.read_bytes()
|
||||||
|
|
||||||
|
base_page = self.pdf_base.pages.p(page_num)
|
||||||
|
|
||||||
|
# The text page always will be oriented up by this stage but the original
|
||||||
|
# content may have a rotation applied. Wrap the text stream with a rotation
|
||||||
|
# so it will be oriented the same way as the rest of the page content.
|
||||||
|
# (Previous versions OCRmyPDF rotated the content layer to match the text.)
|
||||||
|
mediabox = [float(pdf_text.pages[0].MediaBox[v]) for v in range(4)]
|
||||||
|
wt, ht = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
||||||
|
|
||||||
|
mediabox = [float(base_page.MediaBox[v]) for v in range(4)]
|
||||||
|
wp, hp = mediabox[2] - mediabox[0], mediabox[3] - mediabox[1]
|
||||||
|
|
||||||
|
translate = pikepdf.PdfMatrix().translated(-wt / 2, -ht / 2)
|
||||||
|
untranslate = pikepdf.PdfMatrix().translated(wp / 2, hp / 2)
|
||||||
|
corner = pikepdf.PdfMatrix().translated(mediabox[0], mediabox[1])
|
||||||
|
# -rotation because the input is a clockwise angle and this formula
|
||||||
|
# uses CCW
|
||||||
|
rotation = -rotation % 360
|
||||||
|
rotate = pikepdf.PdfMatrix().rotated(rotation)
|
||||||
|
|
||||||
|
# Because of rounding of DPI, we might get a text layer that is not
|
||||||
|
# identically sized to the target page. Scale to adjust. Normally this
|
||||||
|
# is within 0.998.
|
||||||
|
if rotation in (90, 270):
|
||||||
|
wt, ht = ht, wt
|
||||||
|
scale_x = wp / wt
|
||||||
|
scale_y = hp / ht
|
||||||
|
|
||||||
|
# log.debug('%r', scale_x, scale_y)
|
||||||
|
scale = pikepdf.PdfMatrix().scaled(scale_x, scale_y)
|
||||||
|
|
||||||
|
# Translate the text so it is centered at (0, 0), rotate it there, adjust
|
||||||
|
# for a size different between initial and text PDF, then untranslate, and
|
||||||
|
# finally move the lower left corner to match the mediabox
|
||||||
|
ctm = translate @ rotate @ scale @ untranslate @ corner
|
||||||
|
|
||||||
|
pdf_text_contents = (
|
||||||
|
b'q %s cm\n' % ctm.encode() + pdf_text_contents + b'\nQ\n'
|
||||||
|
)
|
||||||
|
|
||||||
|
new_text_layer = pikepdf.Stream(self.pdf_base, pdf_text_contents)
|
||||||
|
|
||||||
|
if strip_old_text:
|
||||||
|
strip_invisible_text(self.pdf_base, base_page)
|
||||||
|
|
||||||
|
base_page.page_contents_add(new_text_layer, prepend=True)
|
||||||
|
|
||||||
|
_update_page_resources(
|
||||||
|
page=base_page, font=font, font_key=font_key, procset=procset
|
||||||
|
)
|
||||||
|
|||||||
+37
-72
@@ -15,110 +15,75 @@
|
|||||||
# You should have received a copy of the GNU General Public License
|
# You should have received a copy of the GNU General Public License
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
import logging
|
|
||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
|
from argparse import Namespace
|
||||||
|
from copy import copy
|
||||||
|
from io import IOBase
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Iterator
|
||||||
|
|
||||||
|
from ocrmypdf.pdfinfo import PdfInfo
|
||||||
|
|
||||||
|
|
||||||
class PicklableLoggerMixin:
|
class PdfContext:
|
||||||
def __init__(self):
|
|
||||||
self._log = None
|
|
||||||
|
|
||||||
@property
|
|
||||||
def log(self):
|
|
||||||
if not self._log:
|
|
||||||
self._log = self.get_logger()
|
|
||||||
return self._log
|
|
||||||
|
|
||||||
def __getstate__(self):
|
|
||||||
# Python 3.6 is incapable of pickling a logger and marshalling it to another
|
|
||||||
# process (threading._RLock error), so we disconnect it before pickling,
|
|
||||||
# and create a new logger in the worker process.
|
|
||||||
state = self.__dict__.copy()
|
|
||||||
state['_log'] = None
|
|
||||||
return state
|
|
||||||
|
|
||||||
|
|
||||||
class PDFContext(PicklableLoggerMixin):
|
|
||||||
"""Holds our context for a particular run of the pipeline"""
|
"""Holds our context for a particular run of the pipeline"""
|
||||||
|
|
||||||
def __init__(self, options, work_folder, origin, pdfinfo):
|
def __init__(
|
||||||
PicklableLoggerMixin.__init__(self)
|
self,
|
||||||
|
options: Namespace,
|
||||||
|
work_folder: Path,
|
||||||
|
origin: Path,
|
||||||
|
pdfinfo: PdfInfo,
|
||||||
|
plugin_manager,
|
||||||
|
):
|
||||||
self.options = options
|
self.options = options
|
||||||
self.work_folder = work_folder
|
self.work_folder = work_folder
|
||||||
self.origin = origin
|
self.origin = origin
|
||||||
self.pdfinfo = pdfinfo
|
self.pdfinfo = pdfinfo
|
||||||
if options:
|
self.plugin_manager = plugin_manager
|
||||||
self.name = os.path.basename(options.input_file)
|
|
||||||
else:
|
|
||||||
self.name = 'origin.pdf'
|
|
||||||
if self.name == '-':
|
|
||||||
self.name = 'stdin'
|
|
||||||
|
|
||||||
def get_logger(self):
|
def get_path(self, name: str) -> Path:
|
||||||
return make_logger(self.options, filename=self.name)
|
return self.work_folder / name
|
||||||
|
|
||||||
def get_path(self, name):
|
def get_page_contexts(self) -> Iterator['PageContext']:
|
||||||
return os.path.join(self.work_folder, name)
|
|
||||||
|
|
||||||
def get_page_contexts(self):
|
|
||||||
npages = len(self.pdfinfo)
|
npages = len(self.pdfinfo)
|
||||||
for n in range(npages):
|
for n in range(npages):
|
||||||
yield PageContext(self, n)
|
yield PageContext(self, n)
|
||||||
|
|
||||||
|
|
||||||
class PageContext(PicklableLoggerMixin):
|
class PageContext:
|
||||||
"""Holds our context for a page
|
"""Holds our context for a page
|
||||||
|
|
||||||
Must be pickable, so only store intrinsic/simple data elements
|
Must be pickable, so stores only intrinsic/simple data elements or those
|
||||||
|
capable of their serializing themselves via __getstate__.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
def __init__(self, pdf_context, pageno):
|
def __init__(self, pdf_context: PdfContext, pageno):
|
||||||
PicklableLoggerMixin.__init__(self)
|
|
||||||
self.work_folder = pdf_context.work_folder
|
self.work_folder = pdf_context.work_folder
|
||||||
self.origin = pdf_context.origin
|
self.origin = pdf_context.origin
|
||||||
self.options = pdf_context.options
|
self.options = pdf_context.options
|
||||||
self.name = pdf_context.name
|
|
||||||
self.pageno = pageno
|
self.pageno = pageno
|
||||||
self.pageinfo = pdf_context.pdfinfo[pageno]
|
self.pageinfo = pdf_context.pdfinfo[pageno]
|
||||||
self._log = None
|
self.plugin_manager = pdf_context.plugin_manager
|
||||||
|
|
||||||
def get_logger(self):
|
def get_path(self, name: str) -> Path:
|
||||||
return make_logger(self.options, filename=self.name, page=self.pageno + 1)
|
return self.work_folder / ("%06d_%s" % (self.pageno + 1, name))
|
||||||
|
|
||||||
def get_path(self, name):
|
def __getstate__(self):
|
||||||
return os.path.join(self.work_folder, "%06d_%s" % (self.pageno + 1, name))
|
state = self.__dict__.copy()
|
||||||
|
|
||||||
|
state['options'] = copy(self.options)
|
||||||
|
if not isinstance(state['options'].input_file, (str, bytes, os.PathLike)):
|
||||||
|
state['options'].input_file = 'stream'
|
||||||
|
if not isinstance(state['options'].output_file, (str, bytes, os.PathLike)):
|
||||||
|
state['options'].output_file = 'stream'
|
||||||
|
return state
|
||||||
|
|
||||||
|
|
||||||
def cleanup_working_files(work_folder, options):
|
def cleanup_working_files(work_folder: Path, options: Namespace):
|
||||||
if options.keep_temporary_files:
|
if options.keep_temporary_files:
|
||||||
print(f"Temporary working files retained at:\n{work_folder}", file=sys.stderr)
|
print(f"Temporary working files retained at:\n{work_folder}", file=sys.stderr)
|
||||||
else:
|
else:
|
||||||
shutil.rmtree(work_folder, ignore_errors=True)
|
shutil.rmtree(work_folder, ignore_errors=True)
|
||||||
|
|
||||||
|
|
||||||
class LogNameAdapter(logging.LoggerAdapter):
|
|
||||||
def process(self, msg, kwargs):
|
|
||||||
# return '[%s] %s' % (self.extra['input_filename'], msg), kwargs
|
|
||||||
return '%s' % (msg,), kwargs
|
|
||||||
|
|
||||||
|
|
||||||
class LogNamePageAdapter(logging.LoggerAdapter):
|
|
||||||
def process(self, msg, kwargs):
|
|
||||||
return (
|
|
||||||
#'[%s:%05u] %s' % (self.extra['input_filename'], self.extra['page'], msg),
|
|
||||||
'%4u: %s' % (self.extra['page'], msg),
|
|
||||||
kwargs,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def make_logger(options=None, prefix='ocrmypdf', filename=None, page=None):
|
|
||||||
log = logging.getLogger(prefix)
|
|
||||||
if filename and page:
|
|
||||||
adapter = LogNamePageAdapter(log, dict(input_filename=filename, page=page))
|
|
||||||
elif filename:
|
|
||||||
adapter = LogNameAdapter(log, dict(input_filename=filename))
|
|
||||||
else:
|
|
||||||
adapter = log
|
|
||||||
return adapter
|
|
||||||
|
|||||||
@@ -0,0 +1,60 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This file is part of OCRmyPDF.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import sys
|
||||||
|
from contextlib import suppress
|
||||||
|
|
||||||
|
from tqdm import tqdm
|
||||||
|
|
||||||
|
|
||||||
|
class PageNumberFilter(logging.Filter):
|
||||||
|
def filter(self, record):
|
||||||
|
pageno = getattr(record, 'pageno', None)
|
||||||
|
if isinstance(pageno, int):
|
||||||
|
record.pageno = f'{pageno:5d} '
|
||||||
|
elif pageno is None:
|
||||||
|
record.pageno = ''
|
||||||
|
return True
|
||||||
|
|
||||||
|
|
||||||
|
class TqdmConsole:
|
||||||
|
"""Wrapper to log messages in a way that is compatible with tqdm progress bar
|
||||||
|
|
||||||
|
This routes log messages through tqdm so that it can print them above the
|
||||||
|
progress bar, and then refresh the progress bar, rather than overwriting
|
||||||
|
it which looks messy.
|
||||||
|
|
||||||
|
For some reason Python 3.6 prints extra empty messages from time to time,
|
||||||
|
so we suppress those.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, file):
|
||||||
|
self.file = file
|
||||||
|
self.py36 = sys.version_info[0:2] == (3, 6)
|
||||||
|
|
||||||
|
def write(self, msg):
|
||||||
|
# When no progress bar is active, tqdm.write() routes to print()
|
||||||
|
if self.py36:
|
||||||
|
if msg.strip() != '':
|
||||||
|
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
||||||
|
else:
|
||||||
|
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
||||||
|
|
||||||
|
def flush(self):
|
||||||
|
with suppress(AttributeError):
|
||||||
|
self.file.flush()
|
||||||
+188
-184
@@ -15,53 +15,59 @@
|
|||||||
# You should have received a copy of the GNU General Public License
|
# You should have received a copy of the GNU General Public License
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
import logging
|
||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
|
from contextlib import suppress
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj
|
from shutil import copyfileobj
|
||||||
|
from typing import BinaryIO, Dict, Iterable, Optional, Union, cast
|
||||||
|
|
||||||
import img2pdf
|
import img2pdf
|
||||||
import pikepdf
|
import pikepdf
|
||||||
from pikepdf.models.metadata import encode_pdf_date
|
from pikepdf.models.metadata import encode_pdf_date
|
||||||
from PIL import Image
|
from PIL import Image, ImageColor, ImageDraw
|
||||||
|
|
||||||
from . import leptonica
|
from ocrmypdf import leptonica
|
||||||
from ._version import PROGRAM_NAME
|
from ocrmypdf._exec import unpaper
|
||||||
from ._version import __version__ as VERSION
|
from ocrmypdf._jobcontext import PageContext, PdfContext
|
||||||
from .exceptions import (
|
from ocrmypdf._version import PROGRAM_NAME
|
||||||
|
from ocrmypdf._version import __version__ as VERSION
|
||||||
|
from ocrmypdf.exceptions import (
|
||||||
DpiError,
|
DpiError,
|
||||||
EncryptedPdfError,
|
EncryptedPdfError,
|
||||||
InputFileError,
|
InputFileError,
|
||||||
PriorOcrFoundError,
|
PriorOcrFoundError,
|
||||||
UnsupportedImageFormatError,
|
UnsupportedImageFormatError,
|
||||||
)
|
)
|
||||||
from .exec import ghostscript, tesseract
|
from ocrmypdf.helpers import Resolution, safe_symlink
|
||||||
from .helpers import safe_symlink
|
from ocrmypdf.hocrtransform import HocrTransform
|
||||||
from .hocrtransform import HocrTransform
|
from ocrmypdf.optimize import optimize
|
||||||
from .optimize import optimize
|
from ocrmypdf.pdfa import generate_pdfa_ps
|
||||||
from .pdfa import generate_pdfa_ps
|
from ocrmypdf.pdfinfo import Colorspace, Encoding, PdfInfo
|
||||||
from .pdfinfo import Colorspace, Encoding, PdfInfo
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
VECTOR_PAGE_DPI = 400
|
VECTOR_PAGE_DPI = 400
|
||||||
|
|
||||||
|
|
||||||
def triage_image_file(input_file, output_file, options, log):
|
def triage_image_file(input_file, output_file, options):
|
||||||
log.info("Input file is not a PDF, checking if it is an image...")
|
log.info("Input file is not a PDF, checking if it is an image...")
|
||||||
try:
|
try:
|
||||||
im = Image.open(input_file)
|
im = Image.open(input_file)
|
||||||
except EnvironmentError as e:
|
except EnvironmentError as e:
|
||||||
# Recover the original filename
|
# Recover the original filename
|
||||||
log.error(str(e).replace(input_file, options.input_file))
|
log.error(str(e).replace(str(input_file), str(options.input_file)))
|
||||||
raise UnsupportedImageFormatError() from e
|
raise UnsupportedImageFormatError() from e
|
||||||
|
|
||||||
with im:
|
with im:
|
||||||
log.info("Input file is an image")
|
log.info("Input file is an image")
|
||||||
if 'dpi' in im.info:
|
if 'dpi' in im.info:
|
||||||
if im.info['dpi'] <= (96, 96) and not options.image_dpi:
|
if im.info['dpi'] <= (96, 96) and not options.image_dpi:
|
||||||
log.info("Image size: (%d, %d)" % im.size)
|
log.info("Image size: (%d, %d)", *im.size)
|
||||||
log.info("Image resolution: (%d, %d)" % im.info['dpi'])
|
log.info("Image resolution: (%d, %d)", *im.info['dpi'])
|
||||||
log.error(
|
log.error(
|
||||||
"Input file is an image, but the resolution (DPI) is "
|
"Input file is an image, but the resolution (DPI) is "
|
||||||
"not credible. Estimate the resolution at which the "
|
"not credible. Estimate the resolution at which the "
|
||||||
@@ -69,7 +75,7 @@ def triage_image_file(input_file, output_file, options, log):
|
|||||||
)
|
)
|
||||||
raise DpiError()
|
raise DpiError()
|
||||||
elif not options.image_dpi:
|
elif not options.image_dpi:
|
||||||
log.info("Image size: (%d, %d)" % im.size)
|
log.info("Image size: (%d, %d)", *im.size)
|
||||||
log.error(
|
log.error(
|
||||||
"Input file is an image, but has no resolution (DPI) "
|
"Input file is an image, but has no resolution (DPI) "
|
||||||
"in its metadata. Estimate the resolution at which "
|
"in its metadata. Estimate the resolution at which "
|
||||||
@@ -96,11 +102,14 @@ def triage_image_file(input_file, output_file, options, log):
|
|||||||
layout_fun = img2pdf.default_layout_fun
|
layout_fun = img2pdf.default_layout_fun
|
||||||
if options.image_dpi:
|
if options.image_dpi:
|
||||||
layout_fun = img2pdf.get_fixed_dpi_layout_fun(
|
layout_fun = img2pdf.get_fixed_dpi_layout_fun(
|
||||||
(options.image_dpi, options.image_dpi)
|
Resolution(options.image_dpi, options.image_dpi)
|
||||||
)
|
)
|
||||||
with open(output_file, 'wb') as outf:
|
with open(output_file, 'wb') as outf:
|
||||||
img2pdf.convert(
|
img2pdf.convert(
|
||||||
input_file, layout_fun=layout_fun, with_pdfrw=False, outputstream=outf
|
os.fspath(input_file),
|
||||||
|
layout_fun=layout_fun,
|
||||||
|
with_pdfrw=False,
|
||||||
|
outputstream=outf,
|
||||||
)
|
)
|
||||||
log.info("Successfully converted to PDF, processing...")
|
log.info("Successfully converted to PDF, processing...")
|
||||||
except img2pdf.ImageOpenError as e:
|
except img2pdf.ImageOpenError as e:
|
||||||
@@ -124,7 +133,7 @@ def _pdf_guess_version(input_file, search_window=1024):
|
|||||||
return ''
|
return ''
|
||||||
|
|
||||||
|
|
||||||
def triage(original_filename, input_file, output_file, options, log):
|
def triage(original_filename, input_file, output_file, options):
|
||||||
try:
|
try:
|
||||||
if _pdf_guess_version(input_file):
|
if _pdf_guess_version(input_file):
|
||||||
if options.image_dpi:
|
if options.image_dpi:
|
||||||
@@ -137,17 +146,27 @@ def triage(original_filename, input_file, output_file, options, log):
|
|||||||
return output_file
|
return output_file
|
||||||
except EnvironmentError as e:
|
except EnvironmentError as e:
|
||||||
log.debug(f"Temporary file was at: {input_file}")
|
log.debug(f"Temporary file was at: {input_file}")
|
||||||
msg = str(e).replace(input_file, original_filename)
|
msg = str(e).replace(str(input_file), original_filename)
|
||||||
raise InputFileError(msg) from e
|
raise InputFileError(msg) from e
|
||||||
|
|
||||||
triage_image_file(input_file, output_file, options, log)
|
triage_image_file(input_file, output_file, options)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def get_pdfinfo(input_file, detailed_page_analysis=False, progbar=False):
|
def get_pdfinfo(
|
||||||
|
input_file,
|
||||||
|
detailed_analysis=False,
|
||||||
|
progbar=False,
|
||||||
|
max_workers=None,
|
||||||
|
check_pages=None,
|
||||||
|
):
|
||||||
try:
|
try:
|
||||||
return PdfInfo(
|
return PdfInfo(
|
||||||
input_file, detailed_page_analysis=detailed_page_analysis, progbar=progbar
|
input_file,
|
||||||
|
detailed_analysis=detailed_analysis,
|
||||||
|
progbar=progbar,
|
||||||
|
max_workers=max_workers,
|
||||||
|
check_pages=check_pages,
|
||||||
)
|
)
|
||||||
except pikepdf.PasswordError:
|
except pikepdf.PasswordError:
|
||||||
raise EncryptedPdfError()
|
raise EncryptedPdfError()
|
||||||
@@ -155,8 +174,7 @@ def get_pdfinfo(input_file, detailed_page_analysis=False, progbar=False):
|
|||||||
raise InputFileError()
|
raise InputFileError()
|
||||||
|
|
||||||
|
|
||||||
def validate_pdfinfo_options(context):
|
def validate_pdfinfo_options(context: PdfContext):
|
||||||
log = context.log
|
|
||||||
pdfinfo = context.pdfinfo
|
pdfinfo = context.pdfinfo
|
||||||
options = context.options
|
options = context.options
|
||||||
|
|
||||||
@@ -194,54 +212,56 @@ def validate_pdfinfo_options(context):
|
|||||||
"form and all filled form fields. The output PDF will be "
|
"form and all filled form fields. The output PDF will be "
|
||||||
"'flattened' and will no longer be fillable."
|
"'flattened' and will no longer be fillable."
|
||||||
)
|
)
|
||||||
|
context.plugin_manager.hook.validate(pdfinfo=pdfinfo, options=options)
|
||||||
|
|
||||||
|
|
||||||
def get_page_dpi(pageinfo, options):
|
def get_page_dpi(pageinfo, options):
|
||||||
"Get the DPI when nonsquare DPI is tolerable"
|
"Get the DPI when nonsquare DPI is tolerable"
|
||||||
xres = max(
|
xres = max(
|
||||||
pageinfo.xres or VECTOR_PAGE_DPI,
|
pageinfo.dpi.x or VECTOR_PAGE_DPI,
|
||||||
options.oversample or 0,
|
options.oversample or 0.0,
|
||||||
VECTOR_PAGE_DPI if pageinfo.has_vector else 0,
|
VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0,
|
||||||
)
|
)
|
||||||
yres = max(
|
yres = max(
|
||||||
pageinfo.yres or VECTOR_PAGE_DPI,
|
pageinfo.dpi.y or VECTOR_PAGE_DPI,
|
||||||
options.oversample or 0,
|
options.oversample or 0,
|
||||||
VECTOR_PAGE_DPI if pageinfo.has_vector else 0,
|
VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0,
|
||||||
)
|
)
|
||||||
return (float(xres), float(yres))
|
return Resolution(float(xres), float(yres))
|
||||||
|
|
||||||
|
|
||||||
def get_page_square_dpi(pageinfo, options):
|
def get_page_square_dpi(pageinfo, options) -> Resolution:
|
||||||
"Get the DPI when we require xres == yres, scaled to physical units"
|
"Get the DPI when we require xres == yres, scaled to physical units"
|
||||||
xres = pageinfo.xres or 0
|
xres = pageinfo.dpi.x or 0.0
|
||||||
yres = pageinfo.yres or 0
|
yres = pageinfo.dpi.y or 0.0
|
||||||
userunit = pageinfo.userunit or 1
|
userunit = float(pageinfo.userunit) or 1.0
|
||||||
return float(
|
units = float(
|
||||||
max(
|
max(
|
||||||
(xres * userunit) or VECTOR_PAGE_DPI,
|
(xres * userunit) or VECTOR_PAGE_DPI,
|
||||||
(yres * userunit) or VECTOR_PAGE_DPI,
|
(yres * userunit) or VECTOR_PAGE_DPI,
|
||||||
VECTOR_PAGE_DPI if pageinfo.has_vector else 0,
|
VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0,
|
||||||
options.oversample or 0,
|
options.oversample or 0.0,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
return Resolution(units, units)
|
||||||
|
|
||||||
|
|
||||||
def get_canvas_square_dpi(pageinfo, options):
|
def get_canvas_square_dpi(pageinfo, options) -> Resolution:
|
||||||
"""Get the DPI when we require xres == yres, in Postscript units"""
|
"""Get the DPI when we require xres == yres, in Postscript units"""
|
||||||
return float(
|
units = float(
|
||||||
max(
|
max(
|
||||||
(pageinfo.xres) or VECTOR_PAGE_DPI,
|
(pageinfo.dpi.x) or VECTOR_PAGE_DPI,
|
||||||
(pageinfo.yres) or VECTOR_PAGE_DPI,
|
(pageinfo.dpi.y) or VECTOR_PAGE_DPI,
|
||||||
VECTOR_PAGE_DPI if pageinfo.has_vector else 0,
|
VECTOR_PAGE_DPI if pageinfo.has_vector else 0.0,
|
||||||
options.oversample or 0,
|
options.oversample or 0.0,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
return Resolution(units, units)
|
||||||
|
|
||||||
|
|
||||||
def is_ocr_required(page_context):
|
def is_ocr_required(page_context: PageContext):
|
||||||
pageinfo = page_context.pageinfo
|
pageinfo = page_context.pageinfo
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
log = page_context.log
|
|
||||||
|
|
||||||
ocr_required = True
|
ocr_required = True
|
||||||
|
|
||||||
@@ -251,14 +271,15 @@ def is_ocr_required(page_context):
|
|||||||
elif pageinfo.has_text:
|
elif pageinfo.has_text:
|
||||||
if not options.force_ocr and not (options.skip_text or options.redo_ocr):
|
if not options.force_ocr and not (options.skip_text or options.redo_ocr):
|
||||||
raise PriorOcrFoundError(
|
raise PriorOcrFoundError(
|
||||||
"page already has text! - aborting (use --force-ocr to force OCR)"
|
"page already has text! - aborting (use --force-ocr to force OCR; "
|
||||||
|
" see also help for the arguments --skip-text and --redo-ocr"
|
||||||
)
|
)
|
||||||
elif options.force_ocr:
|
elif options.force_ocr:
|
||||||
log.info("page already has text! - rasterizing text and running OCR anyway")
|
log.info("page already has text! - rasterizing text and running OCR anyway")
|
||||||
ocr_required = True
|
ocr_required = True
|
||||||
elif options.redo_ocr:
|
elif options.redo_ocr:
|
||||||
if pageinfo.has_corrupt_text:
|
if pageinfo.has_corrupt_text:
|
||||||
log.warn(
|
log.warning(
|
||||||
"some text on this page cannot be mapped to characters: "
|
"some text on this page cannot be mapped to characters: "
|
||||||
"consider using --force-ocr instead"
|
"consider using --force-ocr instead"
|
||||||
)
|
)
|
||||||
@@ -285,7 +306,7 @@ def is_ocr_required(page_context):
|
|||||||
)
|
)
|
||||||
elif options.force_ocr:
|
elif options.force_ocr:
|
||||||
# Warn the user they might not want to do this
|
# Warn the user they might not want to do this
|
||||||
log.warn(
|
log.warning(
|
||||||
"page has no images - "
|
"page has no images - "
|
||||||
"all vector content will be "
|
"all vector content will be "
|
||||||
f"rasterized at {VECTOR_PAGE_DPI} DPI, losing some resolution and likely "
|
f"rasterized at {VECTOR_PAGE_DPI} DPI, losing some resolution and likely "
|
||||||
@@ -305,31 +326,29 @@ def is_ocr_required(page_context):
|
|||||||
pixel_count = pageinfo.width_pixels * pageinfo.height_pixels
|
pixel_count = pageinfo.width_pixels * pageinfo.height_pixels
|
||||||
if pixel_count > (options.skip_big * 1_000_000):
|
if pixel_count > (options.skip_big * 1_000_000):
|
||||||
ocr_required = False
|
ocr_required = False
|
||||||
log.warn(
|
log.warning(
|
||||||
"page too big, skipping OCR "
|
"page too big, skipping OCR "
|
||||||
f"({(pixel_count / 1_000_000):.1f} MPixels > {options.skip_big:.1f} MPixels --skip-big)"
|
f"({(pixel_count / 1_000_000):.1f} MPixels > {options.skip_big:.1f} MPixels --skip-big)"
|
||||||
)
|
)
|
||||||
return ocr_required
|
return ocr_required
|
||||||
|
|
||||||
|
|
||||||
def rasterize_preview(input_file, page_context):
|
def rasterize_preview(input_file: Path, page_context: PageContext):
|
||||||
output_file = page_context.get_path('rasterize_preview.jpg')
|
output_file = page_context.get_path('rasterize_preview.jpg')
|
||||||
canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options)
|
canvas_dpi = get_canvas_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
page_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
ghostscript.rasterize_pdf(
|
page_context.plugin_manager.hook.rasterize_pdf_page(
|
||||||
input_file,
|
input_file=input_file,
|
||||||
output_file,
|
output_file=output_file,
|
||||||
xres=canvas_dpi,
|
|
||||||
yres=canvas_dpi,
|
|
||||||
raster_device='jpeggray',
|
raster_device='jpeggray',
|
||||||
log=page_context.log,
|
raster_dpi=canvas_dpi,
|
||||||
page_dpi=(page_dpi, page_dpi),
|
page_dpi=page_dpi,
|
||||||
pageno=page_context.pageinfo.pageno + 1,
|
pageno=page_context.pageinfo.pageno + 1,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def describe_rotation(page_context, orient_conf, correction):
|
def describe_rotation(page_context: PageContext, orient_conf, correction: int):
|
||||||
"""
|
"""
|
||||||
Describe the page rotation we are going to perform.
|
Describe the page rotation we are going to perform.
|
||||||
"""
|
"""
|
||||||
@@ -358,34 +377,28 @@ def describe_rotation(page_context, orient_conf, correction):
|
|||||||
return f"{facing}, confidence {orient_conf.confidence:.2f} - {action}"
|
return f"{facing}, confidence {orient_conf.confidence:.2f} - {action}"
|
||||||
|
|
||||||
|
|
||||||
def get_orientation_correction(preview, page_context):
|
def get_orientation_correction(preview: Path, page_context: PageContext):
|
||||||
"""
|
"""Work out orientation correct for each page.
|
||||||
Work out orientation correct for each page.
|
|
||||||
|
|
||||||
We ask Ghostscript to draw a preview page, which will rasterize with the
|
We ask Ghostscript to draw a preview page, which will rasterize with the
|
||||||
current /Rotate applied, and then ask Tesseract which way the page is
|
current /Rotate applied, and then ask OCR which way the page is
|
||||||
oriented. If the value of /Rotate is correct (e.g., a user already
|
oriented. If the value of /Rotate is correct (e.g., a user already
|
||||||
manually fixed rotation), then Tesseract will say the page is pointing
|
manually fixed rotation), then OCR will say the page is pointing
|
||||||
up and the correction is zero. Otherwise, the orientation found by
|
up and the correction is zero. Otherwise, the orientation found by
|
||||||
Tesseract represents the clockwise rotation, or the counterclockwise
|
OCR represents the clockwise rotation, or the counterclockwise
|
||||||
correction to rotation.
|
correction to rotation.
|
||||||
|
|
||||||
When we draw the real page for OCR, we rotate it by the CCW correction,
|
When we draw the real page for OCR, we rotate it by the CCW correction,
|
||||||
which points it (hopefully) upright. _graft.py takes care of the orienting
|
which points it (hopefully) upright. _graft.py takes care of the orienting
|
||||||
the image and text layers.
|
the image and text layers.
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
orient_conf = tesseract.get_orientation(
|
orient_conf = page_context.plugin_manager.hook.get_ocr_engine().get_orientation(
|
||||||
preview,
|
preview, page_context.options
|
||||||
engine_mode=page_context.options.tesseract_oem,
|
|
||||||
timeout=page_context.options.tesseract_timeout,
|
|
||||||
log=page_context.log,
|
|
||||||
tesseract_env=page_context.options.tesseract_env,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
correction = orient_conf.angle % 360
|
correction = orient_conf.angle % 360
|
||||||
page_context.log.info(describe_rotation(page_context, orient_conf, correction))
|
log.info(describe_rotation(page_context, orient_conf, correction))
|
||||||
if (
|
if (
|
||||||
orient_conf.confidence >= page_context.options.rotate_pages_threshold
|
orient_conf.confidence >= page_context.options.rotate_pages_threshold
|
||||||
and correction != 0
|
and correction != 0
|
||||||
@@ -396,7 +409,11 @@ def get_orientation_correction(preview, page_context):
|
|||||||
|
|
||||||
|
|
||||||
def rasterize(
|
def rasterize(
|
||||||
input_file, page_context, correction=0, output_tag='', remove_vectors=None
|
input_file: Path,
|
||||||
|
page_context: PageContext,
|
||||||
|
correction: int = 0,
|
||||||
|
output_tag: str = '',
|
||||||
|
remove_vectors=None,
|
||||||
):
|
):
|
||||||
colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m']
|
colorspaces = ['pngmono', 'pnggray', 'png256', 'png16m']
|
||||||
device_idx = 0
|
device_idx = 0
|
||||||
@@ -426,21 +443,19 @@ def rasterize(
|
|||||||
|
|
||||||
device = colorspaces[device_idx]
|
device = colorspaces[device_idx]
|
||||||
|
|
||||||
page_context.log.debug(f"Rasterize with {device}")
|
log.debug(f"Rasterize with {device}")
|
||||||
|
|
||||||
# Produce the page image with square resolution or else deskew and OCR
|
# Produce the page image with square resolution or else deskew and OCR
|
||||||
# will not work properly.
|
# will not work properly.
|
||||||
canvas_dpi = get_canvas_square_dpi(pageinfo, page_context.options)
|
canvas_dpi = get_canvas_square_dpi(pageinfo, page_context.options)
|
||||||
page_dpi = get_page_square_dpi(pageinfo, page_context.options)
|
page_dpi = get_page_square_dpi(pageinfo, page_context.options)
|
||||||
|
|
||||||
ghostscript.rasterize_pdf(
|
page_context.plugin_manager.hook.rasterize_pdf_page(
|
||||||
input_file,
|
input_file=input_file,
|
||||||
output_file,
|
output_file=output_file,
|
||||||
xres=canvas_dpi,
|
|
||||||
yres=canvas_dpi,
|
|
||||||
raster_device=device,
|
raster_device=device,
|
||||||
log=page_context.log,
|
raster_dpi=canvas_dpi,
|
||||||
page_dpi=(page_dpi, page_dpi),
|
page_dpi=page_dpi,
|
||||||
pageno=pageinfo.pageno + 1,
|
pageno=pageinfo.pageno + 1,
|
||||||
rotation=correction,
|
rotation=correction,
|
||||||
filter_vector=remove_vectors,
|
filter_vector=remove_vectors,
|
||||||
@@ -448,39 +463,31 @@ def rasterize(
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def preprocess_remove_background(input_file, page_context):
|
def preprocess_remove_background(input_file: Path, page_context: PageContext):
|
||||||
if any(image.bpc > 1 for image in page_context.pageinfo.images):
|
if any(image.bpc > 1 for image in page_context.pageinfo.images):
|
||||||
output_file = page_context.get_path('pp_rm_bg.png')
|
output_file = page_context.get_path('pp_rm_bg.png')
|
||||||
leptonica.remove_background(input_file, output_file)
|
leptonica.remove_background(input_file, output_file)
|
||||||
return output_file
|
return output_file
|
||||||
else:
|
else:
|
||||||
page_context.log.info("background removal skipped on mono page")
|
log.info("background removal skipped on mono page")
|
||||||
return input_file
|
return input_file
|
||||||
|
|
||||||
|
|
||||||
def preprocess_deskew(input_file, page_context):
|
def preprocess_deskew(input_file: Path, page_context: PageContext):
|
||||||
output_file = page_context.get_path('pp_deskew.png')
|
output_file = page_context.get_path('pp_deskew.png')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
leptonica.deskew(input_file, output_file, dpi)
|
leptonica.deskew(input_file, output_file, dpi.x)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def preprocess_clean(input_file, page_context):
|
def preprocess_clean(input_file: Path, page_context: PageContext):
|
||||||
from .exec import unpaper
|
|
||||||
|
|
||||||
output_file = page_context.get_path('pp_clean.png')
|
output_file = page_context.get_path('pp_clean.png')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
unpaper.clean(
|
unpaper.clean(input_file, output_file, dpi.x, page_context.options.unpaper_args)
|
||||||
input_file,
|
|
||||||
output_file,
|
|
||||||
dpi,
|
|
||||||
page_context.log,
|
|
||||||
page_context.options.unpaper_args,
|
|
||||||
)
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def create_ocr_image(image, page_context):
|
def create_ocr_image(image: Path, page_context: PageContext):
|
||||||
"""Create the image we send for OCR. May not be the same as the display
|
"""Create the image we send for OCR. May not be the same as the display
|
||||||
image depending on preprocessing. This image will never be shown to the
|
image depending on preprocessing. This image will never be shown to the
|
||||||
user."""
|
user."""
|
||||||
@@ -488,15 +495,11 @@ def create_ocr_image(image, page_context):
|
|||||||
output_file = page_context.get_path('ocr.png')
|
output_file = page_context.get_path('ocr.png')
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
with Image.open(image) as im:
|
with Image.open(image) as im:
|
||||||
from PIL import ImageColor
|
|
||||||
from PIL import ImageDraw
|
|
||||||
|
|
||||||
white = ImageColor.getcolor('#ffffff', im.mode)
|
white = ImageColor.getcolor('#ffffff', im.mode)
|
||||||
# pink = ImageColor.getcolor('#ff0080', im.mode)
|
# pink = ImageColor.getcolor('#ff0080', im.mode)
|
||||||
draw = ImageDraw.ImageDraw(im)
|
draw = ImageDraw.ImageDraw(im)
|
||||||
|
|
||||||
xres, yres = im.info['dpi']
|
log.debug('resolution %r', im.info['dpi'])
|
||||||
page_context.log.debug('resolution %r %r' % (xres, yres))
|
|
||||||
|
|
||||||
if not options.force_ocr:
|
if not options.force_ocr:
|
||||||
# Do not mask text areas when forcing OCR, because we need to OCR
|
# Do not mask text areas when forcing OCR, because we need to OCR
|
||||||
@@ -512,15 +515,15 @@ def create_ocr_image(image, page_context):
|
|||||||
# without regard whatever resolution is in pageinfo (may differ or
|
# without regard whatever resolution is in pageinfo (may differ or
|
||||||
# be None)
|
# be None)
|
||||||
bbox = [float(v) for v in textarea]
|
bbox = [float(v) for v in textarea]
|
||||||
xscale, yscale = float(xres) / 72.0, float(yres) / 72.0
|
xyscale = tuple(float(coord) / 72.0 for coord in im.info['dpi'])
|
||||||
pixcoords = [
|
pixcoords = [
|
||||||
bbox[0] * xscale,
|
bbox[0] * xyscale[0],
|
||||||
im.height - bbox[3] * yscale,
|
im.height - bbox[3] * xyscale[1],
|
||||||
bbox[2] * xscale,
|
bbox[2] * xyscale[0],
|
||||||
im.height - bbox[1] * yscale,
|
im.height - bbox[1] * xyscale[1],
|
||||||
]
|
]
|
||||||
pixcoords = [int(round(c)) for c in pixcoords]
|
pixcoords = [int(round(c)) for c in pixcoords]
|
||||||
page_context.log.debug('blanking %r', pixcoords)
|
log.debug('blanking %r', pixcoords)
|
||||||
draw.rectangle(pixcoords, fill=white)
|
draw.rectangle(pixcoords, fill=white)
|
||||||
# draw.rectangle(pixcoords, outline=pink)
|
# draw.rectangle(pixcoords, outline=pink)
|
||||||
|
|
||||||
@@ -530,28 +533,30 @@ def create_ocr_image(image, page_context):
|
|||||||
im = pix.topil()
|
im = pix.topil()
|
||||||
|
|
||||||
del draw
|
del draw
|
||||||
|
|
||||||
|
filter_im = page_context.plugin_manager.hook.filter_ocr_image(
|
||||||
|
page=page_context, image=im
|
||||||
|
)
|
||||||
|
if filter_im is not None:
|
||||||
|
im = filter_im
|
||||||
|
|
||||||
# Pillow requires integer DPI
|
# Pillow requires integer DPI
|
||||||
dpi = round(xres), round(yres)
|
dpi = tuple(round(coord) for coord in im.info['dpi'])
|
||||||
im.save(output_file, dpi=dpi)
|
im.save(output_file, dpi=dpi)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def ocr_tesseract_hocr(input_file, page_context):
|
def ocr_engine_hocr(input_file: Path, page_context: PageContext):
|
||||||
hocr_out = page_context.get_path('ocr_hocr.hocr')
|
hocr_out = page_context.get_path('ocr_hocr.hocr')
|
||||||
hocr_text_out = page_context.get_path('ocr_hocr.txt')
|
hocr_text_out = page_context.get_path('ocr_hocr.txt')
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
tesseract.generate_hocr(
|
|
||||||
|
ocr_engine = page_context.plugin_manager.hook.get_ocr_engine()
|
||||||
|
ocr_engine.generate_hocr(
|
||||||
input_file=input_file,
|
input_file=input_file,
|
||||||
output_files=[hocr_out, hocr_text_out],
|
output_hocr=hocr_out,
|
||||||
language=options.language,
|
output_text=hocr_text_out,
|
||||||
engine_mode=options.tesseract_oem,
|
options=options,
|
||||||
tessconfig=options.tesseract_config,
|
|
||||||
timeout=options.tesseract_timeout,
|
|
||||||
pagesegmode=options.tesseract_pagesegmode,
|
|
||||||
user_words=options.user_words,
|
|
||||||
user_patterns=options.user_patterns,
|
|
||||||
tesseract_env=options.tesseract_env,
|
|
||||||
log=page_context.log,
|
|
||||||
)
|
)
|
||||||
return (hocr_out, hocr_text_out)
|
return (hocr_out, hocr_text_out)
|
||||||
|
|
||||||
@@ -561,24 +566,26 @@ def should_visible_page_image_use_jpg(pageinfo):
|
|||||||
return pageinfo.images and all(im.enc == Encoding.jpeg for im in pageinfo.images)
|
return pageinfo.images and all(im.enc == Encoding.jpeg for im in pageinfo.images)
|
||||||
|
|
||||||
|
|
||||||
def create_visible_page_jpg(image, page_context):
|
def create_visible_page_jpg(image: Path, page_context: PageContext) -> Path:
|
||||||
output_file = page_context.get_path('visible.jpg')
|
output_file = page_context.get_path('visible.jpg')
|
||||||
with Image.open(image) as im:
|
with Image.open(image) as im:
|
||||||
# At this point the image should be a .png, but deskew, unpaper
|
# At this point the image should be a .png, but deskew, unpaper
|
||||||
# might have removed the DPI information. In this case, fall back to
|
# might have removed the DPI information. In this case, fall back to
|
||||||
# square DPI used to rasterize. When the preview image was
|
# square DPI used to rasterize. When the preview image was
|
||||||
# rasterized, it was also converted to square resolution, which is
|
# rasterized, it was also converted to square resolution, which is
|
||||||
# what we want to give tesseract, so keep it square.
|
# what we want to give to the OCR engine, so keep it square.
|
||||||
fallback_dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
if 'dpi' in im.info:
|
||||||
dpi = im.info.get('dpi', (fallback_dpi, fallback_dpi))
|
dpi = Resolution(*im.info['dpi'])
|
||||||
|
else:
|
||||||
|
# Fallback to page-implied DPI
|
||||||
|
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
|
|
||||||
# Pillow requires integer DPI
|
# Pillow requires integer DPI
|
||||||
dpi = round(dpi[0]), round(dpi[1])
|
im.save(output_file, format='JPEG', dpi=dpi.to_int())
|
||||||
im.save(output_file, format='JPEG', dpi=dpi)
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def create_pdf_page_from_image(image, page_context):
|
def create_pdf_page_from_image(image: Path, page_context: PageContext):
|
||||||
# We rasterize a square DPI version of each page because most image
|
# We rasterize a square DPI version of each page because most image
|
||||||
# processing tools don't support rectangular DPI. Use the square DPI as it
|
# processing tools don't support rectangular DPI. Use the square DPI as it
|
||||||
# accurately describes the image. It would be possible to resample the image
|
# accurately describes the image. It would be possible to resample the image
|
||||||
@@ -587,56 +594,50 @@ def create_pdf_page_from_image(image, page_context):
|
|||||||
# sandwich renderer would be fine.
|
# sandwich renderer would be fine.
|
||||||
output_file = page_context.get_path('visible.pdf')
|
output_file = page_context.get_path('visible.pdf')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
layout_fun = img2pdf.get_fixed_dpi_layout_fun((dpi, dpi))
|
layout_fun = img2pdf.get_fixed_dpi_layout_fun(dpi)
|
||||||
|
|
||||||
# This create a single page PDF
|
# This create a single page PDF
|
||||||
with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf:
|
with open(image, 'rb') as imfile, open(output_file, 'wb') as pdf:
|
||||||
page_context.log.debug('convert')
|
log.debug('convert')
|
||||||
img2pdf.convert(
|
img2pdf.convert(
|
||||||
imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf
|
imfile, with_pdfrw=False, layout_fun=layout_fun, outputstream=pdf
|
||||||
)
|
)
|
||||||
page_context.log.debug('convert done')
|
log.debug('convert done')
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def render_hocr_page(hocr, page_context):
|
def render_hocr_page(hocr: Path, page_context: PageContext):
|
||||||
output_file = page_context.get_path('ocr_hocr.pdf')
|
output_file = page_context.get_path('ocr_hocr.pdf')
|
||||||
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
dpi = get_page_square_dpi(page_context.pageinfo, page_context.options)
|
||||||
hocrtransform = HocrTransform(hocr, dpi)
|
hocrtransform = HocrTransform(hocr, dpi.x) # square
|
||||||
hocrtransform.to_pdf(
|
hocrtransform.to_pdf(
|
||||||
output_file,
|
output_file,
|
||||||
imageFileName=None,
|
image_filename=None,
|
||||||
showBoundingboxes=False,
|
show_bounding_boxes=False,
|
||||||
invisibleText=True,
|
invisible_text=True,
|
||||||
interwordSpaces=True,
|
interword_spaces=True,
|
||||||
)
|
)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def ocr_tesseract_textonly_pdf(input_image, page_context):
|
def ocr_engine_textonly_pdf(input_image: Path, page_context: PageContext):
|
||||||
output_pdf = page_context.get_path('ocr_tess.pdf')
|
output_pdf = page_context.get_path('ocr_tess.pdf')
|
||||||
output_text = page_context.get_path('ocr_tess.txt')
|
output_text = page_context.get_path('ocr_tess.txt')
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
tesseract.generate_pdf(
|
|
||||||
input_image=input_image,
|
ocr_engine = page_context.plugin_manager.hook.get_ocr_engine()
|
||||||
skip_pdf=None,
|
ocr_engine.generate_pdf(
|
||||||
|
input_file=input_image,
|
||||||
output_pdf=output_pdf,
|
output_pdf=output_pdf,
|
||||||
output_text=output_text,
|
output_text=output_text,
|
||||||
language=options.language,
|
options=options,
|
||||||
engine_mode=options.tesseract_oem,
|
|
||||||
text_only=True,
|
|
||||||
tessconfig=options.tesseract_config,
|
|
||||||
timeout=options.tesseract_timeout,
|
|
||||||
pagesegmode=options.tesseract_pagesegmode,
|
|
||||||
user_words=options.user_words,
|
|
||||||
user_patterns=options.user_patterns,
|
|
||||||
tesseract_env=options.tesseract_env,
|
|
||||||
log=page_context.log,
|
|
||||||
)
|
)
|
||||||
return (output_pdf, output_text)
|
return (output_pdf, output_text)
|
||||||
|
|
||||||
|
|
||||||
def get_docinfo(base_pdf, options):
|
def get_docinfo(base_pdf: pikepdf.Pdf, context: PdfContext) -> Dict[str, str]:
|
||||||
|
options = context.options
|
||||||
|
|
||||||
def from_document_info(key):
|
def from_document_info(key):
|
||||||
try:
|
try:
|
||||||
s = base_pdf.docinfo[key]
|
s = base_pdf.docinfo[key]
|
||||||
@@ -648,7 +649,6 @@ def get_docinfo(base_pdf, options):
|
|||||||
k: from_document_info(k)
|
k: from_document_info(k)
|
||||||
for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate')
|
for k in ('/Title', '/Author', '/Keywords', '/Subject', '/CreationDate')
|
||||||
}
|
}
|
||||||
renderer_tag = 'OCR'
|
|
||||||
if options is not None:
|
if options is not None:
|
||||||
if options.title:
|
if options.title:
|
||||||
pdfmark['/Title'] = options.title
|
pdfmark['/Title'] = options.title
|
||||||
@@ -659,12 +659,9 @@ def get_docinfo(base_pdf, options):
|
|||||||
if options.subject:
|
if options.subject:
|
||||||
pdfmark['/Subject'] = options.subject
|
pdfmark['/Subject'] = options.subject
|
||||||
|
|
||||||
if options.pdf_renderer == 'sandwich':
|
creator_tag = context.plugin_manager.hook.get_ocr_engine().creator_tag(options)
|
||||||
renderer_tag = 'OCR-PDF'
|
|
||||||
|
|
||||||
pdfmark['/Creator'] = (
|
pdfmark['/Creator'] = f'{PROGRAM_NAME} {VERSION} / {creator_tag}'
|
||||||
f'{PROGRAM_NAME} {VERSION} / ' f'Tesseract {renderer_tag} {tesseract.version()}'
|
|
||||||
)
|
|
||||||
pdfmark['/Producer'] = f'pikepdf {pikepdf.__version__}'
|
pdfmark['/Producer'] = f'pikepdf {pikepdf.__version__}'
|
||||||
if 'OCRMYPDF_CREATOR' in os.environ:
|
if 'OCRMYPDF_CREATOR' in os.environ:
|
||||||
pdfmark['/Creator'] = os.environ['OCRMYPDF_CREATOR']
|
pdfmark['/Creator'] = os.environ['OCRMYPDF_CREATOR']
|
||||||
@@ -675,13 +672,13 @@ def get_docinfo(base_pdf, options):
|
|||||||
return pdfmark
|
return pdfmark
|
||||||
|
|
||||||
|
|
||||||
def generate_postscript_stub(context):
|
def generate_postscript_stub(context: PdfContext):
|
||||||
output_file = context.get_path('pdfa.ps')
|
output_file = context.get_path('pdfa.ps')
|
||||||
generate_pdfa_ps(output_file)
|
generate_pdfa_ps(output_file)
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def convert_to_pdfa(input_pdf, input_ps_stub, context):
|
def convert_to_pdfa(input_pdf: Path, input_ps_stub: Path, context: PdfContext):
|
||||||
options = context.options
|
options = context.options
|
||||||
input_pdfinfo = context.pdfinfo
|
input_pdfinfo = context.pdfinfo
|
||||||
fix_docinfo_file = context.get_path('fix_docinfo.pdf')
|
fix_docinfo_file = context.get_path('fix_docinfo.pdf')
|
||||||
@@ -697,7 +694,7 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context):
|
|||||||
try:
|
try:
|
||||||
len(pdf_file.docinfo)
|
len(pdf_file.docinfo)
|
||||||
except TypeError:
|
except TypeError:
|
||||||
context.log.error(
|
log.error(
|
||||||
"File contains a malformed DocumentInfo block - continuing anyway"
|
"File contains a malformed DocumentInfo block - continuing anyway"
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
@@ -711,26 +708,26 @@ def convert_to_pdfa(input_pdf, input_ps_stub, context):
|
|||||||
else:
|
else:
|
||||||
safe_symlink(input_pdf, fix_docinfo_file)
|
safe_symlink(input_pdf, fix_docinfo_file)
|
||||||
|
|
||||||
ghostscript.generate_pdfa(
|
context.plugin_manager.hook.generate_pdfa(
|
||||||
pdf_version=input_pdfinfo.min_version,
|
pdf_version=input_pdfinfo.min_version,
|
||||||
pdf_pages=[fix_docinfo_file, input_ps_stub],
|
pdf_pages=[fix_docinfo_file],
|
||||||
|
pdfmark=input_ps_stub,
|
||||||
output_file=output_file,
|
output_file=output_file,
|
||||||
compression=options.pdfa_image_compression,
|
compression=options.pdfa_image_compression,
|
||||||
log=context.log,
|
|
||||||
pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3
|
pdfa_part=options.output_type[-1], # is pdfa-1, pdfa-2, or pdfa-3
|
||||||
)
|
)
|
||||||
|
|
||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def should_linearize(working_file, context):
|
def should_linearize(working_file: Path, context: PdfContext):
|
||||||
filesize = os.stat(working_file).st_size
|
filesize = os.stat(working_file).st_size
|
||||||
if filesize > (context.options.fast_web_view * 1_000_000):
|
if filesize > (context.options.fast_web_view * 1_000_000):
|
||||||
return True
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
def metadata_fixup(working_file, context):
|
def metadata_fixup(working_file: Path, context: PdfContext):
|
||||||
output_file = context.get_path('metafix.pdf')
|
output_file = context.get_path('metafix.pdf')
|
||||||
options = context.options
|
options = context.options
|
||||||
|
|
||||||
@@ -738,25 +735,21 @@ def metadata_fixup(working_file, context):
|
|||||||
if not missing:
|
if not missing:
|
||||||
return
|
return
|
||||||
if options.output_type.startswith('pdfa'):
|
if options.output_type.startswith('pdfa'):
|
||||||
context.log.warning(
|
log.warning(
|
||||||
"Some input metadata could not be copied because it is not "
|
"Some input metadata could not be copied because it is not "
|
||||||
"permitted in PDF/A. You may wish to examine the output "
|
"permitted in PDF/A. You may wish to examine the output "
|
||||||
"PDF's XMP metadata."
|
"PDF's XMP metadata."
|
||||||
)
|
)
|
||||||
context.log.debug(
|
log.debug("The following metadata fields were not copied: %r", missing)
|
||||||
"The following metadata fields were not copied: %r", missing
|
|
||||||
)
|
|
||||||
else:
|
else:
|
||||||
context.log.error(
|
log.error(
|
||||||
"Some input metadata could not be copied."
|
"Some input metadata could not be copied."
|
||||||
"You may wish to examine the output PDF's XMP metadata."
|
"You may wish to examine the output PDF's XMP metadata."
|
||||||
)
|
)
|
||||||
context.log.info(
|
log.info("The following metadata fields were not copied: %r", missing)
|
||||||
"The following metadata fields were not copied: %r", missing
|
|
||||||
)
|
|
||||||
|
|
||||||
with pikepdf.open(context.origin) as original, pikepdf.open(working_file) as pdf:
|
with pikepdf.open(context.origin) as original, pikepdf.open(working_file) as pdf:
|
||||||
docinfo = get_docinfo(original, options)
|
docinfo = get_docinfo(original, context)
|
||||||
with pdf.open_metadata() as meta:
|
with pdf.open_metadata() as meta:
|
||||||
meta.load_from_docinfo(docinfo, delete_missing=False, raise_failure=False)
|
meta.load_from_docinfo(docinfo, delete_missing=False, raise_failure=False)
|
||||||
# If xmp:CreateDate is missing, set it to the modify date to
|
# If xmp:CreateDate is missing, set it to the modify date to
|
||||||
@@ -764,6 +757,14 @@ def metadata_fixup(working_file, context):
|
|||||||
if 'xmp:CreateDate' not in meta:
|
if 'xmp:CreateDate' not in meta:
|
||||||
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
meta['xmp:CreateDate'] = meta.get('xmp:ModifyDate', '')
|
||||||
|
|
||||||
|
# Ghostscript likes to set title to Untitled if omitted from input.
|
||||||
|
# Reverse this, because PDF/A TechNote 0003:Metadata in PDF/A-1
|
||||||
|
# and the XMP Spec do not make this recommendation.
|
||||||
|
if meta.get('dc:title') == 'Untitled':
|
||||||
|
with original.open_metadata() as original_meta:
|
||||||
|
if 'dc:title' not in original_meta:
|
||||||
|
del meta['dc:title']
|
||||||
|
|
||||||
meta_original = original.open_metadata()
|
meta_original = original.open_metadata()
|
||||||
missing = set(meta_original.keys()) - set(meta.keys())
|
missing = set(meta_original.keys()) - set(meta.keys())
|
||||||
report_on_metadata(missing)
|
report_on_metadata(missing)
|
||||||
@@ -783,7 +784,7 @@ def metadata_fixup(working_file, context):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def optimize_pdf(input_file, context):
|
def optimize_pdf(input_file: Path, context: PdfContext):
|
||||||
output_file = context.get_path('optimize.pdf')
|
output_file = context.get_path('optimize.pdf')
|
||||||
save_settings = dict(
|
save_settings = dict(
|
||||||
compress_streams=True,
|
compress_streams=True,
|
||||||
@@ -795,7 +796,7 @@ def optimize_pdf(input_file, context):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def merge_sidecars(txt_files, context):
|
def merge_sidecars(txt_files: Iterable[Optional[Path]], context: PdfContext):
|
||||||
output_file = context.get_path('sidecar.txt')
|
output_file = context.get_path('sidecar.txt')
|
||||||
with open(output_file, 'w', encoding="utf-8") as stream:
|
with open(output_file, 'w', encoding="utf-8") as stream:
|
||||||
for page_num, txt_file in enumerate(txt_files):
|
for page_num, txt_file in enumerate(txt_files):
|
||||||
@@ -804,11 +805,9 @@ def merge_sidecars(txt_files, context):
|
|||||||
if txt_file:
|
if txt_file:
|
||||||
with open(txt_file, 'r', encoding="utf-8") as in_:
|
with open(txt_file, 'r', encoding="utf-8") as in_:
|
||||||
txt = in_.read()
|
txt = in_.read()
|
||||||
# Tesseract v4 alpha started adding form feeds in
|
# Some OCR engines (e.g. Tesseract v4 alpha) add form feeds
|
||||||
# commit aa6eb6b
|
# between pages, and some do not. For consistency, we ignore
|
||||||
# No obvious way to detect what binaries will do this, so
|
# any added by the OCR engine and them on our own.
|
||||||
# for consistency just ignore its form feeds and insert our
|
|
||||||
# own
|
|
||||||
if txt.endswith('\f'):
|
if txt.endswith('\f'):
|
||||||
stream.write(txt[:-1])
|
stream.write(txt[:-1])
|
||||||
else:
|
else:
|
||||||
@@ -818,12 +817,17 @@ def merge_sidecars(txt_files, context):
|
|||||||
return output_file
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
def copy_final(input_file, output_file, context):
|
def copy_final(input_file, output_file, _context: PdfContext):
|
||||||
context.log.debug('%s -> %s', input_file, output_file)
|
log.debug('%s -> %s', input_file, output_file)
|
||||||
with open(input_file, 'rb') as input_stream:
|
with open(input_file, 'rb') as input_stream:
|
||||||
if output_file == '-':
|
if output_file == '-':
|
||||||
copyfileobj(input_stream, sys.stdout.buffer)
|
copyfileobj(input_stream, sys.stdout.buffer)
|
||||||
sys.stdout.flush()
|
sys.stdout.flush()
|
||||||
|
elif hasattr(output_file, 'writable'):
|
||||||
|
output_stream = output_file
|
||||||
|
copyfileobj(input_stream, output_stream)
|
||||||
|
with suppress(AttributeError):
|
||||||
|
output_stream.flush()
|
||||||
else:
|
else:
|
||||||
# At this point we overwrite the output_file specified by the user
|
# At this point we overwrite the output_file specified by the user
|
||||||
# use copyfileobj because then we use open() to create the file and
|
# use copyfileobj because then we use open() to create the file and
|
||||||
|
|||||||
@@ -0,0 +1,109 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This file is part of OCRmyPDF.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import importlib
|
||||||
|
import importlib.util
|
||||||
|
import sys
|
||||||
|
from functools import partial
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Callable, List, Tuple
|
||||||
|
|
||||||
|
import pluggy
|
||||||
|
|
||||||
|
from ocrmypdf import pluginspec
|
||||||
|
from ocrmypdf.cli import get_parser, plugins_only_parser
|
||||||
|
|
||||||
|
|
||||||
|
class OcrmypdfPluginManager(pluggy.PluginManager):
|
||||||
|
"""pluggy.PluginManager that can fork.
|
||||||
|
|
||||||
|
Capable of reconstructing itself in child workers.
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
setup_func: callback that initializes the plugin manager with all
|
||||||
|
standard plugins
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self, *args, setup_func: Callable[[pluggy.PluginManager], None], **kwargs
|
||||||
|
):
|
||||||
|
self._init_args = args
|
||||||
|
self._setup_func = setup_func
|
||||||
|
self._init_kwargs = kwargs
|
||||||
|
super().__init__(*args, **kwargs)
|
||||||
|
setup_func(self)
|
||||||
|
|
||||||
|
def __getstate__(self):
|
||||||
|
state = dict(
|
||||||
|
_init_args=self._init_args,
|
||||||
|
_setup_func=self._setup_func,
|
||||||
|
_init_kwargs=self._init_kwargs,
|
||||||
|
)
|
||||||
|
return state
|
||||||
|
|
||||||
|
def __setstate__(self, state):
|
||||||
|
self.__init__(
|
||||||
|
*state['_init_args'],
|
||||||
|
setup_func=state['_setup_func'],
|
||||||
|
**state['_init_kwargs'],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _setup_plugins(pm: pluggy.PluginManager, plugins: List[str], builtins: bool = True):
|
||||||
|
pm.add_hookspecs(pluginspec)
|
||||||
|
|
||||||
|
if builtins:
|
||||||
|
all_plugins = [
|
||||||
|
'ocrmypdf.builtin_plugins.ghostscript',
|
||||||
|
'ocrmypdf.builtin_plugins.tesseract_ocr',
|
||||||
|
] + plugins
|
||||||
|
else:
|
||||||
|
all_plugins = plugins
|
||||||
|
for name in all_plugins:
|
||||||
|
if name.endswith('.py'):
|
||||||
|
# Import by filename
|
||||||
|
module_name = Path(name).stem
|
||||||
|
spec = importlib.util.spec_from_file_location(module_name, name)
|
||||||
|
module = importlib.util.module_from_spec(spec)
|
||||||
|
sys.modules[module_name] = module
|
||||||
|
spec.loader.exec_module(module)
|
||||||
|
else:
|
||||||
|
# Import by dotted module name
|
||||||
|
module = importlib.import_module(name)
|
||||||
|
pm.register(module)
|
||||||
|
|
||||||
|
|
||||||
|
def get_plugin_manager(plugins: List[str], builtins=True):
|
||||||
|
pm = OcrmypdfPluginManager(
|
||||||
|
project_name='ocrmypdf',
|
||||||
|
setup_func=partial(_setup_plugins, plugins=plugins, builtins=builtins),
|
||||||
|
)
|
||||||
|
return pm
|
||||||
|
|
||||||
|
|
||||||
|
def get_parser_options_plugins(
|
||||||
|
args,
|
||||||
|
) -> Tuple[argparse.ArgumentParser, argparse.Namespace, pluggy.PluginManager]:
|
||||||
|
pre_options, _unused = plugins_only_parser.parse_known_args(args=args)
|
||||||
|
plugin_manager = get_plugin_manager(pre_options.plugins)
|
||||||
|
|
||||||
|
parser = get_parser()
|
||||||
|
plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
|
||||||
|
|
||||||
|
options = parser.parse_args(args=args)
|
||||||
|
return parser, options, plugin_manager
|
||||||
+178
-214
@@ -17,21 +17,20 @@
|
|||||||
|
|
||||||
import logging
|
import logging
|
||||||
import logging.handlers
|
import logging.handlers
|
||||||
import multiprocessing
|
|
||||||
import os
|
import os
|
||||||
import signal
|
|
||||||
import sys
|
import sys
|
||||||
import threading
|
import threading
|
||||||
from collections import namedtuple
|
from functools import partial
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from tempfile import mkdtemp
|
from tempfile import mkdtemp
|
||||||
|
from typing import List, NamedTuple, Optional, Tuple
|
||||||
|
|
||||||
import PIL
|
import PIL
|
||||||
from tqdm import tqdm
|
|
||||||
|
|
||||||
from ._graft import OcrGrafter
|
from ocrmypdf._concurrent import exec_progress_pool
|
||||||
from ._jobcontext import PDFContext, cleanup_working_files, make_logger
|
from ocrmypdf._graft import OcrGrafter
|
||||||
from ._pipeline import (
|
from ocrmypdf._jobcontext import PageContext, PdfContext, cleanup_working_files
|
||||||
|
from ocrmypdf._pipeline import (
|
||||||
convert_to_pdfa,
|
convert_to_pdfa,
|
||||||
copy_final,
|
copy_final,
|
||||||
create_ocr_image,
|
create_ocr_image,
|
||||||
@@ -43,8 +42,8 @@ from ._pipeline import (
|
|||||||
is_ocr_required,
|
is_ocr_required,
|
||||||
merge_sidecars,
|
merge_sidecars,
|
||||||
metadata_fixup,
|
metadata_fixup,
|
||||||
ocr_tesseract_hocr,
|
ocr_engine_hocr,
|
||||||
ocr_tesseract_textonly_pdf,
|
ocr_engine_textonly_pdf,
|
||||||
optimize_pdf,
|
optimize_pdf,
|
||||||
preprocess_clean,
|
preprocess_clean,
|
||||||
preprocess_deskew,
|
preprocess_deskew,
|
||||||
@@ -56,22 +55,51 @@ from ._pipeline import (
|
|||||||
triage,
|
triage,
|
||||||
validate_pdfinfo_options,
|
validate_pdfinfo_options,
|
||||||
)
|
)
|
||||||
from ._validation import (
|
from ocrmypdf._plugin_manager import get_plugin_manager
|
||||||
|
from ocrmypdf._validation import (
|
||||||
check_requested_output_file,
|
check_requested_output_file,
|
||||||
create_input_file,
|
create_input_file,
|
||||||
report_output_file_size,
|
report_output_file_size,
|
||||||
)
|
)
|
||||||
from .exceptions import ExitCode, ExitCodeException
|
from ocrmypdf.exceptions import ExitCode, ExitCodeException
|
||||||
from .exec import qpdf
|
from ocrmypdf.helpers import available_cpu_count, check_pdf, samefile
|
||||||
from .helpers import available_cpu_count
|
from ocrmypdf.pdfa import file_claims_pdfa
|
||||||
from .pdfa import file_claims_pdfa
|
|
||||||
|
|
||||||
PageResult = namedtuple(
|
log = logging.getLogger(__name__)
|
||||||
'PageResult', 'pageno, pdf_page_from_image, ocr, text, orientation_correction'
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def preprocess(page_context, image, remove_background, deskew, clean):
|
class PageResult(NamedTuple):
|
||||||
|
pageno: int
|
||||||
|
pdf_page_from_image: Optional[Path]
|
||||||
|
ocr: Optional[Path]
|
||||||
|
text: Optional[Path]
|
||||||
|
orientation_correction: int
|
||||||
|
|
||||||
|
|
||||||
|
tls = threading.local()
|
||||||
|
tls.pageno = None
|
||||||
|
|
||||||
|
|
||||||
|
old_factory = logging.getLogRecordFactory()
|
||||||
|
|
||||||
|
|
||||||
|
def record_factory(*args, **kwargs):
|
||||||
|
record = old_factory(*args, **kwargs)
|
||||||
|
if hasattr(tls, 'pageno'):
|
||||||
|
record.pageno = tls.pageno
|
||||||
|
return record
|
||||||
|
|
||||||
|
|
||||||
|
logging.setLogRecordFactory(record_factory)
|
||||||
|
|
||||||
|
|
||||||
|
def preprocess(
|
||||||
|
page_context: PageContext,
|
||||||
|
image: Path,
|
||||||
|
remove_background: bool,
|
||||||
|
deskew: bool,
|
||||||
|
clean: bool,
|
||||||
|
) -> Path:
|
||||||
if remove_background:
|
if remove_background:
|
||||||
image = preprocess_remove_background(image, page_context)
|
image = preprocess_remove_background(image, page_context)
|
||||||
if deskew:
|
if deskew:
|
||||||
@@ -81,54 +109,55 @@ def preprocess(page_context, image, remove_background, deskew, clean):
|
|||||||
return image
|
return image
|
||||||
|
|
||||||
|
|
||||||
def exec_page_sync(page_context):
|
def make_intermediate_images(
|
||||||
|
page_context: PageContext, orientation_correction: int
|
||||||
|
) -> Tuple[Path, Optional[Path]]:
|
||||||
options = page_context.options
|
options = page_context.options
|
||||||
orientation_correction = 0
|
|
||||||
pdf_page_from_image_out = None
|
|
||||||
ocr_out = None
|
|
||||||
text_out = None
|
|
||||||
if is_ocr_required(page_context):
|
|
||||||
if options.rotate_pages:
|
|
||||||
# Rasterize
|
|
||||||
rasterize_preview_out = rasterize_preview(page_context.origin, page_context)
|
|
||||||
orientation_correction = get_orientation_correction(
|
|
||||||
rasterize_preview_out, page_context
|
|
||||||
)
|
|
||||||
|
|
||||||
rasterize_out = rasterize(
|
ocr_image = preprocess_out = None
|
||||||
page_context.origin,
|
rasterize_out = rasterize(
|
||||||
|
page_context.origin,
|
||||||
|
page_context,
|
||||||
|
correction=orientation_correction,
|
||||||
|
remove_vectors=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
if not any([options.clean, options.clean_final, options.remove_vectors]):
|
||||||
|
ocr_image = preprocess_out = preprocess(
|
||||||
page_context,
|
page_context,
|
||||||
correction=orientation_correction,
|
rasterize_out,
|
||||||
remove_vectors=False,
|
options.remove_background,
|
||||||
|
options.deskew,
|
||||||
|
clean=False,
|
||||||
)
|
)
|
||||||
|
else:
|
||||||
if not any([options.clean, options.clean_final, options.remove_vectors]):
|
if not options.lossless_reconstruction:
|
||||||
ocr_image = preprocess_out = preprocess(
|
preprocess_out = preprocess(
|
||||||
page_context,
|
page_context,
|
||||||
rasterize_out,
|
rasterize_out,
|
||||||
options.remove_background,
|
options.remove_background,
|
||||||
options.deskew,
|
options.deskew,
|
||||||
clean=False,
|
clean=options.clean_final,
|
||||||
|
)
|
||||||
|
if options.remove_vectors:
|
||||||
|
rasterize_ocr_out = rasterize(
|
||||||
|
page_context.origin,
|
||||||
|
page_context,
|
||||||
|
correction=orientation_correction,
|
||||||
|
remove_vectors=True,
|
||||||
|
output_tag='_ocr',
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
if not options.lossless_reconstruction:
|
rasterize_ocr_out = rasterize_out
|
||||||
preprocess_out = preprocess(
|
|
||||||
page_context,
|
if (
|
||||||
rasterize_out,
|
preprocess_out
|
||||||
options.remove_background,
|
and rasterize_ocr_out == rasterize_out
|
||||||
options.deskew,
|
and options.clean == options.clean_final
|
||||||
clean=options.clean_final,
|
):
|
||||||
)
|
# Optimization: image for OCR is identical to presentation image
|
||||||
if options.remove_vectors:
|
ocr_image = preprocess_out
|
||||||
rasterize_ocr_out = rasterize(
|
else:
|
||||||
page_context.origin,
|
|
||||||
page_context,
|
|
||||||
correction=orientation_correction,
|
|
||||||
remove_vectors=True,
|
|
||||||
output_tag='_ocr',
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
rasterize_ocr_out = rasterize_out
|
|
||||||
ocr_image = preprocess(
|
ocr_image = preprocess(
|
||||||
page_context,
|
page_context,
|
||||||
rasterize_ocr_out,
|
rasterize_ocr_out,
|
||||||
@@ -136,28 +165,56 @@ def exec_page_sync(page_context):
|
|||||||
options.deskew,
|
options.deskew,
|
||||||
clean=options.clean,
|
clean=options.clean,
|
||||||
)
|
)
|
||||||
|
return ocr_image, preprocess_out
|
||||||
|
|
||||||
ocr_image_out = create_ocr_image(ocr_image, page_context)
|
|
||||||
|
|
||||||
pdf_page_from_image_out = None
|
def exec_page_sync(page_context: PageContext):
|
||||||
if not options.lossless_reconstruction:
|
options = page_context.options
|
||||||
visible_image_out = preprocess_out
|
tls.pageno = page_context.pageno + 1
|
||||||
if should_visible_page_image_use_jpg(page_context.pageinfo):
|
|
||||||
visible_image_out = create_visible_page_jpg(
|
|
||||||
visible_image_out, page_context
|
|
||||||
)
|
|
||||||
pdf_page_from_image_out = create_pdf_page_from_image(
|
|
||||||
visible_image_out, page_context
|
|
||||||
)
|
|
||||||
|
|
||||||
if options.pdf_renderer == 'hocr':
|
if not is_ocr_required(page_context):
|
||||||
(hocr_out, text_out) = ocr_tesseract_hocr(ocr_image_out, page_context)
|
return PageResult(
|
||||||
ocr_out = render_hocr_page(hocr_out, page_context)
|
pageno=page_context.pageno,
|
||||||
|
pdf_page_from_image=None,
|
||||||
|
ocr=None,
|
||||||
|
text=None,
|
||||||
|
orientation_correction=0,
|
||||||
|
)
|
||||||
|
|
||||||
if options.pdf_renderer == 'sandwich':
|
orientation_correction = 0
|
||||||
(ocr_out, text_out) = ocr_tesseract_textonly_pdf(
|
if options.rotate_pages:
|
||||||
ocr_image_out, page_context
|
# Rasterize
|
||||||
)
|
rasterize_preview_out = rasterize_preview(page_context.origin, page_context)
|
||||||
|
orientation_correction = get_orientation_correction(
|
||||||
|
rasterize_preview_out, page_context
|
||||||
|
)
|
||||||
|
|
||||||
|
ocr_image, preprocess_out = make_intermediate_images(
|
||||||
|
page_context, orientation_correction
|
||||||
|
)
|
||||||
|
ocr_image_out = create_ocr_image(ocr_image, page_context)
|
||||||
|
|
||||||
|
pdf_page_from_image_out = None
|
||||||
|
if not options.lossless_reconstruction:
|
||||||
|
assert preprocess_out
|
||||||
|
visible_image_out = preprocess_out
|
||||||
|
if should_visible_page_image_use_jpg(page_context.pageinfo):
|
||||||
|
visible_image_out = create_visible_page_jpg(visible_image_out, page_context)
|
||||||
|
filtered_image = page_context.plugin_manager.hook.filter_page_image(
|
||||||
|
page=page_context, image_filename=visible_image_out
|
||||||
|
)
|
||||||
|
if filtered_image:
|
||||||
|
visible_image_out = filtered_image
|
||||||
|
pdf_page_from_image_out = create_pdf_page_from_image(
|
||||||
|
visible_image_out, page_context
|
||||||
|
)
|
||||||
|
|
||||||
|
if options.pdf_renderer == 'hocr':
|
||||||
|
(hocr_out, text_out) = ocr_engine_hocr(ocr_image_out, page_context)
|
||||||
|
ocr_out = render_hocr_page(hocr_out, page_context)
|
||||||
|
|
||||||
|
if options.pdf_renderer == 'sandwich':
|
||||||
|
(ocr_out, text_out) = ocr_engine_textonly_pdf(ocr_image_out, page_context)
|
||||||
|
|
||||||
return PageResult(
|
return PageResult(
|
||||||
pageno=page_context.pageno,
|
pageno=page_context.pageno,
|
||||||
@@ -168,7 +225,7 @@ def exec_page_sync(page_context):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def post_process(pdf_file, context):
|
def post_process(pdf_file, context: PdfContext):
|
||||||
pdf_out = pdf_file
|
pdf_out = pdf_file
|
||||||
if context.options.output_type.startswith('pdfa'):
|
if context.options.output_type.startswith('pdfa'):
|
||||||
ps_stub_out = generate_postscript_stub(context)
|
ps_stub_out = generate_postscript_stub(context)
|
||||||
@@ -178,138 +235,50 @@ def post_process(pdf_file, context):
|
|||||||
return optimize_pdf(pdf_out, context)
|
return optimize_pdf(pdf_out, context)
|
||||||
|
|
||||||
|
|
||||||
def worker_init(queue, max_pixels):
|
def worker_init(max_pixels: int):
|
||||||
"""Initialize a process pool worker"""
|
|
||||||
|
|
||||||
# Ignore SIGINT (our parent process will kill us gracefully)
|
|
||||||
signal.signal(signal.SIGINT, signal.SIG_IGN)
|
|
||||||
|
|
||||||
# Reconfigure the root logger for this process to send all messages to a queue
|
|
||||||
h = logging.handlers.QueueHandler(queue)
|
|
||||||
root = logging.getLogger()
|
|
||||||
root.handlers = []
|
|
||||||
root.addHandler(h)
|
|
||||||
|
|
||||||
# In Windows, child process will not inherit our change to this value in
|
# In Windows, child process will not inherit our change to this value in
|
||||||
# the parent process, so ensure workers get it set
|
# the parent process, so ensure workers get it set. Not needed when running
|
||||||
|
# threaded, but harmless to set again.
|
||||||
PIL.Image.MAX_IMAGE_PIXELS = max_pixels
|
PIL.Image.MAX_IMAGE_PIXELS = max_pixels
|
||||||
|
|
||||||
|
|
||||||
def worker_thread_init(_queue, max_pixels):
|
def exec_concurrent(context: PdfContext):
|
||||||
# This is probably not needed since threads should all see the same memory,
|
|
||||||
# but done for consistency.
|
|
||||||
PIL.Image.MAX_IMAGE_PIXELS = max_pixels
|
|
||||||
|
|
||||||
|
|
||||||
def log_listener(queue):
|
|
||||||
"""Listen to the worker processes and forward the messages to logging
|
|
||||||
|
|
||||||
For simplicity this is a thread rather than a process. Only one process
|
|
||||||
should actually write to sys.stderr or whatever we're using, so if this is
|
|
||||||
made into a process the main application needs to be directed to it.
|
|
||||||
|
|
||||||
See https://docs.python.org/3/howto/logging-cookbook.html#logging-to-a-single-file-from-multiple-processes
|
|
||||||
"""
|
|
||||||
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
record = queue.get()
|
|
||||||
if record is None:
|
|
||||||
break
|
|
||||||
logger = logging.getLogger(record.name)
|
|
||||||
logger.handle(record)
|
|
||||||
except Exception:
|
|
||||||
import traceback
|
|
||||||
|
|
||||||
print("Logging problem", file=sys.stderr)
|
|
||||||
traceback.print_exc(file=sys.stderr)
|
|
||||||
|
|
||||||
|
|
||||||
def exec_concurrent(context):
|
|
||||||
"""Execute the pipeline concurrently"""
|
"""Execute the pipeline concurrently"""
|
||||||
|
|
||||||
# Run exec_page_sync on every page context
|
# Run exec_page_sync on every page context
|
||||||
max_workers = min(len(context.pdfinfo), context.options.jobs)
|
max_workers = min(len(context.pdfinfo), context.options.jobs)
|
||||||
if max_workers > 1:
|
if max_workers > 1:
|
||||||
context.log.info("Start processing %d pages concurrently", max_workers)
|
log.info("Start processing %d pages concurrently", max_workers)
|
||||||
|
|
||||||
# Tesseract 4.x can be multithreaded, and we also run multiple workers. We want
|
sidecars: List[Optional[Path]] = [None] * len(context.pdfinfo)
|
||||||
# to manage how many threads it uses to avoid creating total threads than cores.
|
|
||||||
# Performance testing shows we're better off
|
|
||||||
# parallelizing ocrmypdf and forcing Tesseract to be single threaded, which we
|
|
||||||
# get by setting the envvar OMP_THREAD_LIMIT to 1. But if the page count of the
|
|
||||||
# input file is small, then we allow Tesseract to use threads, subject to the
|
|
||||||
# constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers.
|
|
||||||
# As of Tesseract 4.1, 3 threads is the most effective on a 4 core/8 thread system.
|
|
||||||
tess_threads = min(3, context.options.jobs // max_workers)
|
|
||||||
if context.options.tesseract_env is None:
|
|
||||||
context.options.tesseract_env = os.environ.copy()
|
|
||||||
context.options.tesseract_env.setdefault('OMP_THREAD_LIMIT', str(tess_threads))
|
|
||||||
try:
|
|
||||||
tess_threads = int(context.options.tesseract_env['OMP_THREAD_LIMIT'])
|
|
||||||
except ValueError: # OMP_THREAD_LIMIT initialized to non-numeric
|
|
||||||
context.log.error("Environment variable OMP_THREAD_LIMIT is not numeric")
|
|
||||||
if tess_threads > 1:
|
|
||||||
context.log.info("Using Tesseract OpenMP thread limit %d", tess_threads)
|
|
||||||
|
|
||||||
if context.options.use_threads:
|
|
||||||
from multiprocessing.dummy import Pool
|
|
||||||
|
|
||||||
initializer = worker_thread_init
|
|
||||||
else:
|
|
||||||
Pool = multiprocessing.Pool
|
|
||||||
initializer = worker_init
|
|
||||||
|
|
||||||
sidecars = [None] * len(context.pdfinfo)
|
|
||||||
ocrgraft = OcrGrafter(context)
|
ocrgraft = OcrGrafter(context)
|
||||||
|
|
||||||
log_queue = multiprocessing.Queue(-1)
|
def update_page(result: PageResult, pbar):
|
||||||
listener = threading.Thread(target=log_listener, args=(log_queue,))
|
sidecars[result.pageno] = result.text
|
||||||
listener.start()
|
pbar.update()
|
||||||
with tqdm(
|
ocrgraft.graft_page(
|
||||||
total=(2 * len(context.pdfinfo)),
|
pageno=result.pageno,
|
||||||
desc='OCR',
|
image=result.pdf_page_from_image,
|
||||||
unit='page',
|
textpdf=result.ocr,
|
||||||
unit_scale=0.5,
|
autorotate_correction=result.orientation_correction,
|
||||||
disable=not context.options.progress_bar,
|
|
||||||
) as pbar:
|
|
||||||
pool = Pool(
|
|
||||||
processes=max_workers,
|
|
||||||
initializer=initializer,
|
|
||||||
initargs=(log_queue, PIL.Image.MAX_IMAGE_PIXELS),
|
|
||||||
)
|
)
|
||||||
try:
|
pbar.update()
|
||||||
results = pool.imap_unordered(exec_page_sync, context.get_page_contexts())
|
|
||||||
while True:
|
|
||||||
try:
|
|
||||||
page_result = results.next()
|
|
||||||
sidecars[page_result.pageno] = page_result.text
|
|
||||||
pbar.update()
|
|
||||||
ocrgraft.graft_page(page_result)
|
|
||||||
pbar.update()
|
|
||||||
except StopIteration:
|
|
||||||
break
|
|
||||||
except KeyboardInterrupt:
|
|
||||||
# Terminate pool so we exit instantly
|
|
||||||
pool.terminate()
|
|
||||||
# Don't try listener.join() here, will deadlock
|
|
||||||
raise
|
|
||||||
except Exception:
|
|
||||||
if not os.environ.get("PYTEST_CURRENT_TEST", ""):
|
|
||||||
# Unless inside pytest, exit immediately because no one wants
|
|
||||||
# to wait for child processes to finalize results that will be
|
|
||||||
# thrown away. Inside pytest, we want child processes to exit
|
|
||||||
# cleanly so that they output an error messages or coverage data
|
|
||||||
# we need from them.
|
|
||||||
pool.terminate()
|
|
||||||
raise
|
|
||||||
finally:
|
|
||||||
# Terminate log listener
|
|
||||||
log_queue.put_nowait(None)
|
|
||||||
pool.close()
|
|
||||||
pool.join()
|
|
||||||
|
|
||||||
listener.join()
|
exec_progress_pool(
|
||||||
|
use_threads=context.options.use_threads,
|
||||||
|
max_workers=max_workers,
|
||||||
|
tqdm_kwargs=dict(
|
||||||
|
total=(2 * len(context.pdfinfo)),
|
||||||
|
desc='OCR',
|
||||||
|
unit='page',
|
||||||
|
unit_scale=0.5,
|
||||||
|
disable=not context.options.progress_bar,
|
||||||
|
),
|
||||||
|
task_initializer=partial(worker_init, PIL.Image.MAX_IMAGE_PIXELS),
|
||||||
|
task=exec_page_sync,
|
||||||
|
task_arguments=context.get_page_contexts(),
|
||||||
|
task_finished=update_page,
|
||||||
|
)
|
||||||
|
|
||||||
# Output sidecar text
|
# Output sidecar text
|
||||||
if context.options.sidecar:
|
if context.options.sidecar:
|
||||||
@@ -333,34 +302,27 @@ class NeverRaise(Exception):
|
|||||||
pass # pylint: disable=unnecessary-pass
|
pass # pylint: disable=unnecessary-pass
|
||||||
|
|
||||||
|
|
||||||
def samefile(f1, f2):
|
|
||||||
if os.name == 'nt':
|
|
||||||
return f1 == f2
|
|
||||||
else:
|
|
||||||
return os.path.samefile(f1, f2)
|
|
||||||
|
|
||||||
|
|
||||||
def configure_debug_logging(log_filename, prefix=''):
|
def configure_debug_logging(log_filename, prefix=''):
|
||||||
log_file_handler = logging.FileHandler(log_filename, delay=True)
|
log_file_handler = logging.FileHandler(log_filename, delay=True)
|
||||||
log_file_handler.setLevel(logging.DEBUG)
|
log_file_handler.setLevel(logging.DEBUG)
|
||||||
formatter = logging.Formatter(
|
formatter = logging.Formatter(
|
||||||
'[%(asctime)s] - %(name)s - %(levelname)7s - %(message)s'
|
'[%(asctime)s] - %(name)s - %(levelname)7s -%(pageno)s %(message)s'
|
||||||
)
|
)
|
||||||
log_file_handler.setFormatter(formatter)
|
log_file_handler.setFormatter(formatter)
|
||||||
logging.getLogger(prefix).addHandler(log_file_handler)
|
logging.getLogger(prefix).addHandler(log_file_handler)
|
||||||
return log_file_handler
|
return log_file_handler
|
||||||
|
|
||||||
|
|
||||||
def run_pipeline(options, api=False):
|
def run_pipeline(options, *, plugin_manager, api=False):
|
||||||
log = make_logger(options, __name__)
|
|
||||||
|
|
||||||
# Any changes to options will not take effect for options that are already
|
# Any changes to options will not take effect for options that are already
|
||||||
# bound to function parameters in the pipeline. (For example
|
# bound to function parameters in the pipeline. (For example
|
||||||
# options.input_file, options.pdf_renderer are already bound.)
|
# options.input_file, options.pdf_renderer are already bound.)
|
||||||
if not options.jobs:
|
if not options.jobs:
|
||||||
options.jobs = available_cpu_count()
|
options.jobs = available_cpu_count()
|
||||||
|
if not plugin_manager:
|
||||||
|
plugin_manager = get_plugin_manager(options.plugins)
|
||||||
|
|
||||||
work_folder = mkdtemp(prefix="com.github.ocrmypdf.")
|
work_folder = Path(mkdtemp(prefix="com.github.ocrmypdf."))
|
||||||
debug_log_handler = None
|
debug_log_handler = None
|
||||||
if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get(
|
if (options.keep_temporary_files or options.verbose >= 1) and not os.environ.get(
|
||||||
'PYTEST_CURRENT_TEST', ''
|
'PYTEST_CURRENT_TEST', ''
|
||||||
@@ -373,21 +335,19 @@ def run_pipeline(options, api=False):
|
|||||||
|
|
||||||
# Triage image or pdf
|
# Triage image or pdf
|
||||||
origin_pdf = triage(
|
origin_pdf = triage(
|
||||||
original_filename,
|
original_filename, start_input_file, work_folder / 'origin.pdf', options
|
||||||
start_input_file,
|
|
||||||
os.path.join(work_folder, 'origin.pdf'),
|
|
||||||
options,
|
|
||||||
log,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# Gather pdfinfo and create context
|
# Gather pdfinfo and create context
|
||||||
pdfinfo = get_pdfinfo(
|
pdfinfo = get_pdfinfo(
|
||||||
origin_pdf,
|
origin_pdf,
|
||||||
detailed_page_analysis=options.redo_ocr,
|
detailed_analysis=options.redo_ocr,
|
||||||
progbar=options.progress_bar,
|
progbar=options.progress_bar,
|
||||||
|
max_workers=options.jobs if not options.use_threads else 1, # To help debug
|
||||||
|
check_pages=options.pages,
|
||||||
)
|
)
|
||||||
|
|
||||||
context = PDFContext(options, work_folder, origin_pdf, pdfinfo)
|
context = PdfContext(options, work_folder, origin_pdf, pdfinfo, plugin_manager)
|
||||||
|
|
||||||
# Validate options are okay for this pdf
|
# Validate options are okay for this pdf
|
||||||
validate_pdfinfo_options(context)
|
validate_pdfinfo_options(context)
|
||||||
@@ -397,6 +357,10 @@ def run_pipeline(options, api=False):
|
|||||||
|
|
||||||
if options.output_file == '-':
|
if options.output_file == '-':
|
||||||
log.info("Output sent to stdout")
|
log.info("Output sent to stdout")
|
||||||
|
elif (
|
||||||
|
hasattr(options.output_file, 'writable') and options.output_file.writable()
|
||||||
|
):
|
||||||
|
log.info("Output written to stream")
|
||||||
elif samefile(options.output_file, os.devnull):
|
elif samefile(options.output_file, os.devnull):
|
||||||
pass # Say nothing when sending to dev null
|
pass # Say nothing when sending to dev null
|
||||||
else:
|
else:
|
||||||
@@ -412,7 +376,7 @@ def run_pipeline(options, api=False):
|
|||||||
pdfa_info['conformance'],
|
pdfa_info['conformance'],
|
||||||
)
|
)
|
||||||
return ExitCode.pdfa_conversion_failed
|
return ExitCode.pdfa_conversion_failed
|
||||||
if not qpdf.check(options.output_file, log):
|
if not check_pdf(options.output_file):
|
||||||
log.warning('Output file: The generated PDF is INVALID')
|
log.warning('Output file: The generated PDF is INVALID')
|
||||||
return ExitCode.invalid_output_pdf
|
return ExitCode.invalid_output_pdf
|
||||||
report_output_file_size(options, start_input_file, options.output_file)
|
report_output_file_size(options, start_input_file, options.output_file)
|
||||||
@@ -429,7 +393,7 @@ def run_pipeline(options, api=False):
|
|||||||
else:
|
else:
|
||||||
log.error(type(e).__name__)
|
log.error(type(e).__name__)
|
||||||
return e.exit_code
|
return e.exit_code
|
||||||
except (Exception if not api else NeverRaise) as e:
|
except (Exception if not api else NeverRaise) as e: # pylint: disable=broad-except
|
||||||
log.exception("An exception occurred while executing the pipeline")
|
log.exception("An exception occurred while executing the pipeline")
|
||||||
return ExitCode.other_error
|
return ExitCode.other_error
|
||||||
finally:
|
finally:
|
||||||
|
|||||||
+53
-106
@@ -21,28 +21,29 @@ import locale
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
import unicodedata
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from shutil import copyfileobj
|
from shutil import copyfileobj
|
||||||
|
from typing import Tuple
|
||||||
|
|
||||||
|
import pikepdf
|
||||||
import PIL
|
import PIL
|
||||||
|
|
||||||
from ._unicodefun import verify_python3_env
|
from ocrmypdf._exec import jbig2enc, pngquant, unpaper
|
||||||
from .exceptions import (
|
from ocrmypdf._unicodefun import verify_python3_env
|
||||||
|
from ocrmypdf.exceptions import (
|
||||||
BadArgsError,
|
BadArgsError,
|
||||||
InputFileError,
|
InputFileError,
|
||||||
MissingDependencyError,
|
MissingDependencyError,
|
||||||
OutputFileAccessError,
|
OutputFileAccessError,
|
||||||
)
|
)
|
||||||
from .exec import (
|
from ocrmypdf.helpers import (
|
||||||
check_external_program,
|
is_file_writable,
|
||||||
ghostscript,
|
is_iterable_notstr,
|
||||||
jbig2enc,
|
monotonic,
|
||||||
pngquant,
|
safe_symlink,
|
||||||
qpdf,
|
|
||||||
tesseract,
|
|
||||||
unpaper,
|
|
||||||
)
|
)
|
||||||
from .helpers import is_file_writable, is_iterable_notstr, monotonic, safe_symlink
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
# -------------
|
# -------------
|
||||||
# External dependencies
|
# External dependencies
|
||||||
@@ -67,35 +68,26 @@ def check_platform():
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def check_options_languages(options):
|
def check_options_languages(options, ocr_engine_languages):
|
||||||
if not options.language:
|
if not options.languages:
|
||||||
options.language = [DEFAULT_LANGUAGE]
|
options.languages = {DEFAULT_LANGUAGE}
|
||||||
system_lang = locale.getlocale()[0]
|
system_lang = locale.getlocale()[0]
|
||||||
if system_lang and not system_lang.startswith('en'):
|
if system_lang and not system_lang.startswith('en'):
|
||||||
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
log.debug("No language specified; assuming --language %s", DEFAULT_LANGUAGE)
|
||||||
|
if not ocr_engine_languages:
|
||||||
# Support v2.x "eng+deu" language syntax
|
return
|
||||||
if '+' in options.language[0]:
|
if not options.languages.issubset(ocr_engine_languages):
|
||||||
options.language = options.language[0].split('+')
|
|
||||||
|
|
||||||
languages = set(options.language)
|
|
||||||
if not languages.issubset(tesseract.languages()):
|
|
||||||
msg = (
|
msg = (
|
||||||
"The installed version of tesseract does not have language "
|
f"OCR engine does not have language data for the following "
|
||||||
"data for the following requested languages: \n"
|
"requested languages: \n"
|
||||||
)
|
)
|
||||||
for lang in languages - tesseract.languages():
|
for lang in options.languages - ocr_engine_languages:
|
||||||
msg += lang + '\n'
|
msg += lang + '\n'
|
||||||
raise MissingDependencyError(msg)
|
raise MissingDependencyError(msg)
|
||||||
|
|
||||||
|
|
||||||
def check_options_output(options):
|
def check_options_output(options):
|
||||||
# We have these constraints to check for.
|
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
||||||
# 1. Ghostscript < 9.20 mangles multibyte Unicode
|
|
||||||
# 2. hocr doesn't work on non-Latin languages (so don't select it)
|
|
||||||
|
|
||||||
languages = set(options.language)
|
|
||||||
is_latin = languages.issubset(HOCR_OK_LANGS)
|
|
||||||
|
|
||||||
if options.pdf_renderer == 'hocr' and not is_latin:
|
if options.pdf_renderer == 'hocr' and not is_latin:
|
||||||
msg = (
|
msg = (
|
||||||
@@ -105,37 +97,6 @@ def check_options_output(options):
|
|||||||
)
|
)
|
||||||
log.warning(msg)
|
log.warning(msg)
|
||||||
|
|
||||||
if ghostscript.version() < '9.20' and options.output_type != 'pdf' and not is_latin:
|
|
||||||
# https://bugs.ghostscript.com/show_bug.cgi?id=696874
|
|
||||||
# Ghostscript < 9.20 fails to encode multibyte characters properly
|
|
||||||
msg = (
|
|
||||||
"The installed version of Ghostscript does not work correctly "
|
|
||||||
"with the OCR languages you specified. Use --output-type pdf or "
|
|
||||||
"upgrade to Ghostscript 9.20 or later to avoid this issue."
|
|
||||||
)
|
|
||||||
msg += f"Found Ghostscript {ghostscript.version()}"
|
|
||||||
log.warning(msg)
|
|
||||||
|
|
||||||
# Decide on what renderer to use
|
|
||||||
if options.pdf_renderer == 'auto':
|
|
||||||
options.pdf_renderer = 'sandwich'
|
|
||||||
|
|
||||||
if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf(
|
|
||||||
options.tesseract_env, languages
|
|
||||||
):
|
|
||||||
raise MissingDependencyError(
|
|
||||||
"You are using an alpha version of Tesseract 4.0 that does not support "
|
|
||||||
"the textonly_pdf parameter. We don't support versions this old."
|
|
||||||
)
|
|
||||||
|
|
||||||
if options.output_type == 'pdfa':
|
|
||||||
options.output_type = 'pdfa-2'
|
|
||||||
|
|
||||||
if options.output_type == 'pdfa-3' and ghostscript.version() < '9.19':
|
|
||||||
raise MissingDependencyError(
|
|
||||||
"--output-type pdfa-3 requires Ghostscript 9.19 or later"
|
|
||||||
)
|
|
||||||
|
|
||||||
lossless_reconstruction = False
|
lossless_reconstruction = False
|
||||||
if not any(
|
if not any(
|
||||||
(
|
(
|
||||||
@@ -270,18 +231,9 @@ def check_options_advanced(options):
|
|||||||
"--pdfa-image-compression argument has no effect when "
|
"--pdfa-image-compression argument has no effect when "
|
||||||
"--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'"
|
"--output-type is not 'pdfa', 'pdfa-1', or 'pdfa-2'"
|
||||||
)
|
)
|
||||||
if not tesseract.has_user_words(options.tesseract_env) and (
|
|
||||||
options.user_words or options.user_patterns
|
|
||||||
):
|
|
||||||
log.warning(
|
|
||||||
"Tesseract 4.0 ignores --user-words and --user-patterns, so these "
|
|
||||||
"arguments have no effect."
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def check_options_metadata(options):
|
def check_options_metadata(options):
|
||||||
import unicodedata
|
|
||||||
|
|
||||||
docinfo = [options.title, options.author, options.keywords, options.subject]
|
docinfo = [options.title, options.author, options.keywords, options.subject]
|
||||||
for s in (m for m in docinfo if m):
|
for s in (m for m in docinfo if m):
|
||||||
for c in s:
|
for c in s:
|
||||||
@@ -300,9 +252,9 @@ def check_options_pillow(options):
|
|||||||
PIL.Image.MAX_IMAGE_PIXELS = None
|
PIL.Image.MAX_IMAGE_PIXELS = None
|
||||||
|
|
||||||
|
|
||||||
def check_options(options):
|
def _check_options(options, plugin_manager, ocr_engine_languages):
|
||||||
check_platform()
|
check_platform()
|
||||||
check_options_languages(options)
|
check_options_languages(options, ocr_engine_languages)
|
||||||
check_options_metadata(options)
|
check_options_metadata(options)
|
||||||
check_options_output(options)
|
check_options_output(options)
|
||||||
check_options_sidecar(options)
|
check_options_sidecar(options)
|
||||||
@@ -311,7 +263,12 @@ def check_options(options):
|
|||||||
check_options_optimizing(options)
|
check_options_optimizing(options)
|
||||||
check_options_advanced(options)
|
check_options_advanced(options)
|
||||||
check_options_pillow(options)
|
check_options_pillow(options)
|
||||||
check_dependency_versions(options)
|
plugin_manager.hook.check_options(options=options)
|
||||||
|
|
||||||
|
|
||||||
|
def check_options(options, plugin_manager):
|
||||||
|
ocr_engine_languages = plugin_manager.hook.get_ocr_engine().languages(options)
|
||||||
|
_check_options(options, plugin_manager, ocr_engine_languages)
|
||||||
|
|
||||||
|
|
||||||
def check_closed_streams(options): # pragma: no cover
|
def check_closed_streams(options): # pragma: no cover
|
||||||
@@ -373,17 +330,25 @@ def log_page_orientations(pdfinfo):
|
|||||||
log.info('Page orientations detected: %s', ' '.join(orientations))
|
log.info('Page orientations detected: %s', ' '.join(orientations))
|
||||||
|
|
||||||
|
|
||||||
def create_input_file(options, work_folder):
|
def create_input_file(options, work_folder: Path) -> Tuple[Path, str]:
|
||||||
if options.input_file == '-':
|
if options.input_file == '-':
|
||||||
# stdin
|
# stdin
|
||||||
log.info('reading file from standard input')
|
log.info('reading file from standard input')
|
||||||
target = os.path.join(work_folder, 'stdin')
|
target = work_folder / 'stdin'
|
||||||
with open(target, 'wb') as stream_buffer:
|
with open(target, 'wb') as stream_buffer:
|
||||||
copyfileobj(sys.stdin.buffer, stream_buffer)
|
copyfileobj(sys.stdin.buffer, stream_buffer)
|
||||||
return target, "<stdin>"
|
return target, "stdin"
|
||||||
|
elif hasattr(options.input_file, 'readable'):
|
||||||
|
if not options.input_file.readable():
|
||||||
|
raise InputFileError("Input file stream is not readable")
|
||||||
|
log.info('reading file from input stream')
|
||||||
|
target = work_folder / 'stream'
|
||||||
|
with open(target, 'wb') as stream_buffer:
|
||||||
|
copyfileobj(options.input_file, stream_buffer)
|
||||||
|
return target, "stream"
|
||||||
else:
|
else:
|
||||||
try:
|
try:
|
||||||
target = os.path.join(work_folder, 'origin')
|
target = work_folder / 'origin'
|
||||||
safe_symlink(options.input_file, target)
|
safe_symlink(options.input_file, target)
|
||||||
return target, os.fspath(options.input_file)
|
return target, os.fspath(options.input_file)
|
||||||
except FileNotFoundError:
|
except FileNotFoundError:
|
||||||
@@ -398,6 +363,9 @@ def check_requested_output_file(options):
|
|||||||
"is connected to a terminal. Please redirect stdout to a "
|
"is connected to a terminal. Please redirect stdout to a "
|
||||||
"file."
|
"file."
|
||||||
)
|
)
|
||||||
|
elif hasattr(options.output_file, 'writable'):
|
||||||
|
if not options.output_file.writable():
|
||||||
|
raise OutputFileAccessError("Output stream is not writable")
|
||||||
elif not is_file_writable(options.output_file):
|
elif not is_file_writable(options.output_file):
|
||||||
raise OutputFileAccessError(
|
raise OutputFileAccessError(
|
||||||
f"Output file location ({options.output_file}) is not a writable file."
|
f"Output file location ({options.output_file}) is not a writable file."
|
||||||
@@ -410,8 +378,15 @@ def report_output_file_size(options, input_file, output_file):
|
|||||||
input_size = Path(input_file).stat().st_size
|
input_size = Path(input_file).stat().st_size
|
||||||
except FileNotFoundError:
|
except FileNotFoundError:
|
||||||
return # Outputting to stream or something
|
return # Outputting to stream or something
|
||||||
|
with pikepdf.open(output_file) as p:
|
||||||
|
# Overhead constants obtained by estimating amount of data added by OCR
|
||||||
|
# PDF/A conversion, and possible XMP metadata addition, with compression
|
||||||
|
FILE_OVERHEAD = 4000
|
||||||
|
OCR_PER_PAGE_OVERHEAD = 3000
|
||||||
|
reasonable_overhead = FILE_OVERHEAD + OCR_PER_PAGE_OVERHEAD * len(p.pages)
|
||||||
ratio = output_size / input_size
|
ratio = output_size / input_size
|
||||||
if ratio < 1.35 or input_size < 25000:
|
reasonable_ratio = output_size / (input_size + reasonable_overhead)
|
||||||
|
if reasonable_ratio < 1.35 or input_size < 25000:
|
||||||
return # Seems fine
|
return # Seems fine
|
||||||
|
|
||||||
reasons = []
|
reasons = []
|
||||||
@@ -451,31 +426,3 @@ def report_output_file_size(options, input_file, output_file):
|
|||||||
f"The output file size is {ratio:.2f}× larger than the input file.\n"
|
f"The output file size is {ratio:.2f}× larger than the input file.\n"
|
||||||
f"{explanation}"
|
f"{explanation}"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def check_dependency_versions(options):
|
|
||||||
check_external_program(
|
|
||||||
program='tesseract',
|
|
||||||
package={'linux': 'tesseract-ocr'},
|
|
||||||
version_checker=tesseract.version,
|
|
||||||
need_version='4.0.0', # using backport for Travis CI
|
|
||||||
)
|
|
||||||
check_external_program(
|
|
||||||
program='gs',
|
|
||||||
package='ghostscript',
|
|
||||||
version_checker=ghostscript.version,
|
|
||||||
need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports
|
|
||||||
)
|
|
||||||
gs_version = ghostscript.version()
|
|
||||||
if gs_version in ('9.24', '9.51'):
|
|
||||||
raise MissingDependencyError(
|
|
||||||
f"Ghostscript {gs_version} contains serious regressions and is not "
|
|
||||||
"supported. Please upgrade to a newer version, or downgrade to the "
|
|
||||||
"previous version."
|
|
||||||
)
|
|
||||||
check_external_program(
|
|
||||||
program='qpdf',
|
|
||||||
package='qpdf',
|
|
||||||
version_checker=qpdf.version,
|
|
||||||
need_version='8.0.2',
|
|
||||||
)
|
|
||||||
|
|||||||
+84
-50
@@ -18,44 +18,24 @@
|
|||||||
import logging
|
import logging
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
from contextlib import suppress
|
from argparse import ArgumentParser
|
||||||
from enum import IntEnum
|
from enum import IntEnum
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Dict, Iterable
|
from typing import BinaryIO, Iterable, Union
|
||||||
|
|
||||||
from tqdm import tqdm
|
from ocrmypdf._logging import PageNumberFilter, TqdmConsole
|
||||||
|
from ocrmypdf._plugin_manager import get_plugin_manager
|
||||||
|
from ocrmypdf._sync import run_pipeline
|
||||||
|
from ocrmypdf._validation import check_options
|
||||||
|
from ocrmypdf.cli import get_parser
|
||||||
|
|
||||||
from ._sync import run_pipeline
|
try:
|
||||||
from ._validation import check_options
|
import coloredlogs
|
||||||
from .cli import parser
|
except ModuleNotFoundError:
|
||||||
|
coloredlogs = None
|
||||||
|
|
||||||
|
|
||||||
class TqdmConsole:
|
PathOrIO = Union[BinaryIO, os.PathLike, str, bytes]
|
||||||
"""Wrapper to log messages in a way that is compatible with tqdm progress bar
|
|
||||||
|
|
||||||
This routes log messages through tqdm so that it can print them above the
|
|
||||||
progress bar, and then refresh the progress bar, rather than overwriting
|
|
||||||
it which looks messy.
|
|
||||||
|
|
||||||
For some reason Python 3.6 prints extra empty messages from time to time,
|
|
||||||
so we suppress those.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self, file):
|
|
||||||
self.file = file
|
|
||||||
self.py36 = sys.version_info[0:2] == (3, 6)
|
|
||||||
|
|
||||||
def write(self, msg):
|
|
||||||
# When no progress bar is active, tqdm.write() routes to print()
|
|
||||||
if self.py36:
|
|
||||||
if msg.strip() != '':
|
|
||||||
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
|
||||||
else:
|
|
||||||
tqdm.write(msg.rstrip(), end='\n', file=self.file)
|
|
||||||
|
|
||||||
def flush(self):
|
|
||||||
with suppress(AttributeError):
|
|
||||||
self.file.flush()
|
|
||||||
|
|
||||||
|
|
||||||
class Verbosity(IntEnum):
|
class Verbosity(IntEnum):
|
||||||
@@ -98,6 +78,7 @@ def configure_logging(
|
|||||||
"""
|
"""
|
||||||
|
|
||||||
prefix = '' if manage_root_logger else 'ocrmypdf'
|
prefix = '' if manage_root_logger else 'ocrmypdf'
|
||||||
|
|
||||||
log = logging.getLogger(prefix)
|
log = logging.getLogger(prefix)
|
||||||
log.setLevel(logging.DEBUG)
|
log.setLevel(logging.DEBUG)
|
||||||
|
|
||||||
@@ -113,9 +94,25 @@ def configure_logging(
|
|||||||
else:
|
else:
|
||||||
console.setLevel(logging.INFO)
|
console.setLevel(logging.INFO)
|
||||||
|
|
||||||
formatter = logging.Formatter('%(levelname)7s - %(message)s')
|
console.addFilter(PageNumberFilter())
|
||||||
|
|
||||||
if verbosity >= 2:
|
if verbosity >= 2:
|
||||||
formatter = logging.Formatter('%(name)s - %(levelname)7s - %(message)s')
|
fmt = '%(levelname)7s %(name)s -%(pageno)s %(message)s'
|
||||||
|
else:
|
||||||
|
fmt = '%(pageno)s%(message)s'
|
||||||
|
|
||||||
|
use_colors = progress_bar_friendly
|
||||||
|
if not coloredlogs:
|
||||||
|
use_colors = False
|
||||||
|
if use_colors:
|
||||||
|
if os.name == 'nt':
|
||||||
|
use_colors = coloredlogs.enable_ansi_support()
|
||||||
|
if use_colors:
|
||||||
|
use_colors = coloredlogs.terminal_supports_colors()
|
||||||
|
if use_colors:
|
||||||
|
formatter = coloredlogs.ColoredFormatter(fmt=fmt)
|
||||||
|
else:
|
||||||
|
formatter = logging.Formatter(fmt=fmt)
|
||||||
|
|
||||||
console.setFormatter(formatter)
|
console.setFormatter(formatter)
|
||||||
log.addHandler(console)
|
log.addHandler(console)
|
||||||
@@ -132,7 +129,9 @@ def configure_logging(
|
|||||||
return log
|
return log
|
||||||
|
|
||||||
|
|
||||||
def create_options(*, input_file: os.PathLike, output_file: os.PathLike, **kwargs):
|
def create_options(
|
||||||
|
*, input_file: PathOrIO, output_file: PathOrIO, parser: ArgumentParser, **kwargs
|
||||||
|
):
|
||||||
cmdline = []
|
cmdline = []
|
||||||
deferred = []
|
deferred = []
|
||||||
|
|
||||||
@@ -142,7 +141,7 @@ def create_options(*, input_file: os.PathLike, output_file: os.PathLike, **kwarg
|
|||||||
|
|
||||||
# These arguments with special handling for which we bypass
|
# These arguments with special handling for which we bypass
|
||||||
# argparse
|
# argparse
|
||||||
if arg in {'tesseract_env', 'progress_bar'}:
|
if arg in {'progress_bar', 'plugins'}:
|
||||||
deferred.append((arg, val))
|
deferred.append((arg, val))
|
||||||
continue
|
continue
|
||||||
|
|
||||||
@@ -171,24 +170,31 @@ def create_options(*, input_file: os.PathLike, output_file: os.PathLike, **kwarg
|
|||||||
else:
|
else:
|
||||||
raise TypeError(f"{arg}: {val} ({type(val)})")
|
raise TypeError(f"{arg}: {val} ({type(val)})")
|
||||||
|
|
||||||
cmdline.append(str(input_file))
|
try:
|
||||||
cmdline.append(str(output_file))
|
cmdline.append(os.fspath(input_file))
|
||||||
|
except TypeError:
|
||||||
|
cmdline.append('stream://input_file')
|
||||||
|
try:
|
||||||
|
cmdline.append(os.fspath(output_file))
|
||||||
|
except TypeError:
|
||||||
|
cmdline.append('stream://output_file')
|
||||||
|
|
||||||
parser.api_mode = True
|
parser._api_mode = True
|
||||||
options = parser.parse_args(cmdline)
|
options = parser.parse_args(cmdline)
|
||||||
for keyword, val in deferred:
|
for keyword, val in deferred:
|
||||||
setattr(options, keyword, val)
|
setattr(options, keyword, val)
|
||||||
|
|
||||||
# If we are running a Tesseract spoof, ensure it knows what the input file is
|
if options.input_file == 'stream://input_file':
|
||||||
if os.environ.get('PYTEST_CURRENT_TEST') and options.tesseract_env:
|
options.input_file = input_file
|
||||||
options.tesseract_env['_OCRMYPDF_TEST_INFILE'] = os.fspath(input_file)
|
if options.output_file == 'stream://output_file':
|
||||||
|
options.output_file = output_file
|
||||||
|
|
||||||
return options
|
return options
|
||||||
|
|
||||||
|
|
||||||
def ocr( # pylint: disable=unused-argument
|
def ocr( # pylint: disable=unused-argument
|
||||||
input_file: os.PathLike,
|
input_file: PathOrIO,
|
||||||
output_file: os.PathLike,
|
output_file: PathOrIO,
|
||||||
*,
|
*,
|
||||||
language: Iterable[str] = None,
|
language: Iterable[str] = None,
|
||||||
image_dpi: int = None,
|
image_dpi: int = None,
|
||||||
@@ -230,9 +236,10 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
user_words: os.PathLike = None,
|
user_words: os.PathLike = None,
|
||||||
user_patterns: os.PathLike = None,
|
user_patterns: os.PathLike = None,
|
||||||
fast_web_view: float = None,
|
fast_web_view: float = None,
|
||||||
|
plugins: Iterable[str] = None,
|
||||||
keep_temporary_files: bool = None,
|
keep_temporary_files: bool = None,
|
||||||
progress_bar: bool = None,
|
progress_bar: bool = None,
|
||||||
tesseract_env: Dict[str, str] = None,
|
**kwargs,
|
||||||
):
|
):
|
||||||
"""Run OCRmyPDF on one PDF or image.
|
"""Run OCRmyPDF on one PDF or image.
|
||||||
|
|
||||||
@@ -240,10 +247,24 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
A few specific arguments are discussed here:
|
A few specific arguments are discussed here:
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
use_threads (bool): Use worker threads instead of processes. This reduces
|
use_threads: Use worker threads instead of processes. This reduces
|
||||||
performance but may make debugging easier since it is easier to set
|
performance but may make debugging easier since it is easier to set
|
||||||
breakpoints.
|
breakpoints.
|
||||||
tesseract_env (dict): Override environment variables for Tesseract
|
input_file: If a :class:`pathlib.Path`, ``str`` or ``bytes``, this is
|
||||||
|
interpreted as file system path to the input file. If the object
|
||||||
|
appears to be a readable stream (with methods such as ``.read()``
|
||||||
|
and ``.seek()``), the object will be read in its entirety and saved to
|
||||||
|
a temporary file. If ``input_file`` is ``"-"``, standard input will be
|
||||||
|
read.
|
||||||
|
output_file: If a :class:`pathlib.Path`, ``str`` or ``bytes``, this is
|
||||||
|
interpreted as file system path to the output file. If the object
|
||||||
|
appears to be a writable stream (with methods such as ``.read()`` and
|
||||||
|
``.seek()``), the output will be written to this stream. If
|
||||||
|
``output_file`` is ``"-"``, the output will be written to ``sys.stdout``
|
||||||
|
(provided that standard output does not seem to be a terminal device).
|
||||||
|
When a stream is used as output, whether via a writable object or
|
||||||
|
``"-"``, some final validation steps are not performed (we do not read
|
||||||
|
back the stream after it is written).
|
||||||
Raises:
|
Raises:
|
||||||
ocrmypdf.PdfMergeFailedError: If the input PDF is malformed, preventing merging
|
ocrmypdf.PdfMergeFailedError: If the input PDF is malformed, preventing merging
|
||||||
with the OCR layer.
|
with the OCR layer.
|
||||||
@@ -267,7 +288,20 @@ def ocr( # pylint: disable=unused-argument
|
|||||||
Returns:
|
Returns:
|
||||||
:class:`ocrmypdf.ExitCode`
|
:class:`ocrmypdf.ExitCode`
|
||||||
"""
|
"""
|
||||||
|
if not plugins:
|
||||||
|
plugins = []
|
||||||
|
else:
|
||||||
|
plugins = list(plugins)
|
||||||
|
|
||||||
options = create_options(**locals())
|
parser = get_parser()
|
||||||
check_options(options)
|
_plugin_manager = get_plugin_manager(plugins)
|
||||||
return run_pipeline(options, api=True)
|
_plugin_manager.hook.add_options(parser=parser) # pylint: disable=no-member
|
||||||
|
|
||||||
|
create_options_kwargs = {
|
||||||
|
k: v for k, v in locals().items() if not k.startswith('_') and k != 'kwargs'
|
||||||
|
}
|
||||||
|
create_options_kwargs.update(kwargs)
|
||||||
|
|
||||||
|
options = create_options(**create_options_kwargs)
|
||||||
|
check_options(options, _plugin_manager)
|
||||||
|
return run_pipeline(options=options, plugin_manager=_plugin_manager, api=True)
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This file is part of OCRmyPDF.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This file is part of OCRmyPDF.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
import logging
|
||||||
|
|
||||||
|
from ocrmypdf import hookimpl
|
||||||
|
from ocrmypdf._exec import ghostscript
|
||||||
|
from ocrmypdf._validation import HOCR_OK_LANGS
|
||||||
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def check_options(options):
|
||||||
|
gs_version = ghostscript.version()
|
||||||
|
check_external_program(
|
||||||
|
program='gs',
|
||||||
|
package='ghostscript',
|
||||||
|
version_checker=gs_version,
|
||||||
|
need_version='9.15', # limited by Travis CI / Ubuntu 14.04 backports
|
||||||
|
)
|
||||||
|
if gs_version in ('9.24', '9.51'):
|
||||||
|
raise MissingDependencyError(
|
||||||
|
f"Ghostscript {gs_version} contains serious regressions and is not "
|
||||||
|
"supported. Please upgrade to a newer version, or downgrade to the "
|
||||||
|
"previous version."
|
||||||
|
)
|
||||||
|
|
||||||
|
# We have these constraints to check for.
|
||||||
|
# 1. Ghostscript < 9.20 mangles multibyte Unicode
|
||||||
|
# 2. hocr doesn't work on non-Latin languages (so don't select it)
|
||||||
|
is_latin = options.languages.issubset(HOCR_OK_LANGS)
|
||||||
|
if gs_version < '9.20' and options.output_type != 'pdf' and not is_latin:
|
||||||
|
# https://bugs.ghostscript.com/show_bug.cgi?id=696874
|
||||||
|
# Ghostscript < 9.20 fails to encode multibyte characters properly
|
||||||
|
msg = (
|
||||||
|
"The installed version of Ghostscript does not work correctly "
|
||||||
|
"with the OCR languages you specified. Use --output-type pdf or "
|
||||||
|
"upgrade to Ghostscript 9.20 or later to avoid this issue."
|
||||||
|
)
|
||||||
|
msg += f"Found Ghostscript {gs_version}"
|
||||||
|
log.warning(msg)
|
||||||
|
|
||||||
|
if options.output_type == 'pdfa':
|
||||||
|
options.output_type = 'pdfa-2'
|
||||||
|
|
||||||
|
if options.output_type == 'pdfa-3' and ghostscript.version() < '9.19':
|
||||||
|
raise MissingDependencyError(
|
||||||
|
"--output-type pdfa-3 requires Ghostscript 9.19 or later"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def rasterize_pdf_page(
|
||||||
|
input_file,
|
||||||
|
output_file,
|
||||||
|
raster_device,
|
||||||
|
raster_dpi,
|
||||||
|
pageno,
|
||||||
|
page_dpi=None,
|
||||||
|
rotation=None,
|
||||||
|
filter_vector=False,
|
||||||
|
):
|
||||||
|
ghostscript.rasterize_pdf(
|
||||||
|
input_file,
|
||||||
|
output_file,
|
||||||
|
raster_device=raster_device,
|
||||||
|
raster_dpi=raster_dpi,
|
||||||
|
pageno=pageno,
|
||||||
|
page_dpi=page_dpi,
|
||||||
|
rotation=rotation,
|
||||||
|
filter_vector=filter_vector,
|
||||||
|
)
|
||||||
|
return output_file
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def generate_pdfa(pdf_pages, pdfmark, output_file, compression, pdf_version, pdfa_part):
|
||||||
|
ghostscript.generate_pdfa(
|
||||||
|
pdf_pages=[*pdf_pages, pdfmark],
|
||||||
|
output_file=output_file,
|
||||||
|
compression=compression,
|
||||||
|
pdf_version=pdf_version,
|
||||||
|
pdfa_part=pdfa_part,
|
||||||
|
)
|
||||||
|
return output_file
|
||||||
@@ -0,0 +1,197 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This file is part of OCRmyPDF.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
|
||||||
|
from ocrmypdf import hookimpl
|
||||||
|
from ocrmypdf._exec import tesseract
|
||||||
|
from ocrmypdf.cli import numeric
|
||||||
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
|
from ocrmypdf.helpers import clamp
|
||||||
|
from ocrmypdf.pluginspec import OcrEngine
|
||||||
|
from ocrmypdf.subprocess import check_external_program
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def add_options(parser):
|
||||||
|
tess = parser.add_argument_group("Tesseract", "Advanced control of Tesseract OCR")
|
||||||
|
tess.add_argument(
|
||||||
|
'--tesseract-config',
|
||||||
|
action='append',
|
||||||
|
metavar='CFG',
|
||||||
|
default=[],
|
||||||
|
help="Additional Tesseract configuration files -- see documentation",
|
||||||
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--tesseract-pagesegmode',
|
||||||
|
action='store',
|
||||||
|
type=int,
|
||||||
|
metavar='PSM',
|
||||||
|
choices=range(0, 14),
|
||||||
|
help="Set Tesseract page segmentation mode (see tesseract --help)",
|
||||||
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--tesseract-oem',
|
||||||
|
action='store',
|
||||||
|
type=int,
|
||||||
|
metavar='MODE',
|
||||||
|
choices=range(0, 4),
|
||||||
|
help=(
|
||||||
|
"Set Tesseract 4.0 OCR engine mode: "
|
||||||
|
"0 - original Tesseract only; "
|
||||||
|
"1 - neural nets LSTM only; "
|
||||||
|
"2 - Tesseract + LSTM; "
|
||||||
|
"3 - default."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--tesseract-timeout',
|
||||||
|
default=180.0,
|
||||||
|
type=numeric(float, 0),
|
||||||
|
metavar='SECONDS',
|
||||||
|
help='Give up on OCR after the timeout, but copy the preprocessed page '
|
||||||
|
'into the final output',
|
||||||
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--user-words',
|
||||||
|
metavar='FILE',
|
||||||
|
help="Specify the location of the Tesseract user words file. This is a "
|
||||||
|
"list of words Tesseract should consider while performing OCR in "
|
||||||
|
"addition to its standard language dictionaries. This can improve "
|
||||||
|
"OCR quality especially for specialized and technical documents.",
|
||||||
|
)
|
||||||
|
tess.add_argument(
|
||||||
|
'--user-patterns',
|
||||||
|
metavar='FILE',
|
||||||
|
help="Specify the location of the Tesseract user patterns file.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def check_options(options):
|
||||||
|
check_external_program(
|
||||||
|
program='tesseract',
|
||||||
|
package={'linux': 'tesseract-ocr'},
|
||||||
|
version_checker=tesseract.version,
|
||||||
|
need_version='4.0.0', # using backport for Travis CI
|
||||||
|
)
|
||||||
|
|
||||||
|
# Decide on what renderer to use
|
||||||
|
if options.pdf_renderer == 'auto':
|
||||||
|
options.pdf_renderer = 'sandwich'
|
||||||
|
|
||||||
|
if options.pdf_renderer == 'sandwich' and not tesseract.has_textonly_pdf(
|
||||||
|
set(options.languages)
|
||||||
|
):
|
||||||
|
raise MissingDependencyError(
|
||||||
|
"You are using an alpha version of Tesseract 4.0 that does not support "
|
||||||
|
"the textonly_pdf parameter. We don't support versions this old."
|
||||||
|
)
|
||||||
|
if not tesseract.has_user_words() and (options.user_words or options.user_patterns):
|
||||||
|
log.warning(
|
||||||
|
"Tesseract 4.0 ignores --user-words and --user-patterns, so these "
|
||||||
|
"arguments have no effect."
|
||||||
|
)
|
||||||
|
if options.tesseract_pagesegmode in (0, 2):
|
||||||
|
log.warning(
|
||||||
|
"The --tesseract-pagesegmode argument you select will disable OCR. "
|
||||||
|
"This may cause processing to fail."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def validate(pdfinfo, options):
|
||||||
|
# Tesseract 4.x can be multithreaded, and we also run multiple workers. We want
|
||||||
|
# to manage how many threads it uses to avoid creating total threads than cores.
|
||||||
|
# Performance testing shows we're better off
|
||||||
|
# parallelizing ocrmypdf and forcing Tesseract to be single threaded, which we
|
||||||
|
# get by setting the envvar OMP_THREAD_LIMIT to 1. But if the page count of the
|
||||||
|
# input file is small, then we allow Tesseract to use threads, subject to the
|
||||||
|
# constraint: (ocrmypdf workers) * (tesseract threads) <= max_workers.
|
||||||
|
# As of Tesseract 4.1, 3 threads is the most effective on a 4 core/8 thread system.
|
||||||
|
if not os.environ.get('OMP_THREAD_LIMIT', '').isnumeric():
|
||||||
|
tess_threads = clamp(options.jobs // len(pdfinfo), 1, 3)
|
||||||
|
os.environ['OMP_THREAD_LIMIT'] = str(tess_threads)
|
||||||
|
else:
|
||||||
|
tess_threads = int(os.environ['OMP_THREAD_LIMIT'])
|
||||||
|
|
||||||
|
if tess_threads > 1:
|
||||||
|
log.info("Using Tesseract OpenMP thread limit %d", tess_threads)
|
||||||
|
|
||||||
|
|
||||||
|
class TesseractOcrEngine(OcrEngine):
|
||||||
|
@staticmethod
|
||||||
|
def version():
|
||||||
|
return tesseract.version()
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def creator_tag(options):
|
||||||
|
tag = '-PDF' if options.pdf_renderer == 'sandwich' else ''
|
||||||
|
return f"Tesseract OCR{tag} {TesseractOcrEngine.version()}"
|
||||||
|
|
||||||
|
def __str__(self):
|
||||||
|
return f"Tesseract OCR {TesseractOcrEngine.version()}"
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def languages(options):
|
||||||
|
return tesseract.get_languages()
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def get_orientation(input_file, options):
|
||||||
|
return tesseract.get_orientation(
|
||||||
|
input_file,
|
||||||
|
engine_mode=options.tesseract_oem,
|
||||||
|
timeout=options.tesseract_timeout,
|
||||||
|
)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def generate_hocr(input_file, output_hocr, output_text, options):
|
||||||
|
tesseract.generate_hocr(
|
||||||
|
input_file=input_file,
|
||||||
|
output_hocr=output_hocr,
|
||||||
|
output_text=output_text,
|
||||||
|
languages=options.languages,
|
||||||
|
engine_mode=options.tesseract_oem,
|
||||||
|
tessconfig=options.tesseract_config,
|
||||||
|
timeout=options.tesseract_timeout,
|
||||||
|
pagesegmode=options.tesseract_pagesegmode,
|
||||||
|
user_words=options.user_words,
|
||||||
|
user_patterns=options.user_patterns,
|
||||||
|
)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def generate_pdf(input_file, output_pdf, output_text, options):
|
||||||
|
tesseract.generate_pdf(
|
||||||
|
input_file=input_file,
|
||||||
|
output_pdf=output_pdf,
|
||||||
|
output_text=output_text,
|
||||||
|
languages=options.languages,
|
||||||
|
engine_mode=options.tesseract_oem,
|
||||||
|
tessconfig=options.tesseract_config,
|
||||||
|
timeout=options.tesseract_timeout,
|
||||||
|
pagesegmode=options.tesseract_pagesegmode,
|
||||||
|
user_words=options.user_words,
|
||||||
|
user_patterns=options.user_patterns,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@hookimpl
|
||||||
|
def get_ocr_engine():
|
||||||
|
return TesseractOcrEngine()
|
||||||
+386
-390
@@ -17,10 +17,8 @@
|
|||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
|
|
||||||
from ._version import PROGRAM_NAME as _PROGRAM_NAME
|
from ocrmypdf._version import PROGRAM_NAME as _PROGRAM_NAME
|
||||||
from ._version import __version__ as _VERSION
|
from ocrmypdf._version import __version__ as _VERSION
|
||||||
|
|
||||||
__all__ = ['parser']
|
|
||||||
|
|
||||||
|
|
||||||
def numeric(basetype, min_=None, max_=None):
|
def numeric(basetype, min_=None, max_=None):
|
||||||
@@ -47,27 +45,43 @@ class ArgumentParser(argparse.ArgumentParser):
|
|||||||
|
|
||||||
def __init__(self, *args, **kwargs):
|
def __init__(self, *args, **kwargs):
|
||||||
super().__init__(*args, **kwargs)
|
super().__init__(*args, **kwargs)
|
||||||
self.api_mode = False
|
self._api_mode = False
|
||||||
|
|
||||||
def error(self, message):
|
def error(self, message):
|
||||||
if not self.api_mode:
|
if not self._api_mode:
|
||||||
super().error(message)
|
super().error(message)
|
||||||
return
|
return
|
||||||
raise ValueError(message)
|
raise ValueError(message)
|
||||||
|
|
||||||
|
|
||||||
parser = ArgumentParser(
|
class LanguageSetAction(argparse.Action):
|
||||||
prog=_PROGRAM_NAME,
|
def __init__(self, option_strings, dest, default=None, **kwargs):
|
||||||
fromfile_prefix_chars='@',
|
if default is None:
|
||||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
default = set()
|
||||||
description="""\
|
super().__init__(option_strings, dest, default=default, **kwargs)
|
||||||
|
|
||||||
|
def __call__(self, parser, namespace, values, option_string=None):
|
||||||
|
dest = getattr(namespace, self.dest)
|
||||||
|
if '+' in values:
|
||||||
|
dest.update(lang for lang in values.split('+'))
|
||||||
|
else:
|
||||||
|
dest.add(values)
|
||||||
|
|
||||||
|
|
||||||
|
def get_parser():
|
||||||
|
parser = ArgumentParser(
|
||||||
|
prog=_PROGRAM_NAME,
|
||||||
|
allow_abbrev=True,
|
||||||
|
fromfile_prefix_chars='@',
|
||||||
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||||
|
description="""\
|
||||||
Generates a searchable PDF or PDF/A from a regular PDF.
|
Generates a searchable PDF or PDF/A from a regular PDF.
|
||||||
|
|
||||||
OCRmyPDF rasterizes each page of the input PDF, optionally corrects page
|
OCRmyPDF rasterizes each page of the input PDF, optionally corrects page
|
||||||
rotation and performs image processing, runs the Tesseract OCR engine on the
|
rotation and performs image processing, runs the Tesseract OCR engine on the
|
||||||
image, and then creates a PDF from the OCR information.
|
image, and then creates a PDF from the OCR information.
|
||||||
""",
|
""",
|
||||||
epilog="""\
|
epilog="""\
|
||||||
OCRmyPDF attempts to keep the output file at about the same size. If a file
|
OCRmyPDF attempts to keep the output file at about the same size. If a file
|
||||||
contains losslessly compressed images, and output file will be losslessly
|
contains losslessly compressed images, and output file will be losslessly
|
||||||
compressed as well.
|
compressed as well.
|
||||||
@@ -108,386 +122,368 @@ Online documentation is located at:
|
|||||||
https://ocrmypdf.readthedocs.io/en/latest/introduction.html
|
https://ocrmypdf.readthedocs.io/en/latest/introduction.html
|
||||||
|
|
||||||
""",
|
""",
|
||||||
)
|
)
|
||||||
|
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
'input_file',
|
'input_file',
|
||||||
metavar="input_pdf_or_image",
|
metavar="input_pdf_or_image",
|
||||||
help="PDF file containing the images to be OCRed (or '-' to read from "
|
help="PDF file containing the images to be OCRed (or '-' to read from "
|
||||||
"standard input)",
|
"standard input)",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
'output_file',
|
||||||
|
metavar="output_pdf",
|
||||||
|
help="Output searchable PDF file (or '-' to write to standard output). "
|
||||||
|
"Existing files will be ovewritten. If same as input file, the "
|
||||||
|
"input file will be updated only if processing is successful.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
'-l',
|
||||||
|
'--language',
|
||||||
|
dest='languages',
|
||||||
|
action=LanguageSetAction,
|
||||||
|
help="Language(s) of the file to be OCRed (see tesseract --list-langs for "
|
||||||
|
"all language packs installed in your system). Use -l eng+deu for "
|
||||||
|
"multiple languages.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
'--image-dpi',
|
||||||
|
metavar='DPI',
|
||||||
|
type=int,
|
||||||
|
help="For input image instead of PDF, use this DPI instead of file's.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
'--output-type',
|
||||||
|
choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3'],
|
||||||
|
default='pdfa',
|
||||||
|
help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for "
|
||||||
|
"long term archiving (default, recommended) but may not suitable "
|
||||||
|
"for users who want their file altered as little as possible. 'pdfa' "
|
||||||
|
"also has problems with full Unicode text. 'pdf' attempts to "
|
||||||
|
"preserve file contents as much as possible. 'pdf-a1' creates a "
|
||||||
|
"PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a "
|
||||||
|
"PDF/A3-b file.",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Use null string '\0' as sentinel to indicate the user supplied no argument,
|
||||||
|
# since that is the only invalid character for filepaths on all platforms
|
||||||
|
# bool('\0') is True in Python
|
||||||
|
parser.add_argument(
|
||||||
|
'--sidecar',
|
||||||
|
nargs='?',
|
||||||
|
const='\0',
|
||||||
|
default=None,
|
||||||
|
metavar='FILE',
|
||||||
|
help="Generate sidecar text files that contain the same text recognized "
|
||||||
|
"by Tesseract. This may be useful for building a OCR text database. "
|
||||||
|
"If FILE is omitted, the sidecar file be named {output_file}.txt "
|
||||||
|
"If FILE is set to '-', the sidecar is written to stdout (a "
|
||||||
|
"convenient way to preview OCR quality). The output file and sidecar "
|
||||||
|
"may not both use stdout at the same time.",
|
||||||
|
)
|
||||||
|
|
||||||
|
parser.add_argument(
|
||||||
|
'--version',
|
||||||
|
action='version',
|
||||||
|
version=_VERSION,
|
||||||
|
help="Print program version and exit",
|
||||||
|
)
|
||||||
|
|
||||||
|
jobcontrol = parser.add_argument_group("Job control options")
|
||||||
|
jobcontrol.add_argument(
|
||||||
|
'-j',
|
||||||
|
'--jobs',
|
||||||
|
metavar='N',
|
||||||
|
type=numeric(int, 0, 256),
|
||||||
|
help="Use up to N CPU cores simultaneously (default: use all).",
|
||||||
|
)
|
||||||
|
jobcontrol.add_argument(
|
||||||
|
'-q', '--quiet', action='store_true', help="Suppress INFO messages"
|
||||||
|
)
|
||||||
|
jobcontrol.add_argument(
|
||||||
|
'-v',
|
||||||
|
'--verbose',
|
||||||
|
type=numeric(int, 0, 2),
|
||||||
|
default=0,
|
||||||
|
const=1,
|
||||||
|
nargs='?',
|
||||||
|
help="Print more verbose messages for each additional verbose level. Use "
|
||||||
|
"`-v 1` typically for much more detailed logging. Higher numbers "
|
||||||
|
"are probably only useful in debugging.",
|
||||||
|
)
|
||||||
|
jobcontrol.add_argument(
|
||||||
|
'--no-progress-bar',
|
||||||
|
action='store_false',
|
||||||
|
dest='progress_bar',
|
||||||
|
help=argparse.SUPPRESS,
|
||||||
|
)
|
||||||
|
jobcontrol.add_argument(
|
||||||
|
'--use-threads', action='store_true', help=argparse.SUPPRESS
|
||||||
|
)
|
||||||
|
|
||||||
|
metadata = parser.add_argument_group(
|
||||||
|
"Metadata options",
|
||||||
|
"Set output PDF/A metadata (default: copy input document's metadata)",
|
||||||
|
)
|
||||||
|
metadata.add_argument(
|
||||||
|
'--title', type=str, help="Set document title (place multiple words in quotes)"
|
||||||
|
)
|
||||||
|
metadata.add_argument('--author', type=str, help="Set document author")
|
||||||
|
metadata.add_argument(
|
||||||
|
'--subject', type=str, help="Set document subject description"
|
||||||
|
)
|
||||||
|
metadata.add_argument('--keywords', type=str, help="Set document keywords")
|
||||||
|
|
||||||
|
preprocessing = parser.add_argument_group(
|
||||||
|
"Image preprocessing options",
|
||||||
|
"Options to improve the quality of the final PDF and OCR",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'-r',
|
||||||
|
'--rotate-pages',
|
||||||
|
action='store_true',
|
||||||
|
help="Automatically rotate pages based on detected text orientation",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'--remove-background',
|
||||||
|
action='store_true',
|
||||||
|
help="Attempt to remove background from gray or color pages, setting it "
|
||||||
|
"to white ",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'-d',
|
||||||
|
'--deskew',
|
||||||
|
action='store_true',
|
||||||
|
help="Deskew each page before performing OCR",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'-c',
|
||||||
|
'--clean',
|
||||||
|
action='store_true',
|
||||||
|
help="Clean pages from scanning artifacts before performing OCR, and send "
|
||||||
|
"the cleaned page to OCR, but do not include the cleaned page in "
|
||||||
|
"the output",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'-i',
|
||||||
|
'--clean-final',
|
||||||
|
action='store_true',
|
||||||
|
help="Clean page as above, and incorporate the cleaned image in the final "
|
||||||
|
"PDF. Might remove desired content.",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'--unpaper-args',
|
||||||
|
type=str,
|
||||||
|
default=None,
|
||||||
|
help="A quoted string of arguments to pass to unpaper. Requires --clean. "
|
||||||
|
"Example: --unpaper-args '--layout double'.",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'--oversample',
|
||||||
|
metavar='DPI',
|
||||||
|
type=numeric(int, 0, 5000),
|
||||||
|
default=0,
|
||||||
|
help="Oversample images to at least the specified DPI, to improve OCR "
|
||||||
|
"results slightly",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'--remove-vectors',
|
||||||
|
action='store_true',
|
||||||
|
help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they "
|
||||||
|
"will not be included in OCR. This can eliminate false characters.",
|
||||||
|
)
|
||||||
|
preprocessing.add_argument(
|
||||||
|
'--threshold',
|
||||||
|
action='store_true',
|
||||||
|
help=(
|
||||||
|
"EXPERIMENTAL. Threshold image to 1bpp before sending it to Tesseract "
|
||||||
|
"for OCR. Can improve OCR quality compared to Tesseract's thresholder."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
ocrsettings = parser.add_argument_group("OCR options", "Control how OCR is applied")
|
||||||
|
ocrsettings.add_argument(
|
||||||
|
'-f',
|
||||||
|
'--force-ocr',
|
||||||
|
action='store_true',
|
||||||
|
help="Rasterize any text or vector objects on each page, apply OCR, and "
|
||||||
|
"save the rastered output (this rewrites the PDF)",
|
||||||
|
)
|
||||||
|
ocrsettings.add_argument(
|
||||||
|
'-s',
|
||||||
|
'--skip-text',
|
||||||
|
action='store_true',
|
||||||
|
help="Skip OCR on any pages that already contain text, but include the "
|
||||||
|
"page in final output; useful for PDFs that contain a mix of "
|
||||||
|
"images, text pages, and/or previously OCRed pages",
|
||||||
|
)
|
||||||
|
ocrsettings.add_argument(
|
||||||
|
'--redo-ocr',
|
||||||
|
action='store_true',
|
||||||
|
help="Attempt to detect and remove the hidden OCR layer from files that "
|
||||||
|
"were previously OCRed with OCRmyPDF or another program. Apply OCR "
|
||||||
|
"to text found in raster images. Existing visible text objects will "
|
||||||
|
"not be changed. If there is no existing OCR, OCR will be added.",
|
||||||
|
)
|
||||||
|
ocrsettings.add_argument(
|
||||||
|
'--skip-big',
|
||||||
|
type=numeric(float, 0, 5000),
|
||||||
|
metavar='MPixels',
|
||||||
|
help="Skip OCR on pages larger than the specified amount of megapixels, "
|
||||||
|
"but include skipped pages in final output",
|
||||||
|
)
|
||||||
|
|
||||||
|
optimizing = parser.add_argument_group(
|
||||||
|
"Optimization options", "Control how the PDF is optimized after OCR"
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'-O',
|
||||||
|
'--optimize',
|
||||||
|
type=int,
|
||||||
|
choices=range(0, 4),
|
||||||
|
default=1,
|
||||||
|
help=(
|
||||||
|
"Control how PDF is optimized after processing:"
|
||||||
|
"0 - do not optimize; "
|
||||||
|
"1 - do safe, lossless optimizations (default); "
|
||||||
|
"2 - do some lossy optimizations; "
|
||||||
|
"3 - do aggressive lossy optimizations (including lossy JBIG2)"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jpeg-quality',
|
||||||
|
type=numeric(int, 0, 100),
|
||||||
|
default=0,
|
||||||
|
metavar='Q',
|
||||||
|
help=(
|
||||||
|
"Adjust JPEG quality level for JPEG optimization. "
|
||||||
|
"100 is best quality and largest output size; "
|
||||||
|
"1 is lowest quality and smallest output; "
|
||||||
|
"0 uses the default."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jpg-quality',
|
||||||
|
type=numeric(int, 0, 100),
|
||||||
|
default=0,
|
||||||
|
metavar='Q',
|
||||||
|
dest='jpeg_quality',
|
||||||
|
help=argparse.SUPPRESS, # Alias for --jpeg-quality
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--png-quality',
|
||||||
|
type=numeric(int, 0, 100),
|
||||||
|
default=0,
|
||||||
|
metavar='Q',
|
||||||
|
help=(
|
||||||
|
"Adjust PNG quality level to use when quantizing PNGs. "
|
||||||
|
"Values have same meaning as with --jpeg-quality"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jbig2-lossy',
|
||||||
|
action='store_true',
|
||||||
|
help=(
|
||||||
|
"Enable JBIG2 lossy mode (better compression, not suitable for some "
|
||||||
|
"use cases - see documentation)."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
optimizing.add_argument(
|
||||||
|
'--jbig2-page-group-size',
|
||||||
|
type=numeric(int, 1, 10000),
|
||||||
|
default=0,
|
||||||
|
metavar='N',
|
||||||
|
# Adjust number of pages to consider at once for JBIG2 compression
|
||||||
|
help=argparse.SUPPRESS,
|
||||||
|
)
|
||||||
|
|
||||||
|
advanced = parser.add_argument_group(
|
||||||
|
"Advanced", "Advanced options to control OCRmyPDF"
|
||||||
|
)
|
||||||
|
advanced.add_argument(
|
||||||
|
'--pages',
|
||||||
|
type=str,
|
||||||
|
help=(
|
||||||
|
"Limit OCR to the specified pages (ranges or comma separated), "
|
||||||
|
"skipping others"
|
||||||
|
),
|
||||||
|
)
|
||||||
|
advanced.add_argument(
|
||||||
|
'--max-image-mpixels',
|
||||||
|
action='store',
|
||||||
|
type=numeric(float, 0),
|
||||||
|
metavar='MPixels',
|
||||||
|
help="Set maximum number of pixels to unpack before treating an image as a "
|
||||||
|
"decompression bomb",
|
||||||
|
default=128.0,
|
||||||
|
)
|
||||||
|
advanced.add_argument(
|
||||||
|
'--pdf-renderer',
|
||||||
|
choices=['auto', 'hocr', 'sandwich'],
|
||||||
|
default='auto',
|
||||||
|
help="Choose OCR PDF renderer - the default option is to let OCRmyPDF "
|
||||||
|
"choose. See documentation for discussion.",
|
||||||
|
)
|
||||||
|
advanced.add_argument(
|
||||||
|
'--rotate-pages-threshold',
|
||||||
|
default=14.0,
|
||||||
|
type=numeric(float, 0, 1000),
|
||||||
|
metavar='CONFIDENCE',
|
||||||
|
help="Only rotate pages when confidence is above this value (arbitrary "
|
||||||
|
"units reported by tesseract)",
|
||||||
|
)
|
||||||
|
advanced.add_argument(
|
||||||
|
'--pdfa-image-compression',
|
||||||
|
choices=['auto', 'jpeg', 'lossless'],
|
||||||
|
default='auto',
|
||||||
|
help="Specify how to compress images in the output PDF/A. 'auto' lets "
|
||||||
|
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
|
||||||
|
"JPEG compression. 'lossless' uses PNG-style lossless compression "
|
||||||
|
"for all images. Monochrome images are always compressed using a "
|
||||||
|
"lossless codec. Compression settings "
|
||||||
|
"are applied to all pages, including those for which OCR was "
|
||||||
|
"skipped. Not supported for --output-type=pdf ; that setting "
|
||||||
|
"preserves the original compression of all images.",
|
||||||
|
)
|
||||||
|
advanced.add_argument(
|
||||||
|
'--fast-web-view',
|
||||||
|
type=numeric(float, 0),
|
||||||
|
default=1.0,
|
||||||
|
metavar="MEGABYTES",
|
||||||
|
help="If the size of file is more than this threshold (in MB), then "
|
||||||
|
"linearize the PDF for fast web viewing. This allows the PDF to be "
|
||||||
|
"displayed before it is fully downloaded in web browsers, but increases "
|
||||||
|
"the space required slightly. By default we skip this for small files "
|
||||||
|
"which do not benefit. If the threshold is 0 it will be apply to all files. "
|
||||||
|
"Set the threshold very high to disable.",
|
||||||
|
)
|
||||||
|
advanced.add_argument(
|
||||||
|
'--plugin',
|
||||||
|
dest='plugins',
|
||||||
|
action='append',
|
||||||
|
default=[],
|
||||||
|
help="Name of plugin to import.",
|
||||||
|
)
|
||||||
|
|
||||||
|
debugging = parser.add_argument_group(
|
||||||
|
"Debugging", "Arguments to help with troubleshooting and debugging"
|
||||||
|
)
|
||||||
|
debugging.add_argument(
|
||||||
|
'-k',
|
||||||
|
'--keep-temporary-files',
|
||||||
|
action='store_true',
|
||||||
|
help="Keep temporary files (helpful for debugging)",
|
||||||
|
)
|
||||||
|
return parser
|
||||||
|
|
||||||
|
|
||||||
|
plugins_only_parser = ArgumentParser(
|
||||||
|
prog=_PROGRAM_NAME, fromfile_prefix_chars='@', add_help=False, allow_abbrev=False
|
||||||
)
|
)
|
||||||
parser.add_argument(
|
plugins_only_parser.add_argument(
|
||||||
'output_file',
|
'--plugin',
|
||||||
metavar="output_pdf",
|
dest='plugins',
|
||||||
help="Output searchable PDF file (or '-' to write to standard output). "
|
|
||||||
"Existing files will be ovewritten. If same as input file, the "
|
|
||||||
"input file will be updated only if processing is successful.",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'-l',
|
|
||||||
'--language',
|
|
||||||
action='append',
|
action='append',
|
||||||
help="Language(s) of the file to be OCRed (see tesseract --list-langs for "
|
|
||||||
"all language packs installed in your system). Use -l eng+deu for "
|
|
||||||
"multiple languages.",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'--image-dpi',
|
|
||||||
metavar='DPI',
|
|
||||||
type=int,
|
|
||||||
help="For input image instead of PDF, use this DPI instead of file's.",
|
|
||||||
)
|
|
||||||
parser.add_argument(
|
|
||||||
'--output-type',
|
|
||||||
choices=['pdfa', 'pdf', 'pdfa-1', 'pdfa-2', 'pdfa-3'],
|
|
||||||
default='pdfa',
|
|
||||||
help="Choose output type. 'pdfa' creates a PDF/A-2b compliant file for "
|
|
||||||
"long term archiving (default, recommended) but may not suitable "
|
|
||||||
"for users who want their file altered as little as possible. 'pdfa' "
|
|
||||||
"also has problems with full Unicode text. 'pdf' attempts to "
|
|
||||||
"preserve file contents as much as possible. 'pdf-a1' creates a "
|
|
||||||
"PDF/A1-b file. 'pdf-a2' is equivalent to 'pdfa'. 'pdf-a3' creates a "
|
|
||||||
"PDF/A3-b file.",
|
|
||||||
)
|
|
||||||
|
|
||||||
# Use null string '\0' as sentinel to indicate the user supplied no argument,
|
|
||||||
# since that is the only invalid character for filepaths on all platforms
|
|
||||||
# bool('\0') is True in Python
|
|
||||||
parser.add_argument(
|
|
||||||
'--sidecar',
|
|
||||||
nargs='?',
|
|
||||||
const='\0',
|
|
||||||
default=None,
|
|
||||||
metavar='FILE',
|
|
||||||
help="Generate sidecar text files that contain the same text recognized "
|
|
||||||
"by Tesseract. This may be useful for building a OCR text database. "
|
|
||||||
"If FILE is omitted, the sidecar file be named {output_file}.txt "
|
|
||||||
"If FILE is set to '-', the sidecar is written to stdout (a "
|
|
||||||
"convenient way to preview OCR quality). The output file and sidecar "
|
|
||||||
"may not both use stdout at the same time.",
|
|
||||||
)
|
|
||||||
|
|
||||||
parser.add_argument(
|
|
||||||
'--version',
|
|
||||||
action='version',
|
|
||||||
version=_VERSION,
|
|
||||||
help="Print program version and exit",
|
|
||||||
)
|
|
||||||
|
|
||||||
jobcontrol = parser.add_argument_group("Job control options")
|
|
||||||
jobcontrol.add_argument(
|
|
||||||
'-j',
|
|
||||||
'--jobs',
|
|
||||||
metavar='N',
|
|
||||||
type=numeric(int, 0, 256),
|
|
||||||
help="Use up to N CPU cores simultaneously (default: use all).",
|
|
||||||
)
|
|
||||||
jobcontrol.add_argument(
|
|
||||||
'-q', '--quiet', action='store_true', help="Suppress INFO messages"
|
|
||||||
)
|
|
||||||
jobcontrol.add_argument(
|
|
||||||
'-v',
|
|
||||||
'--verbose',
|
|
||||||
type=numeric(int, 0, 2),
|
|
||||||
default=0,
|
|
||||||
const=1,
|
|
||||||
nargs='?',
|
|
||||||
help="Print more verbose messages for each additional verbose level. Use "
|
|
||||||
"`-v 1` typically for much more detailed logging. Higher numbers "
|
|
||||||
"are probably only useful in debugging.",
|
|
||||||
)
|
|
||||||
jobcontrol.add_argument(
|
|
||||||
'--no-progress-bar',
|
|
||||||
action='store_false',
|
|
||||||
dest='progress_bar',
|
|
||||||
help=argparse.SUPPRESS,
|
|
||||||
)
|
|
||||||
jobcontrol.add_argument('--use-threads', action='store_true', help=argparse.SUPPRESS)
|
|
||||||
|
|
||||||
metadata = parser.add_argument_group(
|
|
||||||
"Metadata options",
|
|
||||||
"Set output PDF/A metadata (default: copy input document's metadata)",
|
|
||||||
)
|
|
||||||
metadata.add_argument(
|
|
||||||
'--title', type=str, help="Set document title (place multiple words in quotes)"
|
|
||||||
)
|
|
||||||
metadata.add_argument('--author', type=str, help="Set document author")
|
|
||||||
metadata.add_argument('--subject', type=str, help="Set document subject description")
|
|
||||||
metadata.add_argument('--keywords', type=str, help="Set document keywords")
|
|
||||||
|
|
||||||
preprocessing = parser.add_argument_group(
|
|
||||||
"Image preprocessing options",
|
|
||||||
"Options to improve the quality of the final PDF and OCR",
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'-r',
|
|
||||||
'--rotate-pages',
|
|
||||||
action='store_true',
|
|
||||||
help="Automatically rotate pages based on detected text orientation",
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'--remove-background',
|
|
||||||
action='store_true',
|
|
||||||
help="Attempt to remove background from gray or color pages, setting it "
|
|
||||||
"to white ",
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'-d', '--deskew', action='store_true', help="Deskew each page before performing OCR"
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'-c',
|
|
||||||
'--clean',
|
|
||||||
action='store_true',
|
|
||||||
help="Clean pages from scanning artifacts before performing OCR, and send "
|
|
||||||
"the cleaned page to OCR, but do not include the cleaned page in "
|
|
||||||
"the output",
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'-i',
|
|
||||||
'--clean-final',
|
|
||||||
action='store_true',
|
|
||||||
help="Clean page as above, and incorporate the cleaned image in the final "
|
|
||||||
"PDF. Might remove desired content.",
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'--unpaper-args',
|
|
||||||
type=str,
|
|
||||||
default=None,
|
|
||||||
help="A quoted string of arguments to pass to unpaper. Requires --clean. "
|
|
||||||
"Example: --unpaper-args '--layout double'.",
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'--oversample',
|
|
||||||
metavar='DPI',
|
|
||||||
type=numeric(int, 0, 5000),
|
|
||||||
default=0,
|
|
||||||
help="Oversample images to at least the specified DPI, to improve OCR "
|
|
||||||
"results slightly",
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'--remove-vectors',
|
|
||||||
action='store_true',
|
|
||||||
help="EXPERIMENTAL. Mask out any vector objects in the PDF so that they "
|
|
||||||
"will not be included in OCR. This can eliminate false characters.",
|
|
||||||
)
|
|
||||||
preprocessing.add_argument(
|
|
||||||
'--threshold',
|
|
||||||
action='store_true',
|
|
||||||
help="EXPERIMENTAL. Threshold image to 1bpp before sending it to Tesseract for OCR. Can "
|
|
||||||
"improve OCR quality compared to Tesseract's thresholder.",
|
|
||||||
)
|
|
||||||
|
|
||||||
ocrsettings = parser.add_argument_group("OCR options", "Control how OCR is applied")
|
|
||||||
ocrsettings.add_argument(
|
|
||||||
'-f',
|
|
||||||
'--force-ocr',
|
|
||||||
action='store_true',
|
|
||||||
help="Rasterize any text or vector objects on each page, apply OCR, and "
|
|
||||||
"save the rastered output (this rewrites the PDF)",
|
|
||||||
)
|
|
||||||
ocrsettings.add_argument(
|
|
||||||
'-s',
|
|
||||||
'--skip-text',
|
|
||||||
action='store_true',
|
|
||||||
help="Skip OCR on any pages that already contain text, but include the "
|
|
||||||
"page in final output; useful for PDFs that contain a mix of "
|
|
||||||
"images, text pages, and/or previously OCRed pages",
|
|
||||||
)
|
|
||||||
ocrsettings.add_argument(
|
|
||||||
'--redo-ocr',
|
|
||||||
action='store_true',
|
|
||||||
help="Attempt to detect and remove the hidden OCR layer from files that "
|
|
||||||
"were previously OCRed with OCRmyPDF or another program. Apply OCR "
|
|
||||||
"to text found in raster images. Existing visible text objects will "
|
|
||||||
"not be changed. If there is no existing OCR, OCR will be added.",
|
|
||||||
)
|
|
||||||
ocrsettings.add_argument(
|
|
||||||
'--skip-big',
|
|
||||||
type=numeric(float, 0, 5000),
|
|
||||||
metavar='MPixels',
|
|
||||||
help="Skip OCR on pages larger than the specified amount of megapixels, "
|
|
||||||
"but include skipped pages in final output",
|
|
||||||
)
|
|
||||||
|
|
||||||
optimizing = parser.add_argument_group(
|
|
||||||
"Optimization options", "Control how the PDF is optimized after OCR"
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'-O',
|
|
||||||
'--optimize',
|
|
||||||
type=int,
|
|
||||||
choices=range(0, 4),
|
|
||||||
default=1,
|
|
||||||
help=(
|
|
||||||
"Control how PDF is optimized after processing:"
|
|
||||||
"0 - do not optimize; "
|
|
||||||
"1 - do safe, lossless optimizations (default); "
|
|
||||||
"2 - do some lossy optimizations; "
|
|
||||||
"3 - do aggressive lossy optimizations (including lossy JBIG2)"
|
|
||||||
),
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--jpeg-quality',
|
|
||||||
type=numeric(int, 0, 100),
|
|
||||||
default=0,
|
|
||||||
metavar='Q',
|
|
||||||
help=(
|
|
||||||
"Adjust JPEG quality level for JPEG optimization. "
|
|
||||||
"100 is best quality and largest output size; "
|
|
||||||
"1 is lowest quality and smallest output; "
|
|
||||||
"0 uses the default."
|
|
||||||
),
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--jpg-quality',
|
|
||||||
type=numeric(int, 0, 100),
|
|
||||||
default=0,
|
|
||||||
metavar='Q',
|
|
||||||
dest='jpeg_quality',
|
|
||||||
help=argparse.SUPPRESS, # Alias for --jpeg-quality
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--png-quality',
|
|
||||||
type=numeric(int, 0, 100),
|
|
||||||
default=0,
|
|
||||||
metavar='Q',
|
|
||||||
help=(
|
|
||||||
"Adjust PNG quality level to use when quantizing PNGs. "
|
|
||||||
"Values have same meaning as with --jpeg-quality"
|
|
||||||
),
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--jbig2-lossy',
|
|
||||||
action='store_true',
|
|
||||||
help=(
|
|
||||||
"Enable JBIG2 lossy mode (better compression, not suitable for some "
|
|
||||||
"use cases - see documentation)."
|
|
||||||
),
|
|
||||||
)
|
|
||||||
optimizing.add_argument(
|
|
||||||
'--jbig2-page-group-size',
|
|
||||||
type=numeric(int, 1, 10000),
|
|
||||||
default=0,
|
|
||||||
metavar='N',
|
|
||||||
# Adjust number of pages to consider at once for JBIG2 compression
|
|
||||||
help=argparse.SUPPRESS,
|
|
||||||
)
|
|
||||||
|
|
||||||
advanced = parser.add_argument_group(
|
|
||||||
"Advanced", "Advanced options to control Tesseract's OCR behavior"
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--pages',
|
|
||||||
type=str,
|
|
||||||
help="Limit OCR to the specified pages (ranges or comma separated), skipping others",
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--max-image-mpixels',
|
|
||||||
action='store',
|
|
||||||
type=numeric(float, 0),
|
|
||||||
metavar='MPixels',
|
|
||||||
help="Set maximum number of pixels to unpack before treating an image as a "
|
|
||||||
"decompression bomb",
|
|
||||||
default=128.0,
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--tesseract-config',
|
|
||||||
action='append',
|
|
||||||
metavar='CFG',
|
|
||||||
default=[],
|
default=[],
|
||||||
help="Additional Tesseract configuration files -- see documentation",
|
help="Name of plugin to import.",
|
||||||
)
|
)
|
||||||
advanced.add_argument(
|
|
||||||
'--tesseract-pagesegmode',
|
|
||||||
action='store',
|
|
||||||
type=int,
|
|
||||||
metavar='PSM',
|
|
||||||
choices=range(0, 14),
|
|
||||||
help="Set Tesseract page segmentation mode (see tesseract --help)",
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--tesseract-oem',
|
|
||||||
action='store',
|
|
||||||
type=int,
|
|
||||||
metavar='MODE',
|
|
||||||
choices=range(0, 4),
|
|
||||||
help=(
|
|
||||||
"Set Tesseract 4.0 OCR engine mode: "
|
|
||||||
"0 - original Tesseract only; "
|
|
||||||
"1 - neural nets LSTM only; "
|
|
||||||
"2 - Tesseract + LSTM; "
|
|
||||||
"3 - default."
|
|
||||||
),
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--pdf-renderer',
|
|
||||||
choices=['auto', 'hocr', 'sandwich'],
|
|
||||||
default='auto',
|
|
||||||
help="Choose OCR PDF renderer - the default option is to let OCRmyPDF "
|
|
||||||
"choose. See documentation for discussion.",
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--tesseract-timeout',
|
|
||||||
default=180.0,
|
|
||||||
type=numeric(float, 0),
|
|
||||||
metavar='SECONDS',
|
|
||||||
help='Give up on OCR after the timeout, but copy the preprocessed page '
|
|
||||||
'into the final output',
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--rotate-pages-threshold',
|
|
||||||
default=14.0,
|
|
||||||
type=numeric(float, 0, 1000),
|
|
||||||
metavar='CONFIDENCE',
|
|
||||||
help="Only rotate pages when confidence is above this value (arbitrary "
|
|
||||||
"units reported by tesseract)",
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--pdfa-image-compression',
|
|
||||||
choices=['auto', 'jpeg', 'lossless'],
|
|
||||||
default='auto',
|
|
||||||
help="Specify how to compress images in the output PDF/A. 'auto' lets "
|
|
||||||
"OCRmyPDF decide. 'jpeg' changes all grayscale and color images to "
|
|
||||||
"JPEG compression. 'lossless' uses PNG-style lossless compression "
|
|
||||||
"for all images. Monochrome images are always compressed using a "
|
|
||||||
"lossless codec. Compression settings "
|
|
||||||
"are applied to all pages, including those for which OCR was "
|
|
||||||
"skipped. Not supported for --output-type=pdf ; that setting "
|
|
||||||
"preserves the original compression of all images.",
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--user-words',
|
|
||||||
metavar='FILE',
|
|
||||||
help="Specify the location of the Tesseract user words file. This is a "
|
|
||||||
"list of words Tesseract should consider while performing OCR in "
|
|
||||||
"addition to its standard language dictionaries. This can improve "
|
|
||||||
"OCR quality especially for specialized and technical documents.",
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--user-patterns',
|
|
||||||
metavar='FILE',
|
|
||||||
help="Specify the location of the Tesseract user patterns file.",
|
|
||||||
)
|
|
||||||
advanced.add_argument(
|
|
||||||
'--fast-web-view',
|
|
||||||
type=numeric(float, 0),
|
|
||||||
default=1.0,
|
|
||||||
metavar="MEGABYTES",
|
|
||||||
help="If the size of file is more than this threshold (in MB), then "
|
|
||||||
"linearize the PDF for fast web viewing. This allows the PDF to be "
|
|
||||||
"displayed before it is fully downloaded in web browsers, but increases "
|
|
||||||
"the space required slightly. By default we skip this for small files "
|
|
||||||
"which do not benefit. If the threshold is 0 it will be apply to all files. "
|
|
||||||
"Set the threshold very high to disable.",
|
|
||||||
)
|
|
||||||
|
|
||||||
debugging = parser.add_argument_group(
|
|
||||||
"Debugging", "Arguments to help with troubleshooting and debugging"
|
|
||||||
)
|
|
||||||
debugging.add_argument(
|
|
||||||
'-k',
|
|
||||||
'--keep-temporary-files',
|
|
||||||
action='store_true',
|
|
||||||
help="Keep temporary files (helpful for debugging)",
|
|
||||||
)
|
|
||||||
debugging.add_argument('--tesseract-env', type=str, help=argparse.SUPPRESS)
|
|
||||||
|
|||||||
@@ -1,61 +0,0 @@
|
|||||||
# © 2017 James R. Barlow: github.com/jbarlow83
|
|
||||||
#
|
|
||||||
# This file is part of OCRmyPDF.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
"""Interface to qpdf executable"""
|
|
||||||
|
|
||||||
from io import StringIO
|
|
||||||
|
|
||||||
import pikepdf
|
|
||||||
|
|
||||||
|
|
||||||
def version():
|
|
||||||
return pikepdf.__libqpdf_version__
|
|
||||||
|
|
||||||
|
|
||||||
def check(input_file, log=None):
|
|
||||||
pdf = None
|
|
||||||
try:
|
|
||||||
pdf = pikepdf.open(input_file)
|
|
||||||
except pikepdf.PdfError as e:
|
|
||||||
if log:
|
|
||||||
log.error(e)
|
|
||||||
return False
|
|
||||||
else:
|
|
||||||
messages = pdf.check()
|
|
||||||
for msg in messages:
|
|
||||||
if 'error' in msg.lower():
|
|
||||||
log.error(msg)
|
|
||||||
else:
|
|
||||||
log.warning(msg)
|
|
||||||
|
|
||||||
sio = StringIO()
|
|
||||||
linearize = None
|
|
||||||
try:
|
|
||||||
pdf.check_linearization(sio)
|
|
||||||
except RuntimeError:
|
|
||||||
pass
|
|
||||||
else:
|
|
||||||
linearize = sio.getvalue()
|
|
||||||
if linearize:
|
|
||||||
log.warning(linearize)
|
|
||||||
|
|
||||||
if not messages and not linearize:
|
|
||||||
return True
|
|
||||||
return False
|
|
||||||
finally:
|
|
||||||
if pdf:
|
|
||||||
pdf.close()
|
|
||||||
+93
-16
@@ -20,23 +20,56 @@ import multiprocessing
|
|||||||
import os
|
import os
|
||||||
import shutil
|
import shutil
|
||||||
import warnings
|
import warnings
|
||||||
|
from collections import namedtuple
|
||||||
from collections.abc import Iterable
|
from collections.abc import Iterable
|
||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from functools import wraps
|
from functools import wraps
|
||||||
|
from io import StringIO
|
||||||
|
from math import isclose
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from typing import Any, Sequence, TypeVar
|
||||||
|
|
||||||
|
import pikepdf
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike, *args, **kwargs):
|
class Resolution(namedtuple('Resolution', ('x', 'y'))):
|
||||||
|
__slots__ = ()
|
||||||
|
|
||||||
|
def round(self, ndigits: int):
|
||||||
|
return Resolution(round(self.x, ndigits), round(self.y, ndigits))
|
||||||
|
|
||||||
|
def to_int(self):
|
||||||
|
return Resolution(int(round(self.x)), int(round(self.y)))
|
||||||
|
|
||||||
|
@property
|
||||||
|
def is_square(self) -> bool:
|
||||||
|
return isclose(self.x, self.y, rel_tol=1e-3)
|
||||||
|
|
||||||
|
def take_max(self, vals, yvals=None):
|
||||||
|
if yvals is not None:
|
||||||
|
return Resolution(max(self.x, *vals), max(self.y, *yvals))
|
||||||
|
max_x, max_y = self.x, self.y
|
||||||
|
for x, y in vals:
|
||||||
|
max_x = max(x, max_x)
|
||||||
|
max_y = max(y, max_y)
|
||||||
|
return Resolution(max_x, max_y)
|
||||||
|
|
||||||
|
def flip_axis(self):
|
||||||
|
return Resolution(self.y, self.x)
|
||||||
|
|
||||||
|
def __str__(self):
|
||||||
|
return f"{self.x:f}x{self.y:f}"
|
||||||
|
|
||||||
|
def __repr__(self):
|
||||||
|
return f"Resolution({self.x}x{self.y} dpi)"
|
||||||
|
|
||||||
|
|
||||||
|
def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike):
|
||||||
"""
|
"""
|
||||||
Helper function: relinks soft symbolic link if necessary
|
Helper function: relinks soft symbolic link if necessary
|
||||||
"""
|
"""
|
||||||
if len(args) == 1 and isinstance(args[0], logging.Logger):
|
|
||||||
log.warning("Deprecated: safe_symlink(,log)")
|
|
||||||
if 'log' in kwargs:
|
|
||||||
log.warning('Deprecated: safe_symlink(...log=)')
|
|
||||||
|
|
||||||
input_file = os.fspath(input_file)
|
input_file = os.fspath(input_file)
|
||||||
soft_link_name = os.fspath(soft_link_name)
|
soft_link_name = os.fspath(soft_link_name)
|
||||||
|
|
||||||
@@ -72,21 +105,28 @@ def safe_symlink(input_file: os.PathLike, soft_link_name: os.PathLike, *args, **
|
|||||||
os.symlink(os.path.abspath(input_file), soft_link_name)
|
os.symlink(os.path.abspath(input_file), soft_link_name)
|
||||||
|
|
||||||
|
|
||||||
def is_iterable_notstr(thing):
|
def samefile(f1: os.PathLike, f2: os.PathLike):
|
||||||
|
if os.name == 'nt':
|
||||||
|
return f1 == f2
|
||||||
|
else:
|
||||||
|
return os.path.samefile(f1, f2)
|
||||||
|
|
||||||
|
|
||||||
|
def is_iterable_notstr(thing: Any) -> bool:
|
||||||
return isinstance(thing, Iterable) and not isinstance(thing, str)
|
return isinstance(thing, Iterable) and not isinstance(thing, str)
|
||||||
|
|
||||||
|
|
||||||
def monotonic(L: Iterable):
|
def monotonic(L: Sequence) -> bool:
|
||||||
"""Does list increase monotonically?"""
|
"""Does list increase monotonically?"""
|
||||||
return all(b > a for a, b in zip(L, L[1:]))
|
return all(b > a for a, b in zip(L, L[1:]))
|
||||||
|
|
||||||
|
|
||||||
def page_number(input_file: os.PathLike):
|
def page_number(input_file: os.PathLike) -> int:
|
||||||
"""Get one-based page number implied by filename (000002.pdf -> 2)"""
|
"""Get one-based page number implied by filename (000002.pdf -> 2)"""
|
||||||
return int(os.path.basename(os.fspath(input_file))[0:6])
|
return int(os.path.basename(os.fspath(input_file))[0:6])
|
||||||
|
|
||||||
|
|
||||||
def available_cpu_count():
|
def available_cpu_count() -> int:
|
||||||
try:
|
try:
|
||||||
return multiprocessing.cpu_count()
|
return multiprocessing.cpu_count()
|
||||||
except NotImplementedError:
|
except NotImplementedError:
|
||||||
@@ -97,7 +137,7 @@ def available_cpu_count():
|
|||||||
return 1
|
return 1
|
||||||
|
|
||||||
|
|
||||||
def is_file_writable(test_file: os.PathLike):
|
def is_file_writable(test_file: os.PathLike) -> bool:
|
||||||
"""Intentionally racy test if target is writable.
|
"""Intentionally racy test if target is writable.
|
||||||
|
|
||||||
We intend to write to the output file if and only if we succeed and
|
We intend to write to the output file if and only if we succeed and
|
||||||
@@ -105,11 +145,7 @@ def is_file_writable(test_file: os.PathLike):
|
|||||||
the location is writable.
|
the location is writable.
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
if not isinstance(test_file, Path):
|
p = Path(test_file)
|
||||||
p = Path(test_file)
|
|
||||||
else:
|
|
||||||
p = test_file
|
|
||||||
|
|
||||||
if p.is_symlink():
|
if p.is_symlink():
|
||||||
p = p.resolve(strict=False)
|
p = p.resolve(strict=False)
|
||||||
|
|
||||||
@@ -136,6 +172,47 @@ def is_file_writable(test_file: os.PathLike):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def check_pdf(input_file: Path) -> bool:
|
||||||
|
pdf = None
|
||||||
|
try:
|
||||||
|
pdf = pikepdf.open(input_file)
|
||||||
|
except pikepdf.PdfError as e:
|
||||||
|
log.error(e)
|
||||||
|
return False
|
||||||
|
else:
|
||||||
|
messages = pdf.check()
|
||||||
|
for msg in messages:
|
||||||
|
if 'error' in msg.lower():
|
||||||
|
log.error(msg)
|
||||||
|
else:
|
||||||
|
log.warning(msg)
|
||||||
|
|
||||||
|
sio = StringIO()
|
||||||
|
linearize = None
|
||||||
|
try:
|
||||||
|
pdf.check_linearization(sio)
|
||||||
|
except RuntimeError:
|
||||||
|
pass
|
||||||
|
else:
|
||||||
|
linearize = sio.getvalue()
|
||||||
|
if linearize:
|
||||||
|
log.warning(linearize)
|
||||||
|
|
||||||
|
if not messages and not linearize:
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
finally:
|
||||||
|
if pdf:
|
||||||
|
pdf.close()
|
||||||
|
|
||||||
|
|
||||||
|
T = TypeVar('T')
|
||||||
|
|
||||||
|
|
||||||
|
def clamp(n: T, smallest: T, largest: T) -> T:
|
||||||
|
return max(smallest, min(n, largest))
|
||||||
|
|
||||||
|
|
||||||
def deprecated(func):
|
def deprecated(func):
|
||||||
"""Warn that function is deprecated"""
|
"""Warn that function is deprecated"""
|
||||||
|
|
||||||
|
|||||||
@@ -29,11 +29,16 @@
|
|||||||
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
# SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
|
import os
|
||||||
import re
|
import re
|
||||||
from collections import namedtuple
|
from collections import namedtuple
|
||||||
|
from itertools import chain
|
||||||
from math import atan, cos, sin
|
from math import atan, cos, sin
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Union
|
||||||
from xml.etree import ElementTree
|
from xml.etree import ElementTree
|
||||||
|
|
||||||
|
from reportlab.lib.colors import black, cyan, magenta, red
|
||||||
from reportlab.lib.units import inch
|
from reportlab.lib.units import inch
|
||||||
from reportlab.pdfgen.canvas import Canvas
|
from reportlab.pdfgen.canvas import Canvas
|
||||||
|
|
||||||
@@ -64,9 +69,9 @@ class HocrTransform:
|
|||||||
{'ff': 'ff', 'ffi': 'ffi', 'ffl': 'ffl', 'fi': 'fi', 'fl': 'fl'}
|
{'ff': 'ff', 'ffi': 'ffi', 'ffl': 'ffl', 'fi': 'fi', 'fl': 'fl'}
|
||||||
)
|
)
|
||||||
|
|
||||||
def __init__(self, hocrFileName, dpi):
|
def __init__(self, hocr_filename: Union[str, Path], dpi: float):
|
||||||
self.dpi = dpi
|
self.dpi = dpi
|
||||||
self.hocr = ElementTree.parse(hocrFileName)
|
self.hocr = ElementTree.parse(os.fspath(hocr_filename))
|
||||||
|
|
||||||
# if the hOCR file has a namespace, ElementTree requires its use to
|
# if the hOCR file has a namespace, ElementTree requires its use to
|
||||||
# find elements
|
# find elements
|
||||||
@@ -77,7 +82,7 @@ class HocrTransform:
|
|||||||
|
|
||||||
# get dimension in pt (not pixel!!!!) of the OCRed image
|
# get dimension in pt (not pixel!!!!) of the OCRed image
|
||||||
self.width, self.height = None, None
|
self.width, self.height = None, None
|
||||||
for div in self.hocr.findall(".//%sdiv[@class='ocr_page']" % (self.xmlns)):
|
for div in self.hocr.findall(self._child_xpath('div', 'ocr_page')):
|
||||||
coords = self.element_coordinates(div)
|
coords = self.element_coordinates(div)
|
||||||
pt_coords = self.pt_from_pixel(coords)
|
pt_coords = self.pt_from_pixel(coords)
|
||||||
self.width = pt_coords.x2 - pt_coords.x1
|
self.width = pt_coords.x2 - pt_coords.x1
|
||||||
@@ -94,7 +99,7 @@ class HocrTransform:
|
|||||||
"""
|
"""
|
||||||
if self.hocr is None:
|
if self.hocr is None:
|
||||||
return ''
|
return ''
|
||||||
body = self.hocr.find(".//%sbody" % (self.xmlns))
|
body = self.hocr.find(self._child_xpath('body'))
|
||||||
if body:
|
if body:
|
||||||
return self._get_element_text(body)
|
return self._get_element_text(body)
|
||||||
else:
|
else:
|
||||||
@@ -107,19 +112,19 @@ class HocrTransform:
|
|||||||
text = ''
|
text = ''
|
||||||
if element.text is not None:
|
if element.text is not None:
|
||||||
text += element.text
|
text += element.text
|
||||||
for child in element.getchildren():
|
for child in element:
|
||||||
text += self._get_element_text(child)
|
text += self._get_element_text(child)
|
||||||
if element.tail is not None:
|
if element.tail is not None:
|
||||||
text += element.tail
|
text += element.tail
|
||||||
return text
|
return text
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def element_coordinates(cls, element):
|
def element_coordinates(cls, element) -> Rect:
|
||||||
"""
|
"""
|
||||||
Returns a tuple containing the coordinates of the bounding box around
|
Returns a tuple containing the coordinates of the bounding box around
|
||||||
an element
|
an element
|
||||||
"""
|
"""
|
||||||
out = (0, 0, 0, 0)
|
out = Rect._make(0 for _ in range(4))
|
||||||
if 'title' in element.attrib:
|
if 'title' in element.attrib:
|
||||||
matches = cls.box_pattern.search(element.attrib['title'])
|
matches = cls.box_pattern.search(element.attrib['title'])
|
||||||
if matches:
|
if matches:
|
||||||
@@ -136,7 +141,7 @@ class HocrTransform:
|
|||||||
matches = cls.baseline_pattern.search(element.attrib['title'])
|
matches = cls.baseline_pattern.search(element.attrib['title'])
|
||||||
if matches:
|
if matches:
|
||||||
return float(matches.group(1)), int(matches.group(2))
|
return float(matches.group(1)), int(matches.group(2))
|
||||||
return (0, 0)
|
return (0.0, 0.0)
|
||||||
|
|
||||||
def pt_from_pixel(self, pxl):
|
def pt_from_pixel(self, pxl):
|
||||||
"""
|
"""
|
||||||
@@ -144,8 +149,14 @@ class HocrTransform:
|
|||||||
"""
|
"""
|
||||||
return Rect._make((c / self.dpi * inch) for c in pxl)
|
return Rect._make((c / self.dpi * inch) for c in pxl)
|
||||||
|
|
||||||
|
def _child_xpath(self, html_tag, html_class=None):
|
||||||
|
xpath = f".//{self.xmlns}{html_tag}"
|
||||||
|
if html_class:
|
||||||
|
xpath += f"[@class='{html_class}']"
|
||||||
|
return xpath
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def replace_unsupported_chars(cls, s):
|
def replace_unsupported_chars(cls, s: str):
|
||||||
"""
|
"""
|
||||||
Given an input string, returns the corresponding string that:
|
Given an input string, returns the corresponding string that:
|
||||||
- is available in the helvetica facetype
|
- is available in the helvetica facetype
|
||||||
@@ -153,14 +164,19 @@ class HocrTransform:
|
|||||||
"""
|
"""
|
||||||
return s.translate(cls.ligatures)
|
return s.translate(cls.ligatures)
|
||||||
|
|
||||||
|
def topdown_position(self, element):
|
||||||
|
pxl_line_coords = self.element_coordinates(element)
|
||||||
|
line_box = self.pt_from_pixel(pxl_line_coords)
|
||||||
|
return -line_box.y2
|
||||||
|
|
||||||
def to_pdf(
|
def to_pdf(
|
||||||
self,
|
self,
|
||||||
outFileName,
|
out_filename: Path,
|
||||||
imageFileName=None,
|
image_filename: Path = None,
|
||||||
showBoundingboxes=False,
|
show_bounding_boxes: bool = False,
|
||||||
fontname="Helvetica",
|
fontname: str = "Helvetica",
|
||||||
invisibleText=False,
|
invisible_text: bool = False,
|
||||||
interwordSpaces=False,
|
interword_spaces: bool = False,
|
||||||
):
|
):
|
||||||
"""
|
"""
|
||||||
Creates a PDF file with an image superimposed on top of the text.
|
Creates a PDF file with an image superimposed on top of the text.
|
||||||
@@ -172,16 +188,19 @@ class HocrTransform:
|
|||||||
"""
|
"""
|
||||||
# create the PDF file
|
# create the PDF file
|
||||||
# page size in points (1/72 in.)
|
# page size in points (1/72 in.)
|
||||||
pdf = Canvas(outFileName, pagesize=(self.width, self.height), pageCompression=1)
|
pdf = Canvas(
|
||||||
|
os.fspath(out_filename),
|
||||||
|
pagesize=(self.width, self.height),
|
||||||
|
pageCompression=1,
|
||||||
|
)
|
||||||
|
|
||||||
# draw bounding box for each paragraph
|
# draw bounding box for each paragraph
|
||||||
# light blue for bounding box of paragraph
|
# light blue for bounding box of paragraph
|
||||||
pdf.setStrokeColorRGB(0, 1, 1)
|
pdf.setStrokeColor(cyan)
|
||||||
# light blue for bounding box of paragraph
|
# light blue for bounding box of paragraph
|
||||||
pdf.setFillColorRGB(0, 1, 1)
|
pdf.setFillColor(cyan)
|
||||||
pdf.setLineWidth(0) # no line for bounding box
|
pdf.setLineWidth(0) # no line for bounding box
|
||||||
for elem in self.hocr.findall(".//%sp[@class='%s']" % (self.xmlns, "ocr_par")):
|
for elem in self.hocr.iterfind(self._child_xpath('p', 'ocr_par')):
|
||||||
|
|
||||||
elemtxt = self._get_element_text(elem).rstrip()
|
elemtxt = self._get_element_text(elem).rstrip()
|
||||||
if len(elemtxt) == 0:
|
if len(elemtxt) == 0:
|
||||||
continue
|
continue
|
||||||
@@ -190,14 +209,19 @@ class HocrTransform:
|
|||||||
pt = self.pt_from_pixel(pxl_coords)
|
pt = self.pt_from_pixel(pxl_coords)
|
||||||
|
|
||||||
# draw the bbox border
|
# draw the bbox border
|
||||||
if showBoundingboxes: # pragma: no cover
|
if show_bounding_boxes: # pragma: no cover
|
||||||
pdf.rect(
|
pdf.rect(
|
||||||
pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, fill=1
|
pt.x1, self.height - pt.y2, pt.x2 - pt.x1, pt.y2 - pt.y1, fill=1
|
||||||
)
|
)
|
||||||
|
|
||||||
found_lines = False
|
found_lines = False
|
||||||
for line in self.hocr.findall(
|
for line in sorted(
|
||||||
".//%sspan[@class='%s']" % (self.xmlns, "ocr_line")
|
chain(
|
||||||
|
self.hocr.iterfind(self._child_xpath('span', 'ocr_header')),
|
||||||
|
self.hocr.iterfind(self._child_xpath('span', 'ocr_line')),
|
||||||
|
self.hocr.iterfind(self._child_xpath('span', 'ocr_textfloat')),
|
||||||
|
),
|
||||||
|
key=self.topdown_position,
|
||||||
):
|
):
|
||||||
found_lines = True
|
found_lines = True
|
||||||
self._do_line(
|
self._do_line(
|
||||||
@@ -205,26 +229,28 @@ class HocrTransform:
|
|||||||
line,
|
line,
|
||||||
"ocrx_word",
|
"ocrx_word",
|
||||||
fontname,
|
fontname,
|
||||||
invisibleText,
|
invisible_text,
|
||||||
interwordSpaces,
|
interword_spaces,
|
||||||
showBoundingboxes,
|
show_bounding_boxes,
|
||||||
)
|
)
|
||||||
|
|
||||||
if not found_lines:
|
if not found_lines:
|
||||||
# Tesseract did not report any lines (just words)
|
# Tesseract did not report any lines (just words)
|
||||||
root = self.hocr.find(".//%sdiv[@class='%s']" % (self.xmlns, "ocr_page"))
|
root = self.hocr.find(self._child_xpath('div', 'ocr_page'))
|
||||||
self._do_line(
|
self._do_line(
|
||||||
pdf,
|
pdf,
|
||||||
root,
|
root,
|
||||||
"ocrx_word",
|
"ocrx_word",
|
||||||
fontname,
|
fontname,
|
||||||
invisibleText,
|
invisible_text,
|
||||||
interwordSpaces,
|
interword_spaces,
|
||||||
showBoundingboxes,
|
show_bounding_boxes,
|
||||||
)
|
)
|
||||||
# put the image on the page, scaled to fill the page
|
# put the image on the page, scaled to fill the page
|
||||||
if imageFileName is not None:
|
if image_filename is not None:
|
||||||
pdf.drawImage(imageFileName, 0, 0, width=self.width, height=self.height)
|
pdf.drawImage(
|
||||||
|
os.fspath(image_filename), 0, 0, width=self.width, height=self.height
|
||||||
|
)
|
||||||
|
|
||||||
# finish up the page and save it
|
# finish up the page and save it
|
||||||
pdf.showPage()
|
pdf.showPage()
|
||||||
@@ -236,13 +262,13 @@ class HocrTransform:
|
|||||||
|
|
||||||
def _do_line(
|
def _do_line(
|
||||||
self,
|
self,
|
||||||
pdf,
|
pdf: Canvas,
|
||||||
line,
|
line,
|
||||||
elemclass,
|
elemclass: str,
|
||||||
fontname,
|
fontname: str,
|
||||||
invisibleText,
|
invisible_text: bool,
|
||||||
interwordSpaces,
|
interword_spaces: bool,
|
||||||
showBoundingboxes,
|
show_bounding_boxes: bool,
|
||||||
):
|
):
|
||||||
pxl_line_coords = self.element_coordinates(line)
|
pxl_line_coords = self.element_coordinates(line)
|
||||||
line_box = self.pt_from_pixel(pxl_line_coords)
|
line_box = self.pt_from_pixel(pxl_line_coords)
|
||||||
@@ -262,17 +288,17 @@ class HocrTransform:
|
|||||||
# on a sloped baseline and the edge of the bounding box.
|
# on a sloped baseline and the edge of the bounding box.
|
||||||
fontsize = (line_height - abs(intercept)) / cos_a
|
fontsize = (line_height - abs(intercept)) / cos_a
|
||||||
text.setFont(fontname, fontsize)
|
text.setFont(fontname, fontsize)
|
||||||
if invisibleText:
|
if invisible_text:
|
||||||
text.setTextRenderMode(3) # Invisible (indicates OCR text)
|
text.setTextRenderMode(3) # Invisible (indicates OCR text)
|
||||||
|
|
||||||
# Intercept is normally negative, so this places it above the bottom
|
# Intercept is normally negative, so this places it above the bottom
|
||||||
# of the line box
|
# of the line box
|
||||||
baseline_y2 = self.height - (line_box.y2 + intercept)
|
baseline_y2 = self.height - (line_box.y2 + intercept)
|
||||||
|
|
||||||
if showBoundingboxes: # pragma: no cover
|
if show_bounding_boxes: # pragma: no cover
|
||||||
# draw the baseline in magenta, dashed
|
# draw the baseline in magenta, dashed
|
||||||
pdf.setDash()
|
pdf.setDash()
|
||||||
pdf.setStrokeColorRGB(0.95, 0.65, 0.95)
|
pdf.setStrokeColor(magenta)
|
||||||
pdf.setLineWidth(0.5)
|
pdf.setLineWidth(0.5)
|
||||||
# negate slope because it is defined as a rise/run in pixel
|
# negate slope because it is defined as a rise/run in pixel
|
||||||
# coordinates and page coordinates have the y axis flipped
|
# coordinates and page coordinates have the y axis flipped
|
||||||
@@ -284,12 +310,12 @@ class HocrTransform:
|
|||||||
)
|
)
|
||||||
# light green for bounding box of word/line
|
# light green for bounding box of word/line
|
||||||
pdf.setDash(6, 3)
|
pdf.setDash(6, 3)
|
||||||
pdf.setStrokeColorRGB(1, 0, 0)
|
pdf.setStrokeColor(red)
|
||||||
|
|
||||||
text.setTextTransform(cos_a, -sin_a, sin_a, cos_a, line_box.x1, baseline_y2)
|
text.setTextTransform(cos_a, -sin_a, sin_a, cos_a, line_box.x1, baseline_y2)
|
||||||
pdf.setFillColorRGB(0, 0, 0) # text in black
|
pdf.setFillColor(black) # text in black
|
||||||
|
|
||||||
elements = line.findall(".//%sspan[@class='%s']" % (self.xmlns, elemclass))
|
elements = line.findall(self._child_xpath('span', elemclass))
|
||||||
for elem in elements:
|
for elem in elements:
|
||||||
elemtxt = self._get_element_text(elem).strip()
|
elemtxt = self._get_element_text(elem).strip()
|
||||||
elemtxt = self.replace_unsupported_chars(elemtxt)
|
elemtxt = self.replace_unsupported_chars(elemtxt)
|
||||||
@@ -298,7 +324,7 @@ class HocrTransform:
|
|||||||
|
|
||||||
pxl_coords = self.element_coordinates(elem)
|
pxl_coords = self.element_coordinates(elem)
|
||||||
box = self.pt_from_pixel(pxl_coords)
|
box = self.pt_from_pixel(pxl_coords)
|
||||||
if interwordSpaces:
|
if interword_spaces:
|
||||||
# if `--interword-spaces` is true, append a space
|
# if `--interword-spaces` is true, append a space
|
||||||
# to the end of each text element to allow simpler PDF viewers
|
# to the end of each text element to allow simpler PDF viewers
|
||||||
# such as PDF.js to better recognize words in search and copy
|
# such as PDF.js to better recognize words in search and copy
|
||||||
@@ -318,7 +344,7 @@ class HocrTransform:
|
|||||||
font_width = pdf.stringWidth(elemtxt, fontname, fontsize)
|
font_width = pdf.stringWidth(elemtxt, fontname, fontsize)
|
||||||
|
|
||||||
# draw the bbox border
|
# draw the bbox border
|
||||||
if showBoundingboxes: # pragma: no cover
|
if show_bounding_boxes: # pragma: no cover
|
||||||
pdf.rect(
|
pdf.rect(
|
||||||
box.x1, self.height - line_box.y2, box_width, line_height, fill=0
|
box.x1, self.height - line_box.y2, box_width, line_height, fill=0
|
||||||
)
|
)
|
||||||
@@ -385,5 +411,5 @@ if __name__ == "__main__":
|
|||||||
args.outputfile,
|
args.outputfile,
|
||||||
args.image,
|
args.image,
|
||||||
args.boundingboxes,
|
args.boundingboxes,
|
||||||
interwordSpaces=args.interword_spaces,
|
interword_spaces=args.interword_spaces,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -29,13 +29,13 @@ from collections.abc import Sequence
|
|||||||
from contextlib import suppress
|
from contextlib import suppress
|
||||||
from ctypes.util import find_library
|
from ctypes.util import find_library
|
||||||
from functools import lru_cache
|
from functools import lru_cache
|
||||||
from io import BytesIO
|
from io import BytesIO, UnsupportedOperation
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from tempfile import TemporaryFile
|
from tempfile import TemporaryFile
|
||||||
|
|
||||||
from .exceptions import MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
from .exec import shim_paths_with_program_files
|
from ocrmypdf.lib._leptonica import ffi
|
||||||
from .lib._leptonica import ffi
|
from ocrmypdf.subprocess import shim_paths_with_program_files
|
||||||
|
|
||||||
# pylint: disable=protected-access
|
# pylint: disable=protected-access
|
||||||
|
|
||||||
@@ -96,7 +96,6 @@ class _LeptonicaErrorTrap:
|
|||||||
self.no_stderr = False
|
self.no_stderr = False
|
||||||
|
|
||||||
def __enter__(self):
|
def __enter__(self):
|
||||||
from io import UnsupportedOperation
|
|
||||||
|
|
||||||
self.tmpfile = TemporaryFile()
|
self.tmpfile = TemporaryFile()
|
||||||
|
|
||||||
@@ -351,7 +350,7 @@ class Pix(LeptonicaObject):
|
|||||||
py_file.write(buffer)
|
py_file.write(buffer)
|
||||||
|
|
||||||
@classmethod
|
@classmethod
|
||||||
def frompil(self, pillow_image):
|
def frompil(cls, pillow_image):
|
||||||
"""Create a copy of a PIL.Image from this Pix"""
|
"""Create a copy of a PIL.Image from this Pix"""
|
||||||
bio = BytesIO()
|
bio = BytesIO()
|
||||||
pillow_image.save(bio, format='png', compress_level=1)
|
pillow_image.save(bio, format='png', compress_level=1)
|
||||||
@@ -363,7 +362,7 @@ class Pix(LeptonicaObject):
|
|||||||
|
|
||||||
def topil(self):
|
def topil(self):
|
||||||
"""Returns a PIL.Image version of this Pix"""
|
"""Returns a PIL.Image version of this Pix"""
|
||||||
from PIL import Image
|
from PIL import Image # pylint: disable=import-outside-toplevel
|
||||||
|
|
||||||
# Leptonica manages data in words, so it implicitly does an endian
|
# Leptonica manages data in words, so it implicitly does an endian
|
||||||
# swap. Tell Pillow about this when it reads the data.
|
# swap. Tell Pillow about this when it reads the data.
|
||||||
@@ -534,16 +533,7 @@ class Pix(LeptonicaObject):
|
|||||||
)
|
)
|
||||||
return Pix(thresh_pix)
|
return Pix(thresh_pix)
|
||||||
|
|
||||||
def crop_to_foreground(
|
def crop_to_foreground(self, threshold=128, mindist=70, erasedist=30, showmorph=0):
|
||||||
self,
|
|
||||||
threshold=128,
|
|
||||||
mindist=70,
|
|
||||||
erasedist=30,
|
|
||||||
pagenum=0,
|
|
||||||
showmorph=0,
|
|
||||||
display=0,
|
|
||||||
pdfdir=ffi.NULL,
|
|
||||||
):
|
|
||||||
if get_leptonica_version() < 'leptonica-1.76':
|
if get_leptonica_version() < 'leptonica-1.76':
|
||||||
# Leptonica 1.76 changed the API for pixFindPageForeground; we don't
|
# Leptonica 1.76 changed the API for pixFindPageForeground; we don't
|
||||||
# support the old version
|
# support the old version
|
||||||
|
|||||||
+79
-75
@@ -15,10 +15,11 @@
|
|||||||
# You should have received a copy of the GNU General Public License
|
# You should have received a copy of the GNU General Public License
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
import concurrent.futures
|
import logging
|
||||||
import sys
|
import sys
|
||||||
import tempfile
|
import tempfile
|
||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
|
from functools import partial
|
||||||
from os import fspath
|
from os import fspath
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -27,11 +28,14 @@ from pikepdf import Dictionary, Name
|
|||||||
from PIL import Image
|
from PIL import Image
|
||||||
from tqdm import tqdm
|
from tqdm import tqdm
|
||||||
|
|
||||||
from . import leptonica
|
from ocrmypdf import leptonica
|
||||||
from ._jobcontext import PDFContext
|
from ocrmypdf._concurrent import exec_progress_pool
|
||||||
from .exceptions import OutputFileAccessError
|
from ocrmypdf._exec import jbig2enc, pngquant
|
||||||
from .exec import jbig2enc, pngquant
|
from ocrmypdf._jobcontext import PdfContext
|
||||||
from .helpers import safe_symlink
|
from ocrmypdf.exceptions import OutputFileAccessError
|
||||||
|
from ocrmypdf.helpers import safe_symlink
|
||||||
|
|
||||||
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
DEFAULT_JPEG_QUALITY = 75
|
DEFAULT_JPEG_QUALITY = 75
|
||||||
DEFAULT_PNG_QUALITY = 70
|
DEFAULT_PNG_QUALITY = 70
|
||||||
@@ -53,7 +57,7 @@ def tif_name(root, xref):
|
|||||||
return img_name(root, xref, '.tif')
|
return img_name(root, xref, '.tif')
|
||||||
|
|
||||||
|
|
||||||
def extract_image_filter(pike, root, log, image, xref):
|
def extract_image_filter(pike, root, image, xref):
|
||||||
if image.Subtype != Name.Image:
|
if image.Subtype != Name.Image:
|
||||||
return None
|
return None
|
||||||
if image.Length < 100:
|
if image.Length < 100:
|
||||||
@@ -79,8 +83,8 @@ def extract_image_filter(pike, root, log, image, xref):
|
|||||||
return pim, filtdp
|
return pim, filtdp
|
||||||
|
|
||||||
|
|
||||||
def extract_image_jbig2(*, pike, root, log, image, xref, options):
|
def extract_image_jbig2(*, pike, root, image, xref, options):
|
||||||
result = extract_image_filter(pike, root, log, image, xref)
|
result = extract_image_filter(pike, root, image, xref)
|
||||||
if result is None:
|
if result is None:
|
||||||
return None
|
return None
|
||||||
pim, filtdp = result
|
pim, filtdp = result
|
||||||
@@ -101,8 +105,8 @@ def extract_image_jbig2(*, pike, root, log, image, xref, options):
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def extract_image_generic(*, pike, root, log, image, xref, options):
|
def extract_image_generic(*, pike, root, image, xref, options):
|
||||||
result = extract_image_filter(pike, root, log, image, xref)
|
result = extract_image_filter(pike, root, image, xref)
|
||||||
if result is None:
|
if result is None:
|
||||||
return None
|
return None
|
||||||
pim, filtdp = result
|
pim, filtdp = result
|
||||||
@@ -170,7 +174,7 @@ def extract_image_generic(*, pike, root, log, image, xref, options):
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def extract_images(pike, root, log, options, extract_fn):
|
def extract_images(pike, root, options, extract_fn):
|
||||||
"""Extract image using extract_fn
|
"""Extract image using extract_fn
|
||||||
|
|
||||||
Enumerate images on each page, lookup their xref/ID number in the PDF.
|
Enumerate images on each page, lookup their xref/ID number in the PDF.
|
||||||
@@ -212,9 +216,9 @@ def extract_images(pike, root, log, options, extract_fn):
|
|||||||
image = pike.get_object((xref, 0))
|
image = pike.get_object((xref, 0))
|
||||||
try:
|
try:
|
||||||
result = extract_fn(
|
result = extract_fn(
|
||||||
pike=pike, root=root, log=log, image=image, xref=xref, options=options
|
pike=pike, root=root, image=image, xref=xref, options=options
|
||||||
)
|
)
|
||||||
except Exception as e:
|
except Exception as e: # pylint: disable=broad-except
|
||||||
log.debug("Image xref %s, error %s", xref, repr(e))
|
log.debug("Image xref %s, error %s", xref, repr(e))
|
||||||
errors += 1
|
errors += 1
|
||||||
else:
|
else:
|
||||||
@@ -223,12 +227,12 @@ def extract_images(pike, root, log, options, extract_fn):
|
|||||||
yield pageno_for_xref[xref], xref, ext
|
yield pageno_for_xref[xref], xref, ext
|
||||||
|
|
||||||
|
|
||||||
def extract_images_generic(pike, root, log, options):
|
def extract_images_generic(pike, root, options):
|
||||||
"""Extract any >=2bpp image we think we can improve"""
|
"""Extract any >=2bpp image we think we can improve"""
|
||||||
|
|
||||||
jpegs = []
|
jpegs = []
|
||||||
pngs = []
|
pngs = []
|
||||||
for _, xref, ext in extract_images(pike, root, log, options, extract_image_generic):
|
for _, xref, ext in extract_images(pike, root, options, extract_image_generic):
|
||||||
log.debug('xref = %s ext = %s', xref, ext)
|
log.debug('xref = %s ext = %s', xref, ext)
|
||||||
if ext == '.png':
|
if ext == '.png':
|
||||||
pngs.append(xref)
|
pngs.append(xref)
|
||||||
@@ -238,13 +242,11 @@ def extract_images_generic(pike, root, log, options):
|
|||||||
return jpegs, pngs
|
return jpegs, pngs
|
||||||
|
|
||||||
|
|
||||||
def extract_images_jbig2(pike, root, log, options):
|
def extract_images_jbig2(pike, root, options):
|
||||||
"""Extract any bitonal image that we think we can improve as JBIG2"""
|
"""Extract any bitonal image that we think we can improve as JBIG2"""
|
||||||
|
|
||||||
jbig2_groups = defaultdict(list)
|
jbig2_groups = defaultdict(list)
|
||||||
for pageno, xref, ext in extract_images(
|
for pageno, xref, ext in extract_images(pike, root, options, extract_image_jbig2):
|
||||||
pike, root, log, options, extract_image_jbig2
|
|
||||||
):
|
|
||||||
group = pageno // options.jbig2_page_group_size
|
group = pageno // options.jbig2_page_group_size
|
||||||
jbig2_groups[group].append((xref, ext))
|
jbig2_groups[group].append((xref, ext))
|
||||||
|
|
||||||
@@ -256,55 +258,55 @@ def extract_images_jbig2(pike, root, log, options):
|
|||||||
return jbig2_groups
|
return jbig2_groups
|
||||||
|
|
||||||
|
|
||||||
def _produce_jbig2_images(jbig2_groups, root, log, options):
|
def _produce_jbig2_images(jbig2_groups, root, options):
|
||||||
"""Produce JBIG2 images from their groups"""
|
"""Produce JBIG2 images from their groups"""
|
||||||
|
|
||||||
def jbig2_group_futures(executor, root, groups):
|
def jbig2_group_args(root, groups):
|
||||||
for group, xref_exts in groups.items():
|
for group, xref_exts in groups.items():
|
||||||
prefix = f'group{group:08d}'
|
prefix = f'group{group:08d}'
|
||||||
future = executor.submit(
|
yield dict(
|
||||||
jbig2enc.convert_group,
|
|
||||||
cwd=fspath(root),
|
cwd=fspath(root),
|
||||||
infiles=(img_name(root, xref, ext) for xref, ext in xref_exts),
|
infiles=(img_name(root, xref, ext) for xref, ext in xref_exts),
|
||||||
out_prefix=prefix,
|
out_prefix=prefix,
|
||||||
)
|
)
|
||||||
yield future
|
|
||||||
|
|
||||||
def jbig2_single_futures(executor, root, groups):
|
def jbig2_single_args(root, groups):
|
||||||
for group, xref_exts in groups.items():
|
for group, xref_exts in groups.items():
|
||||||
prefix = f'group{group:08d}'
|
prefix = f'group{group:08d}'
|
||||||
# Second loop is to ensure multiple images per page are unpacked
|
# Second loop is to ensure multiple images per page are unpacked
|
||||||
for n, xref_ext in enumerate(xref_exts):
|
for n, xref_ext in enumerate(xref_exts):
|
||||||
xref, ext = xref_ext
|
xref, ext = xref_ext
|
||||||
future = executor.submit(
|
yield dict(
|
||||||
jbig2enc.convert_single,
|
|
||||||
cwd=fspath(root),
|
cwd=fspath(root),
|
||||||
infile=img_name(root, xref, ext),
|
infile=img_name(root, xref, ext),
|
||||||
outfile=root / f'{prefix}.{n:04d}',
|
outfile=root / f'{prefix}.{n:04d}',
|
||||||
)
|
)
|
||||||
yield future
|
|
||||||
|
def convert_generic(fn, kwargs_dict):
|
||||||
|
return fn(**kwargs_dict)
|
||||||
|
|
||||||
if options.jbig2_page_group_size > 1:
|
if options.jbig2_page_group_size > 1:
|
||||||
jbig2_futures = jbig2_group_futures
|
jbig2_args = jbig2_group_args
|
||||||
|
jbig2_convert = partial(convert_generic, jbig2enc.convert_group)
|
||||||
else:
|
else:
|
||||||
jbig2_futures = jbig2_single_futures
|
jbig2_args = jbig2_single_args
|
||||||
|
jbig2_convert = partial(convert_generic, jbig2enc.convert_single)
|
||||||
|
|
||||||
with concurrent.futures.ThreadPoolExecutor(max_workers=options.jobs) as executor:
|
exec_progress_pool(
|
||||||
futures = jbig2_futures(executor, root, jbig2_groups)
|
use_threads=True,
|
||||||
with tqdm(
|
max_workers=options.jobs,
|
||||||
|
tqdm_kwargs=dict(
|
||||||
total=len(jbig2_groups),
|
total=len(jbig2_groups),
|
||||||
desc="JBIG2",
|
desc="JBIG2",
|
||||||
unit='item',
|
unit='item',
|
||||||
disable=not options.progress_bar,
|
disable=not options.progress_bar,
|
||||||
) as pbar:
|
),
|
||||||
for future in concurrent.futures.as_completed(futures):
|
task=jbig2_convert,
|
||||||
proc = future.result()
|
task_arguments=jbig2_args(root, jbig2_groups),
|
||||||
if proc.stderr:
|
)
|
||||||
log.debug(proc.stderr.decode())
|
|
||||||
pbar.update()
|
|
||||||
|
|
||||||
|
|
||||||
def convert_to_jbig2(pike, jbig2_groups, root, log, options):
|
def convert_to_jbig2(pike, jbig2_groups, root, options):
|
||||||
"""Convert images to JBIG2 and insert into PDF.
|
"""Convert images to JBIG2 and insert into PDF.
|
||||||
|
|
||||||
When the JBIG2 page group size is > 1 we do several JBIG2 images at once
|
When the JBIG2 page group size is > 1 we do several JBIG2 images at once
|
||||||
@@ -318,7 +320,7 @@ def convert_to_jbig2(pike, jbig2_groups, root, log, options):
|
|||||||
and needs no dictionary. Currently this must be lossless JBIG2.
|
and needs no dictionary. Currently this must be lossless JBIG2.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
_produce_jbig2_images(jbig2_groups, root, log, options)
|
_produce_jbig2_images(jbig2_groups, root, options)
|
||||||
|
|
||||||
for group, xref_exts in jbig2_groups.items():
|
for group, xref_exts in jbig2_groups.items():
|
||||||
prefix = f'group{group:08d}'
|
prefix = f'group{group:08d}'
|
||||||
@@ -342,7 +344,7 @@ def convert_to_jbig2(pike, jbig2_groups, root, log, options):
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def transcode_jpegs(pike, jpegs, root, log, options):
|
def transcode_jpegs(pike, jpegs, root, options):
|
||||||
for xref in tqdm(
|
for xref in tqdm(
|
||||||
jpegs, desc="JPEGs", unit='image', disable=not options.progress_bar
|
jpegs, desc="JPEGs", unit='image', disable=not options.progress_bar
|
||||||
):
|
):
|
||||||
@@ -365,37 +367,40 @@ def transcode_jpegs(pike, jpegs, root, log, options):
|
|||||||
im_obj.write(compdata.read(), filter=Name.DCTDecode)
|
im_obj.write(compdata.read(), filter=Name.DCTDecode)
|
||||||
|
|
||||||
|
|
||||||
def transcode_pngs(pike, images, image_name_fn, root, log, options):
|
def transcode_pngs(pike, images, image_name_fn, root, options):
|
||||||
modified = set()
|
modified = set()
|
||||||
if options.optimize >= 2:
|
if options.optimize >= 2:
|
||||||
png_quality = (
|
png_quality = (
|
||||||
max(10, options.png_quality - 10),
|
max(10, options.png_quality - 10),
|
||||||
min(100, options.png_quality + 10),
|
min(100, options.png_quality + 10),
|
||||||
)
|
)
|
||||||
with concurrent.futures.ThreadPoolExecutor(
|
|
||||||
max_workers=options.jobs
|
def pngquant_args():
|
||||||
) as executor:
|
|
||||||
futures = []
|
|
||||||
for xref in images:
|
for xref in images:
|
||||||
log.debug(image_name_fn(root, xref))
|
log.debug(image_name_fn(root, xref))
|
||||||
futures.append(
|
yield (
|
||||||
executor.submit(
|
image_name_fn(root, xref),
|
||||||
pngquant.quantize,
|
png_name(root, xref),
|
||||||
image_name_fn(root, xref),
|
png_quality[0],
|
||||||
png_name(root, xref),
|
png_quality[1],
|
||||||
png_quality[0],
|
|
||||||
png_quality[1],
|
|
||||||
)
|
|
||||||
)
|
)
|
||||||
modified.add(xref)
|
modified.add(xref)
|
||||||
with tqdm(
|
|
||||||
|
def pngquant_fn(args):
|
||||||
|
pngquant.quantize(*args)
|
||||||
|
|
||||||
|
exec_progress_pool(
|
||||||
|
use_threads=True,
|
||||||
|
max_workers=options.jobs,
|
||||||
|
tqdm_kwargs=dict(
|
||||||
desc="PNGs",
|
desc="PNGs",
|
||||||
total=len(futures),
|
total=len(images),
|
||||||
unit='image',
|
unit='image',
|
||||||
disable=not options.progress_bar,
|
disable=not options.progress_bar,
|
||||||
) as pbar:
|
),
|
||||||
for _future in concurrent.futures.as_completed(futures):
|
task=pngquant_fn,
|
||||||
pbar.update()
|
task_arguments=pngquant_args(),
|
||||||
|
)
|
||||||
|
|
||||||
for xref in modified:
|
for xref in modified:
|
||||||
im_obj = pike.get_object(xref, 0)
|
im_obj = pike.get_object(xref, 0)
|
||||||
@@ -421,12 +426,12 @@ def transcode_pngs(pike, images, image_name_fn, root, log, options):
|
|||||||
)
|
)
|
||||||
continue
|
continue
|
||||||
if compdata.type == leptonica.lept.L_FLATE_ENCODE:
|
if compdata.type == leptonica.lept.L_FLATE_ENCODE:
|
||||||
rewrite_png(pike, im_obj, compdata, log)
|
rewrite_png(pike, im_obj, compdata)
|
||||||
elif compdata.type == leptonica.lept.L_G4_ENCODE:
|
elif compdata.type == leptonica.lept.L_G4_ENCODE:
|
||||||
rewrite_png_as_g4(pike, im_obj, compdata, log)
|
rewrite_png_as_g4(pike, im_obj, compdata)
|
||||||
|
|
||||||
|
|
||||||
def rewrite_png_as_g4(pike, im_obj, compdata, log):
|
def rewrite_png_as_g4(pike, im_obj, compdata):
|
||||||
im_obj.BitsPerComponent = 1
|
im_obj.BitsPerComponent = 1
|
||||||
im_obj.Width = compdata.w
|
im_obj.Width = compdata.w
|
||||||
im_obj.Height = compdata.h
|
im_obj.Height = compdata.h
|
||||||
@@ -446,7 +451,7 @@ def rewrite_png_as_g4(pike, im_obj, compdata, log):
|
|||||||
return
|
return
|
||||||
|
|
||||||
|
|
||||||
def rewrite_png(pike, im_obj, compdata, log):
|
def rewrite_png(pike, im_obj, compdata):
|
||||||
# When a PNG is inserted into a PDF, we more or less copy the IDAT section from
|
# When a PNG is inserted into a PDF, we more or less copy the IDAT section from
|
||||||
# the PDF and transfer the rest of the PNG headers to PDF image metadata.
|
# the PDF and transfer the rest of the PNG headers to PDF image metadata.
|
||||||
# One thing we have to do is tell the PDF reader whether a predictor was used
|
# One thing we have to do is tell the PDF reader whether a predictor was used
|
||||||
@@ -500,7 +505,6 @@ def rewrite_png(pike, im_obj, compdata, log):
|
|||||||
|
|
||||||
|
|
||||||
def optimize(input_file, output_file, context, save_settings):
|
def optimize(input_file, output_file, context, save_settings):
|
||||||
log = context.log
|
|
||||||
options = context.options
|
options = context.options
|
||||||
if options.optimize == 0:
|
if options.optimize == 0:
|
||||||
safe_symlink(input_file, output_file)
|
safe_symlink(input_file, output_file)
|
||||||
@@ -517,15 +521,15 @@ def optimize(input_file, output_file, context, save_settings):
|
|||||||
root = Path(output_file).parent / 'images'
|
root = Path(output_file).parent / 'images'
|
||||||
root.mkdir(exist_ok=True)
|
root.mkdir(exist_ok=True)
|
||||||
|
|
||||||
jpegs, pngs = extract_images_generic(pike, root, log, options)
|
jpegs, pngs = extract_images_generic(pike, root, options)
|
||||||
transcode_jpegs(pike, jpegs, root, log, options)
|
transcode_jpegs(pike, jpegs, root, options)
|
||||||
# if options.optimize >= 2:
|
# if options.optimize >= 2:
|
||||||
# Try pngifying the jpegs
|
# Try pngifying the jpegs
|
||||||
# transcode_pngs(pike, jpegs, jpg_name, root, log, options)
|
# transcode_pngs(pike, jpegs, jpg_name, root, options)
|
||||||
transcode_pngs(pike, pngs, png_name, root, log, options)
|
transcode_pngs(pike, pngs, png_name, root, options)
|
||||||
|
|
||||||
jbig2_groups = extract_images_jbig2(pike, root, log, options)
|
jbig2_groups = extract_images_jbig2(pike, root, options)
|
||||||
convert_to_jbig2(pike, jbig2_groups, root, log, options)
|
convert_to_jbig2(pike, jbig2_groups, root, options)
|
||||||
|
|
||||||
target_file = Path(output_file).with_suffix('.opt.pdf')
|
target_file = Path(output_file).with_suffix('.opt.pdf')
|
||||||
pike.remove_unreferenced_resources()
|
pike.remove_unreferenced_resources()
|
||||||
@@ -553,8 +557,8 @@ def optimize(input_file, output_file, context, save_settings):
|
|||||||
|
|
||||||
|
|
||||||
def main(infile, outfile, level, jobs=1):
|
def main(infile, outfile, level, jobs=1):
|
||||||
from tempfile import TemporaryDirectory
|
from tempfile import TemporaryDirectory # pylint: disable=import-outside-toplevel
|
||||||
from shutil import copy
|
from shutil import copy # pylint: disable=import-outside-toplevel
|
||||||
|
|
||||||
class OptimizeOptions:
|
class OptimizeOptions:
|
||||||
"""Emulate ocrmypdf's options"""
|
"""Emulate ocrmypdf's options"""
|
||||||
@@ -582,7 +586,7 @@ def main(infile, outfile, level, jobs=1):
|
|||||||
)
|
)
|
||||||
|
|
||||||
with TemporaryDirectory() as td:
|
with TemporaryDirectory() as td:
|
||||||
context = PDFContext(options, td, infile, None)
|
context = PdfContext(options, td, infile, None, None)
|
||||||
tmpout = Path(td) / 'out.pdf'
|
tmpout = Path(td) / 'out.pdf'
|
||||||
optimize(
|
optimize(
|
||||||
infile,
|
infile,
|
||||||
|
|||||||
@@ -16,4 +16,4 @@
|
|||||||
# You should have received a copy of the GNU General Public License
|
# You should have received a copy of the GNU General Public License
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
from .info import Colorspace, Encoding, PdfInfo
|
from ocrmypdf.pdfinfo.info import Colorspace, Encoding, PdfInfo
|
||||||
|
|||||||
@@ -1,102 +0,0 @@
|
|||||||
# © 2018 James R. Barlow: github.com/jbarlow83
|
|
||||||
#
|
|
||||||
# This file is part of OCRmyPDF.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is free software: you can redistribute it and/or modify
|
|
||||||
# it under the terms of the GNU General Public License as published by
|
|
||||||
# the Free Software Foundation, either version 3 of the License, or
|
|
||||||
# (at your option) any later version.
|
|
||||||
#
|
|
||||||
# OCRmyPDF is distributed in the hope that it will be useful,
|
|
||||||
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
||||||
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
||||||
# GNU General Public License for more details.
|
|
||||||
#
|
|
||||||
# You should have received a copy of the GNU General Public License
|
|
||||||
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import re
|
|
||||||
import xml.etree.ElementTree as ET
|
|
||||||
|
|
||||||
from ..exec import ghostscript
|
|
||||||
|
|
||||||
gslog = logging.getLogger()
|
|
||||||
|
|
||||||
# Forgive me for I have sinned
|
|
||||||
# I am using regular expressions to parse XML. However the XML in this case,
|
|
||||||
# generated by Ghostscript, is self-consistent enough to be parseable.
|
|
||||||
regex_remove_char_tags = re.compile(
|
|
||||||
br"""
|
|
||||||
<char\b
|
|
||||||
(?: [^>] # anything single character but >
|
|
||||||
| \">\" # special case: trap ">"
|
|
||||||
)*
|
|
||||||
/> # terminate with '/>'
|
|
||||||
""",
|
|
||||||
re.VERBOSE,
|
|
||||||
)
|
|
||||||
|
|
||||||
|
|
||||||
def page_get_textblocks(infile, pageno, xmltext, height):
|
|
||||||
"""Get text boxes out of Ghostscript txtwrite xml"""
|
|
||||||
|
|
||||||
root = xmltext
|
|
||||||
if not hasattr(xmltext, 'findall'):
|
|
||||||
return []
|
|
||||||
|
|
||||||
def blocks():
|
|
||||||
for span in root.findall('.//span'):
|
|
||||||
bbox_str = span.attrib['bbox']
|
|
||||||
font_size = span.attrib['size']
|
|
||||||
pts = [int(pt) for pt in bbox_str.split()]
|
|
||||||
pts[1] = pts[1] - int(float(font_size) + 0.5)
|
|
||||||
bbox_topdown = tuple(pts)
|
|
||||||
bb = bbox_topdown
|
|
||||||
bbox_bottomup = (bb[0], height - bb[3], bb[2], height - bb[1])
|
|
||||||
yield bbox_bottomup
|
|
||||||
|
|
||||||
def joined_blocks():
|
|
||||||
prev = None
|
|
||||||
for bbox in blocks():
|
|
||||||
if prev is None:
|
|
||||||
prev = bbox
|
|
||||||
if bbox[1] == prev[1] and bbox[3] == prev[3]:
|
|
||||||
gap = prev[2] - bbox[0]
|
|
||||||
height = abs(bbox[3] - bbox[1])
|
|
||||||
if gap < height:
|
|
||||||
# Join boxes
|
|
||||||
prev = (prev[0], prev[1], bbox[2], bbox[3])
|
|
||||||
continue
|
|
||||||
# yield previously joined bboxes and start anew
|
|
||||||
yield prev
|
|
||||||
prev = bbox
|
|
||||||
if prev is not None:
|
|
||||||
yield prev
|
|
||||||
|
|
||||||
return [block for block in joined_blocks()]
|
|
||||||
|
|
||||||
|
|
||||||
def extract_text_xml(infile, pdf, pageno=None, log=gslog):
|
|
||||||
existing_text = ghostscript.extract_text(infile, pageno=None)
|
|
||||||
existing_text = regex_remove_char_tags.sub(b' ', existing_text)
|
|
||||||
|
|
||||||
try:
|
|
||||||
root = ET.fromstringlist([b'<document>\n', existing_text, b'</document>\n'])
|
|
||||||
page_xml = root.findall('page')
|
|
||||||
except ET.ParseError as e:
|
|
||||||
log.error(
|
|
||||||
"An error occurred while attempting to retrieve existing text in "
|
|
||||||
"the input file. Will attempt to continue assuming that there is "
|
|
||||||
"no existing text in the file. The error was:"
|
|
||||||
)
|
|
||||||
log.error(e)
|
|
||||||
page_xml = [None] * len(pdf.pages)
|
|
||||||
|
|
||||||
page_count_difference = len(pdf.pages) - len(page_xml)
|
|
||||||
if page_count_difference != 0:
|
|
||||||
log.error("The number of pages in the input file is inconsistent.")
|
|
||||||
log.error(f"Expected {len(pdf.pages)}, txtwrite says {len(page_xml)}")
|
|
||||||
if page_count_difference > 0:
|
|
||||||
page_xml.extend([None] * page_count_difference)
|
|
||||||
return page_xml
|
|
||||||
+152
-103
@@ -21,18 +21,19 @@ import re
|
|||||||
from collections import defaultdict, namedtuple
|
from collections import defaultdict, namedtuple
|
||||||
from decimal import Decimal
|
from decimal import Decimal
|
||||||
from enum import Enum
|
from enum import Enum
|
||||||
|
from functools import partial
|
||||||
from math import hypot, isclose
|
from math import hypot, isclose
|
||||||
from os import PathLike, fspath
|
from os import PathLike
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from typing import Any, Dict, List
|
||||||
from warnings import warn
|
from warnings import warn
|
||||||
|
|
||||||
import pikepdf
|
import pikepdf
|
||||||
from pikepdf import PdfMatrix
|
from pikepdf import PdfMatrix
|
||||||
from tqdm import tqdm
|
|
||||||
|
|
||||||
|
from ocrmypdf._concurrent import exec_progress_pool
|
||||||
from ocrmypdf.exceptions import EncryptedPdfError
|
from ocrmypdf.exceptions import EncryptedPdfError
|
||||||
from ocrmypdf.exec import ghostscript
|
from ocrmypdf.helpers import Resolution, available_cpu_count
|
||||||
from ocrmypdf.pdfinfo import ghosttext
|
|
||||||
from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes
|
from ocrmypdf.pdfinfo.layout import get_page_analysis, get_text_boxes
|
||||||
|
|
||||||
logger = logging.getLogger()
|
logger = logging.getLogger()
|
||||||
@@ -98,15 +99,19 @@ XobjectSettings = namedtuple('XobjectSettings', ['name', 'shorthand', 'stack_dep
|
|||||||
InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth'])
|
InlineSettings = namedtuple('InlineSettings', ['iimage', 'shorthand', 'stack_depth'])
|
||||||
|
|
||||||
ContentsInfo = namedtuple(
|
ContentsInfo = namedtuple(
|
||||||
'ContentsInfo', ['xobject_settings', 'inline_images', 'found_vector', 'name_index']
|
'ContentsInfo',
|
||||||
|
['xobject_settings', 'inline_images', 'found_vector', 'found_text', 'name_index'],
|
||||||
)
|
)
|
||||||
|
|
||||||
TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt'])
|
TextboxInfo = namedtuple('TextboxInfo', ['bbox', 'is_visible', 'is_corrupt'])
|
||||||
|
|
||||||
|
|
||||||
class VectorInfo:
|
class VectorMarker:
|
||||||
def __init__(self):
|
pass
|
||||||
pass
|
|
||||||
|
|
||||||
|
class TextMarker:
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
def _normalize_stack(graphobjs):
|
def _normalize_stack(graphobjs):
|
||||||
@@ -153,9 +158,11 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
|||||||
inline_images = []
|
inline_images = []
|
||||||
name_index = defaultdict(lambda: [])
|
name_index = defaultdict(lambda: [])
|
||||||
found_vector = False
|
found_vector = False
|
||||||
|
found_text = False
|
||||||
vector_ops = set('S s f F f* B B* b b*'.split())
|
vector_ops = set('S s f F f* B B* b b*'.split())
|
||||||
|
text_showing_ops = set("""TJ Tj " '""".split())
|
||||||
image_ops = set('BI ID EI q Q Do cm'.split())
|
image_ops = set('BI ID EI q Q Do cm'.split())
|
||||||
operator_whitelist = ' '.join(vector_ops | image_ops)
|
operator_whitelist = ' '.join(vector_ops | text_showing_ops | image_ops)
|
||||||
|
|
||||||
for n, graphobj in enumerate(
|
for n, graphobj in enumerate(
|
||||||
_normalize_stack(
|
_normalize_stack(
|
||||||
@@ -195,11 +202,14 @@ def _interpret_contents(contentstream, initial_shorthand=UNIT_SQUARE):
|
|||||||
inline_images.append(inline)
|
inline_images.append(inline)
|
||||||
elif operator in vector_ops:
|
elif operator in vector_ops:
|
||||||
found_vector = True
|
found_vector = True
|
||||||
|
elif operator in text_showing_ops:
|
||||||
|
found_text = True
|
||||||
|
|
||||||
return ContentsInfo(
|
return ContentsInfo(
|
||||||
xobject_settings=xobject_settings,
|
xobject_settings=xobject_settings,
|
||||||
inline_images=inline_images,
|
inline_images=inline_images,
|
||||||
found_vector=found_vector,
|
found_vector=found_vector,
|
||||||
|
found_text=found_text,
|
||||||
name_index=name_index,
|
name_index=name_index,
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -265,7 +275,7 @@ def _get_dpi(ctm_shorthand, image_size):
|
|||||||
dpi_w = scale_w * 72.0
|
dpi_w = scale_w * 72.0
|
||||||
dpi_h = scale_h * 72.0
|
dpi_h = scale_h * 72.0
|
||||||
|
|
||||||
return dpi_w, dpi_h
|
return Resolution(dpi_w, dpi_h)
|
||||||
|
|
||||||
|
|
||||||
class ImageInfo:
|
class ImageInfo:
|
||||||
@@ -356,12 +366,8 @@ class ImageInfo:
|
|||||||
return self._enc
|
return self._enc
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def xres(self):
|
def dpi(self):
|
||||||
return _get_dpi(self._shorthand, (self._width, self._height))[0]
|
return _get_dpi(self._shorthand, (self._width, self._height))
|
||||||
|
|
||||||
@property
|
|
||||||
def yres(self):
|
|
||||||
return _get_dpi(self._shorthand, (self._width, self._height))[1]
|
|
||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
class_locals = {
|
class_locals = {
|
||||||
@@ -371,7 +377,7 @@ class ImageInfo:
|
|||||||
}
|
}
|
||||||
return (
|
return (
|
||||||
"<ImageInfo '{name}' {type_} {width}x{height} {color} "
|
"<ImageInfo '{name}' {type_} {width}x{height} {color} "
|
||||||
"{comp} {bpc} {enc} {xres}x{yres}>"
|
"{comp} {bpc} {enc} {dpi}>"
|
||||||
).format(**class_locals)
|
).format(**class_locals)
|
||||||
|
|
||||||
|
|
||||||
@@ -509,7 +515,9 @@ def _process_content_streams(*, pdf, container, shorthand=None):
|
|||||||
contentsinfo = _interpret_contents(container, initial_shorthand)
|
contentsinfo = _interpret_contents(container, initial_shorthand)
|
||||||
|
|
||||||
if contentsinfo.found_vector:
|
if contentsinfo.found_vector:
|
||||||
yield VectorInfo()
|
yield VectorMarker()
|
||||||
|
if contentsinfo.found_text:
|
||||||
|
yield TextMarker()
|
||||||
yield from _find_inline_images(contentsinfo)
|
yield from _find_inline_images(contentsinfo)
|
||||||
yield from _find_regular_images(container, contentsinfo)
|
yield from _find_regular_images(container, contentsinfo)
|
||||||
yield from _find_form_xobject_images(pdf, container, contentsinfo)
|
yield from _find_form_xobject_images(pdf, container, contentsinfo)
|
||||||
@@ -558,8 +566,10 @@ def simplify_textboxes(miner, textbox_getter):
|
|||||||
yield TextboxInfo(box.bbox, visible, corrupt)
|
yield TextboxInfo(box.bbox, visible, corrupt)
|
||||||
|
|
||||||
|
|
||||||
def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str):
|
def _pdf_get_pageinfo(
|
||||||
pageinfo = {}
|
pdf, pageno: int, infile: PathLike, check_pages, detailed_analysis
|
||||||
|
):
|
||||||
|
pageinfo: Dict[str, Any] = {}
|
||||||
pageinfo['pageno'] = pageno
|
pageinfo['pageno'] = pageno
|
||||||
pageinfo['images'] = []
|
pageinfo['images'] = []
|
||||||
|
|
||||||
@@ -568,18 +578,18 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str):
|
|||||||
width_pt = mediabox[2] - mediabox[0]
|
width_pt = mediabox[2] - mediabox[0]
|
||||||
height_pt = mediabox[3] - mediabox[1]
|
height_pt = mediabox[3] - mediabox[1]
|
||||||
|
|
||||||
if xmltext is not None:
|
check_this_page = pageno in check_pages
|
||||||
bboxes = ghosttext.page_get_textblocks(
|
|
||||||
fspath(infile), pageno, xmltext=xmltext, height=height_pt
|
if check_this_page and detailed_analysis:
|
||||||
)
|
|
||||||
pageinfo['bboxes'] = bboxes
|
|
||||||
else:
|
|
||||||
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
|
pscript5_mode = str(pdf.docinfo.get('/Creator')).startswith('PScript5')
|
||||||
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
miner = get_page_analysis(infile, pageno, pscript5_mode)
|
||||||
pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes))
|
pageinfo['textboxes'] = list(simplify_textboxes(miner, get_text_boxes))
|
||||||
bboxes = (box.bbox for box in pageinfo['textboxes'])
|
bboxes = (box.bbox for box in pageinfo['textboxes'])
|
||||||
|
|
||||||
pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt)
|
pageinfo['has_text'] = _page_has_text(bboxes, width_pt, height_pt)
|
||||||
|
else:
|
||||||
|
pageinfo['textboxes'] = []
|
||||||
|
pageinfo['has_text'] = None # i.e. "no information"
|
||||||
|
|
||||||
userunit = page.get('/UserUnit', Decimal(1.0))
|
userunit = page.get('/UserUnit', Decimal(1.0))
|
||||||
if not isinstance(userunit, Decimal):
|
if not isinstance(userunit, Decimal):
|
||||||
@@ -594,62 +604,96 @@ def _pdf_get_pageinfo(pdf, pageno: int, infile: PathLike, xmltext: str):
|
|||||||
pageinfo['rotate'] = 0
|
pageinfo['rotate'] = 0
|
||||||
|
|
||||||
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
|
userunit_shorthand = (userunit, 0, 0, userunit, 0, 0)
|
||||||
contentsinfo = [
|
|
||||||
ci
|
if check_this_page:
|
||||||
|
pageinfo['has_vector'] = False
|
||||||
|
pageinfo['has_text'] = False
|
||||||
|
pageinfo['images'] = []
|
||||||
for ci in _process_content_streams(
|
for ci in _process_content_streams(
|
||||||
pdf=pdf, container=page, shorthand=userunit_shorthand
|
pdf=pdf, container=page, shorthand=userunit_shorthand
|
||||||
)
|
):
|
||||||
]
|
if isinstance(ci, VectorMarker):
|
||||||
|
pageinfo['has_vector'] = True
|
||||||
|
elif isinstance(ci, TextMarker):
|
||||||
|
pageinfo['has_text'] = True
|
||||||
|
elif isinstance(ci, ImageInfo):
|
||||||
|
pageinfo['images'].append(ci)
|
||||||
|
else:
|
||||||
|
raise NotImplementedError()
|
||||||
|
else:
|
||||||
|
pageinfo['has_vector'] = None # i.e. "no information"
|
||||||
|
pageinfo['has_text'] = None
|
||||||
|
pageinfo['images'] = None
|
||||||
|
|
||||||
pageinfo['has_vector'] = False
|
|
||||||
if any(isinstance(ci, VectorInfo) for ci in contentsinfo):
|
|
||||||
pageinfo['has_vector'] = True
|
|
||||||
|
|
||||||
pageinfo['images'] = [im for im in contentsinfo if isinstance(im, ImageInfo)]
|
|
||||||
if pageinfo['images']:
|
if pageinfo['images']:
|
||||||
xres = Decimal(max(image.xres for image in pageinfo['images']))
|
dpi = Resolution(0.0, 0.0).take_max(image.dpi for image in pageinfo['images'])
|
||||||
yres = Decimal(max(image.yres for image in pageinfo['images']))
|
pageinfo['dpi'] = dpi
|
||||||
pageinfo['xres'], pageinfo['yres'] = xres, yres
|
pageinfo['width_pixels'] = int(round(dpi.x * float(pageinfo['width_inches'])))
|
||||||
pageinfo['width_pixels'] = int(round(xres * pageinfo['width_inches']))
|
pageinfo['height_pixels'] = int(round(dpi.y * float(pageinfo['height_inches'])))
|
||||||
pageinfo['height_pixels'] = int(round(yres * pageinfo['height_inches']))
|
|
||||||
|
|
||||||
return pageinfo
|
return pageinfo
|
||||||
|
|
||||||
|
|
||||||
def _pdf_get_all_pageinfo(infile, detailed_analysis=False, log=None, progbar=False):
|
worker_pdf = None
|
||||||
pdf = pikepdf.open(infile) # Do not close in this function
|
|
||||||
try:
|
|
||||||
if pdf.is_encrypted:
|
|
||||||
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
|
||||||
if detailed_analysis:
|
|
||||||
pages_xml = None
|
|
||||||
else:
|
|
||||||
pages_xml = ghosttext.extract_text_xml(infile, pdf, pageno=None, log=log)
|
|
||||||
|
|
||||||
pages = []
|
|
||||||
for n, _ in tqdm(
|
|
||||||
enumerate(pdf.pages),
|
|
||||||
total=len(pdf.pages),
|
|
||||||
desc="Scan",
|
|
||||||
unit='page',
|
|
||||||
disable=not progbar,
|
|
||||||
):
|
|
||||||
page_xml = pages_xml[n] if pages_xml else None
|
|
||||||
page = PageInfo(pdf, n, infile, page_xml, detailed_analysis)
|
|
||||||
pages.append(page)
|
|
||||||
except Exception:
|
|
||||||
pdf.close()
|
|
||||||
raise
|
|
||||||
|
|
||||||
return pages, pdf
|
def _pdf_pageinfo_sync_init(infile):
|
||||||
|
global worker_pdf # pylint: disable=global-statement
|
||||||
|
worker_pdf = pikepdf.open(infile)
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_pageinfo_sync(args):
|
||||||
|
global worker_pdf # pylint: disable=global-statement
|
||||||
|
pageno, infile, check_pages, detailed_analysis = args
|
||||||
|
page = PageInfo(worker_pdf, pageno, infile, check_pages, detailed_analysis)
|
||||||
|
return page
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_pageinfo_concurrent(
|
||||||
|
pdf, infile, progbar, max_workers, check_pages, detailed_analysis=False
|
||||||
|
):
|
||||||
|
pages = [None] * len(pdf.pages)
|
||||||
|
|
||||||
|
def update_pageinfo(result, pbar):
|
||||||
|
page = result
|
||||||
|
pages[page.pageno] = page
|
||||||
|
pbar.update()
|
||||||
|
|
||||||
|
if max_workers is None:
|
||||||
|
max_workers = available_cpu_count()
|
||||||
|
|
||||||
|
total = len(pdf.pages)
|
||||||
|
contexts = ((n, infile, check_pages, detailed_analysis) for n in range(total))
|
||||||
|
|
||||||
|
use_threads = False # No performance gain if threaded due to GIL
|
||||||
|
n_workers = min(1 + len(pages) // 4, max_workers)
|
||||||
|
if n_workers == 1:
|
||||||
|
# But if we decided on only one worker, there is no point in using
|
||||||
|
# a separate process.
|
||||||
|
use_threads = True
|
||||||
|
|
||||||
|
exec_progress_pool(
|
||||||
|
use_threads=use_threads,
|
||||||
|
max_workers=n_workers,
|
||||||
|
tqdm_kwargs=dict(
|
||||||
|
total=total, desc="Scanning contents", unit='page', disable=not progbar
|
||||||
|
),
|
||||||
|
task_initializer=partial(_pdf_pageinfo_sync_init, infile),
|
||||||
|
task=_pdf_pageinfo_sync,
|
||||||
|
task_arguments=contexts,
|
||||||
|
task_finished=update_pageinfo,
|
||||||
|
)
|
||||||
|
return pages
|
||||||
|
|
||||||
|
|
||||||
class PageInfo:
|
class PageInfo:
|
||||||
def __init__(self, pdf, pageno, infile, xmltext, detailed_analysis=False):
|
def __init__(self, pdf, pageno, infile, check_pages, detailed_analysis=False):
|
||||||
self._pageno = pageno
|
self._pageno = pageno
|
||||||
self._infile = infile
|
self._infile = infile
|
||||||
self._pageinfo = _pdf_get_pageinfo(pdf, pageno, infile, xmltext)
|
|
||||||
self._detailed_analysis = detailed_analysis
|
self._detailed_analysis = detailed_analysis
|
||||||
|
self._pageinfo = _pdf_get_pageinfo(
|
||||||
|
pdf, pageno, infile, check_pages, detailed_analysis
|
||||||
|
)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def pageno(self):
|
def pageno(self):
|
||||||
@@ -679,11 +723,11 @@ class PageInfo:
|
|||||||
|
|
||||||
@property
|
@property
|
||||||
def width_pixels(self):
|
def width_pixels(self):
|
||||||
return int(round(self.width_inches * self.xres))
|
return int(round(float(self.width_inches) * self.dpi.x))
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def height_pixels(self):
|
def height_pixels(self):
|
||||||
return int(round(self.height_inches * self.yres))
|
return int(round(float(self.height_inches) * self.dpi.y))
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def rotation(self):
|
def rotation(self):
|
||||||
@@ -713,7 +757,7 @@ class PageInfo:
|
|||||||
|
|
||||||
if 'textboxes' not in self._pageinfo:
|
if 'textboxes' not in self._pageinfo:
|
||||||
if visible is not None and corrupt is not None:
|
if visible is not None and corrupt is not None:
|
||||||
raise NotImplementedError('Ghostscript textboxes cannot be classified')
|
raise NotImplementedError('Incomplete information on textboxes')
|
||||||
return self._pageinfo['bboxes']
|
return self._pageinfo['bboxes']
|
||||||
|
|
||||||
return (
|
return (
|
||||||
@@ -723,12 +767,8 @@ class PageInfo:
|
|||||||
)
|
)
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def xres(self):
|
def dpi(self):
|
||||||
return self._pageinfo.get('xres', None)
|
return self._pageinfo.get('dpi', Resolution(0.0, 0.0))
|
||||||
|
|
||||||
@property
|
|
||||||
def yres(self):
|
|
||||||
return self._pageinfo.get('yres', None)
|
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def userunit(self):
|
def userunit(self):
|
||||||
@@ -743,36 +783,45 @@ class PageInfo:
|
|||||||
|
|
||||||
def __repr__(self):
|
def __repr__(self):
|
||||||
return (
|
return (
|
||||||
'<PageInfo ' 'pageno={} {}"x{}" rotation={} res={}x{} has_text={}>'
|
f'<PageInfo '
|
||||||
).format(
|
f'pageno={self.pageno} {self.width_inches}"x{self.height_inches}" '
|
||||||
self.pageno,
|
f'rotation={self.rotation} dpi={self.dpi} has_text={self.has_text}>'
|
||||||
self.width_inches,
|
|
||||||
self.height_inches,
|
|
||||||
self.rotation,
|
|
||||||
self.xres,
|
|
||||||
self.yres,
|
|
||||||
self.has_text,
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
class PdfInfo:
|
class PdfInfo:
|
||||||
"""Get summary information about a PDF"""
|
"""Get summary information about a PDF"""
|
||||||
|
|
||||||
def __init__(self, infile, detailed_page_analysis=False, log=logger, progbar=False):
|
def __init__(
|
||||||
|
self,
|
||||||
|
infile,
|
||||||
|
detailed_analysis=False,
|
||||||
|
progbar=False,
|
||||||
|
max_workers=None,
|
||||||
|
check_pages=None,
|
||||||
|
):
|
||||||
self._infile = infile
|
self._infile = infile
|
||||||
if ghostscript.version() in ('9.52',):
|
if check_pages is None:
|
||||||
detailed_page_analysis = True # txtwrite doesn't work in these versions
|
check_pages = range(0, 1_000_000_000)
|
||||||
self._pages, pdf = _pdf_get_all_pageinfo(
|
|
||||||
infile, detailed_page_analysis, log=log, progbar=progbar
|
with pikepdf.open(infile) as pdf:
|
||||||
)
|
if pdf.is_encrypted:
|
||||||
self._needs_rendering = pdf.root.get('/NeedsRendering', False)
|
raise EncryptedPdfError() # Triggered by encryption with empty passwd
|
||||||
self._has_acroform = False
|
self._pages = _pdf_pageinfo_concurrent(
|
||||||
if '/AcroForm' in pdf.root:
|
pdf,
|
||||||
if len(pdf.root.AcroForm.get('/Fields', [])) > 0:
|
infile,
|
||||||
self._has_acroform = True
|
progbar,
|
||||||
elif '/XFA' in pdf.root.AcroForm:
|
max_workers,
|
||||||
self._has_acroform = True
|
check_pages=check_pages,
|
||||||
pdf.close()
|
detailed_analysis=detailed_analysis,
|
||||||
|
)
|
||||||
|
self._needs_rendering = pdf.root.get('/NeedsRendering', False)
|
||||||
|
self._has_acroform = False
|
||||||
|
if '/AcroForm' in pdf.root:
|
||||||
|
if len(pdf.root.AcroForm.get('/Fields', [])) > 0:
|
||||||
|
self._has_acroform = True
|
||||||
|
elif '/XFA' in pdf.root.AcroForm:
|
||||||
|
self._has_acroform = True
|
||||||
|
|
||||||
@property
|
@property
|
||||||
def pages(self):
|
def pages(self):
|
||||||
@@ -812,16 +861,16 @@ class PdfInfo:
|
|||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
import argparse
|
import argparse # pylint: disable=import-outside-toplevel
|
||||||
|
from pprint import pprint # pylint: disable=import-outside-toplevel
|
||||||
|
|
||||||
parser = argparse.ArgumentParser()
|
parser = argparse.ArgumentParser()
|
||||||
parser.add_argument('infile')
|
parser.add_argument('infile')
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
pagesinfo, pdfinfo = _pdf_get_all_pageinfo(args.infile)
|
pdfinfo = PdfInfo(args.infile)
|
||||||
from pprint import pprint
|
|
||||||
|
|
||||||
pprint(pdfinfo)
|
pprint(pdfinfo)
|
||||||
for page in pagesinfo:
|
for page in pdfinfo.pages:
|
||||||
pprint(page)
|
pprint(page)
|
||||||
for im in page.images:
|
for im in page.images:
|
||||||
pprint(im)
|
pprint(im)
|
||||||
|
|||||||
@@ -25,65 +25,16 @@ import pdfminer.encodingdb
|
|||||||
import pdfminer.pdfdevice
|
import pdfminer.pdfdevice
|
||||||
import pdfminer.pdfinterp
|
import pdfminer.pdfinterp
|
||||||
from pdfminer.converter import PDFLayoutAnalyzer
|
from pdfminer.converter import PDFLayoutAnalyzer
|
||||||
from pdfminer.glyphlist import glyphname2unicode
|
|
||||||
from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox
|
from pdfminer.layout import LAParams, LTChar, LTPage, LTTextBox
|
||||||
from pdfminer.pdfdocument import PDFTextExtractionNotAllowed
|
from pdfminer.pdfdocument import PDFTextExtractionNotAllowed
|
||||||
from pdfminer.pdffont import PDFFont, PDFSimpleFont, PDFUnicodeNotDefined
|
from pdfminer.pdffont import PDFSimpleFont, PDFUnicodeNotDefined
|
||||||
from pdfminer.pdfpage import PDFPage
|
from pdfminer.pdfpage import PDFPage
|
||||||
from pdfminer.utils import bbox2str, matrix2str
|
from pdfminer.utils import bbox2str, matrix2str
|
||||||
|
|
||||||
from ..exceptions import EncryptedPdfError
|
from ocrmypdf.exceptions import EncryptedPdfError
|
||||||
|
|
||||||
STRIP_NAME = re.compile(r'[0-9]+')
|
STRIP_NAME = re.compile(r'[0-9]+')
|
||||||
|
|
||||||
#
|
|
||||||
# pdfminer 20181108 patches
|
|
||||||
#
|
|
||||||
|
|
||||||
if pdfminer.__version__ == '20181108':
|
|
||||||
|
|
||||||
def name2unicode(name):
|
|
||||||
"""Fix pdfminer's name2unicode function
|
|
||||||
|
|
||||||
Font cids that are mapped to names of the form /g123 seem to be, by convention
|
|
||||||
characters with no corresponding Unicode entry. These can be subsetted fonts
|
|
||||||
or symbolic fonts. There seems to be no way to map /g123 fonts to Unicode,
|
|
||||||
barring a ToUnicode data structure.
|
|
||||||
"""
|
|
||||||
if name in glyphname2unicode:
|
|
||||||
return glyphname2unicode[name]
|
|
||||||
if name.startswith('g') or name.startswith('a'):
|
|
||||||
raise KeyError(name)
|
|
||||||
if name.startswith('uni'):
|
|
||||||
try:
|
|
||||||
return chr(int(name[3:], 16))
|
|
||||||
except ValueError: # Not hexadecimal
|
|
||||||
raise KeyError(name)
|
|
||||||
m = STRIP_NAME.search(name)
|
|
||||||
if not m:
|
|
||||||
raise KeyError(name)
|
|
||||||
return chr(int(m.group(0)))
|
|
||||||
|
|
||||||
pdfminer.encodingdb.name2unicode = name2unicode
|
|
||||||
|
|
||||||
original_PDFFont_init = PDFFont.__init__
|
|
||||||
|
|
||||||
def PDFFont__init__(self, descriptor, widths, default_width=None):
|
|
||||||
original_PDFFont_init(self, descriptor, widths, default_width)
|
|
||||||
# PDF spec says descent should be negative
|
|
||||||
# A font with a positive descent implies it floats entirely above the
|
|
||||||
# baseline, i.e. it's not really a baseline anymore. I have fonts that
|
|
||||||
# claim a positive descent, but treating descent as positive always seems
|
|
||||||
# to misposition text.
|
|
||||||
if self.descent > 0:
|
|
||||||
self.descent = -self.descent
|
|
||||||
|
|
||||||
PDFFont.__init__ = PDFFont__init__
|
|
||||||
|
|
||||||
#
|
|
||||||
# end of pdfminer 20181108 patches
|
|
||||||
#
|
|
||||||
|
|
||||||
|
|
||||||
original_PDFSimpleFont_init = PDFSimpleFont.__init__
|
original_PDFSimpleFont_init = PDFSimpleFont.__init__
|
||||||
|
|
||||||
@@ -269,7 +220,9 @@ class TextPositionTracker(PDFLayoutAnalyzer):
|
|||||||
|
|
||||||
def get_page_analysis(infile, pageno, pscript5_mode):
|
def get_page_analysis(infile, pageno, pscript5_mode):
|
||||||
rman = pdfminer.pdfinterp.PDFResourceManager(caching=True)
|
rman = pdfminer.pdfinterp.PDFResourceManager(caching=True)
|
||||||
dev = TextPositionTracker(rman, laparams=LAParams())
|
dev = TextPositionTracker(
|
||||||
|
rman, laparams=LAParams(all_texts=True, detect_vertical=True)
|
||||||
|
)
|
||||||
interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev)
|
interp = pdfminer.pdfinterp.PDFPageInterpreter(rman, dev)
|
||||||
|
|
||||||
if pscript5_mode:
|
if pscript5_mode:
|
||||||
|
|||||||
@@ -0,0 +1,252 @@
|
|||||||
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
|
#
|
||||||
|
# This file is part of OCRmyPDF.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is free software: you can redistribute it and/or modify
|
||||||
|
# it under the terms of the GNU General Public License as published by
|
||||||
|
# the Free Software Foundation, either version 3 of the License, or
|
||||||
|
# (at your option) any later version.
|
||||||
|
#
|
||||||
|
# OCRmyPDF is distributed in the hope that it will be useful,
|
||||||
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
# GNU General Public License for more details.
|
||||||
|
#
|
||||||
|
# You should have received a copy of the GNU General Public License
|
||||||
|
# along with OCRmyPDF. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
from abc import ABC, abstractmethod, abstractstaticmethod
|
||||||
|
from argparse import ArgumentParser, Namespace
|
||||||
|
from collections import namedtuple
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import TYPE_CHECKING, AbstractSet, List, Optional
|
||||||
|
|
||||||
|
import pluggy
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
from ocrmypdf.helpers import Resolution
|
||||||
|
|
||||||
|
if TYPE_CHECKING:
|
||||||
|
from ocrmypdf._jobcontext import PageContext
|
||||||
|
from ocrmypdf.pdfinfo import PdfInfo
|
||||||
|
|
||||||
|
hookspec = pluggy.HookspecMarker('ocrmypdf')
|
||||||
|
|
||||||
|
# pylint: disable=unused-argument
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec
|
||||||
|
def add_options(parser: ArgumentParser) -> None:
|
||||||
|
"""Allows the plugin to add its own command line and API arguments.
|
||||||
|
|
||||||
|
OCRmyPDF converts command line arguments to API arguments, so adding
|
||||||
|
arguments here will cause new arguments to be processed for API calls
|
||||||
|
to ``ocrmypdf.ocr``, or when invoked on the command line.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This hook will be called from the main process, and may modify global state
|
||||||
|
before child worker processes are forked.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec
|
||||||
|
def check_options(options: Namespace) -> None:
|
||||||
|
"""Called to ask the plugin to check all of the options.
|
||||||
|
|
||||||
|
The plugin may check if options that it added are valid.
|
||||||
|
|
||||||
|
Warnings or other messages may be passed to the user by creating a logger
|
||||||
|
object using ``log = logging.getLogger(__name__)`` and logging to this.
|
||||||
|
|
||||||
|
The plugin may also modify the *options*. All objects that are in options
|
||||||
|
must be picklable so they can be marshalled to child worker processes.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ocrmypdf.exceptions.ExitCodeException: If options are not acceptable
|
||||||
|
and the application should terminate gracefully with an informative
|
||||||
|
message and error code.
|
||||||
|
Note:
|
||||||
|
This hook will be called from the main process, and may modify global state
|
||||||
|
before child worker processes are forked.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec
|
||||||
|
def validate(pdfinfo: 'PdfInfo', options: Namespace) -> None:
|
||||||
|
"""Called to give a plugin an opportunity to review *options* and *pdfinfo*.
|
||||||
|
|
||||||
|
*options* contains the "work order" to process a particular file. *pdfinfo*
|
||||||
|
contains information about the input file obtained after loading and
|
||||||
|
parsing. The plugin may modify the *options*. For example, you could decide
|
||||||
|
that a certain type of file should be treated with ``options.force_ocr = True``
|
||||||
|
based on information in its *pdfinfo*.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
ocrmypdf.exceptions.ExitCodeException: If options or pdfinfo are not acceptable
|
||||||
|
and the application should terminate gracefully with an informative
|
||||||
|
message and error code.
|
||||||
|
Note:
|
||||||
|
This hook will be called from the main process, and may modify global state
|
||||||
|
before child worker processes are forked.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def rasterize_pdf_page(
|
||||||
|
input_file: Path,
|
||||||
|
output_file: Path,
|
||||||
|
raster_device: str,
|
||||||
|
raster_dpi: Resolution,
|
||||||
|
pageno: int,
|
||||||
|
page_dpi: Optional[Resolution] = None,
|
||||||
|
rotation: Optional[int] = None,
|
||||||
|
filter_vector: bool = False,
|
||||||
|
) -> Path:
|
||||||
|
"""Rasterize one page of a PDF at resolution raster_dpi in canvas units.
|
||||||
|
|
||||||
|
The image is sized to match the integer pixels dimensions implied by
|
||||||
|
raster_dpi even if those numbers are noninteger. The image's DPI will
|
||||||
|
be overridden with the values in page_dpi.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
input_file: The PDF to rasterize.
|
||||||
|
output_file: The desired name of the rasterized image.
|
||||||
|
raster_device: Type of image to produce at output_file
|
||||||
|
raster_dpi: Resolution at which to rasterize page
|
||||||
|
pageno: Page number to rasterize (beginning at page 1)
|
||||||
|
page_dpi: Resolution, overriding output image DPI
|
||||||
|
rotation: Cardinal angle, clockwise, to rotate page
|
||||||
|
filter_vector: If True, remove vector graphics objects
|
||||||
|
Returns:
|
||||||
|
output_file
|
||||||
|
Note:
|
||||||
|
This hook will be called from child processes. Modifying global state
|
||||||
|
will not affect the main process or other child processes.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def filter_ocr_image(page: 'PageContext', image: Image) -> Image:
|
||||||
|
"""Called to filter the image before it is sent to OCR.
|
||||||
|
|
||||||
|
This is the image that OCR sees, not what the user sees when they view the
|
||||||
|
PDF.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This hook will be called from child processes. Modifying global state
|
||||||
|
will not affect the main process or other child processes.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def filter_page_image(page: 'PageContext', image_filename: Path) -> Path:
|
||||||
|
"""Called to filter the whole page before it is inserted into the PDF.
|
||||||
|
|
||||||
|
A whole page image is only produced when preprocessing command line arguments
|
||||||
|
are issued or when ``--force-ocr`` is issued. If no whole page is image is
|
||||||
|
produced for a given page, this function will not be called. This is not
|
||||||
|
the image that will be shown to OCR.
|
||||||
|
|
||||||
|
ocrmypdf will create the PDF page based on the image format used. If you
|
||||||
|
convert the image to a JPEG, the output page will be created as a JPEG, etc.
|
||||||
|
Note that the ocrmypdf image optimization stage may ultimately chose a
|
||||||
|
different format.
|
||||||
|
|
||||||
|
Note:
|
||||||
|
This hook will be called from child processes. Modifying global state
|
||||||
|
will not affect the main process or other child processes.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
OrientationConfidence = namedtuple('OrientationConfidence', ('angle', 'confidence'))
|
||||||
|
|
||||||
|
|
||||||
|
class OcrEngine(ABC):
|
||||||
|
@abstractstaticmethod
|
||||||
|
def version() -> str:
|
||||||
|
"""Returns the version of the OCR engine."""
|
||||||
|
|
||||||
|
@abstractstaticmethod
|
||||||
|
def creator_tag(options: Namespace) -> str:
|
||||||
|
"""Returns the creator tag to identify this software's role in creating the PDF."""
|
||||||
|
|
||||||
|
@abstractmethod
|
||||||
|
def __str__(self):
|
||||||
|
"""Returns name of OCR engine and version."""
|
||||||
|
|
||||||
|
@abstractstaticmethod
|
||||||
|
def languages(options: Namespace) -> AbstractSet[str]:
|
||||||
|
"""Returns the set of all languages that are supported by the engine.
|
||||||
|
|
||||||
|
Languages are typically given in 3-letter ISO 3166-1 codes, but actually
|
||||||
|
can be any value understood by the OCR engine."""
|
||||||
|
|
||||||
|
@abstractstaticmethod
|
||||||
|
def get_orientation(input_file: Path, options: Namespace) -> OrientationConfidence:
|
||||||
|
"""Returns the orientation of the image."""
|
||||||
|
|
||||||
|
@abstractstaticmethod
|
||||||
|
def generate_hocr(
|
||||||
|
input_file: Path, output_hocr: Path, output_text: Path, options: Namespace
|
||||||
|
) -> None:
|
||||||
|
"""Called to produce a hOCR file and sidecar text file."""
|
||||||
|
|
||||||
|
@abstractstaticmethod
|
||||||
|
def generate_pdf(
|
||||||
|
input_file: Path, output_pdf: Path, output_text: Path, options: Namespace
|
||||||
|
) -> None:
|
||||||
|
"""Called to produce a text only PDF.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
input_file: A page image on which to perform OCR.
|
||||||
|
output_pdf: The expected name of the output PDF, which must be
|
||||||
|
a single page PDF with no visible content of any kind, sized
|
||||||
|
to the dimensions implied by the input_file's width, height
|
||||||
|
and DPI. The image will be grafted onto the input PDF page.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def get_ocr_engine() -> OcrEngine:
|
||||||
|
"""Returns an OcrEngine to use for processing this file.
|
||||||
|
|
||||||
|
The OcrEngine may be instantiated multiple times, by both the main process
|
||||||
|
and child process. As such, it must be obtain store any state in ``options``
|
||||||
|
or some common location.
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
@hookspec(firstresult=True)
|
||||||
|
def generate_pdfa(
|
||||||
|
pdf_pages: List[Path],
|
||||||
|
pdfmark: Path,
|
||||||
|
output_file: Path,
|
||||||
|
compression: str,
|
||||||
|
pdf_version: str,
|
||||||
|
pdfa_part: str,
|
||||||
|
) -> Path:
|
||||||
|
"""Generate a PDF/A.
|
||||||
|
|
||||||
|
This API strongly assumes a PDF/A generator with Ghostscript's semantics.
|
||||||
|
|
||||||
|
OCRmyPDF will modify the metadata and possibly linearize the PDF/A after it
|
||||||
|
is generated.
|
||||||
|
|
||||||
|
Arguments:
|
||||||
|
pdf_pages: A list of one or more filenames, will be merged into output_file.
|
||||||
|
pdfmark: A PostScript file intended for Ghostscript with details on
|
||||||
|
how to perform the PDF/A conversion.
|
||||||
|
output_file: The name of the desired output file.
|
||||||
|
compression: One of ``'jpeg'``, ``'lossless'``, ``''``. For ``'jpeg'``,
|
||||||
|
the PDF/A generator should convert all images to JPEG encoding where
|
||||||
|
possible. For lossless, all images should be converted to FlateEncode
|
||||||
|
(lossless PNG). If an empty string, the PDF generator should make its
|
||||||
|
own decisions about how to encode images.
|
||||||
|
pdf_version: The minimum PDF version that the output file should be.
|
||||||
|
At its own discretion, the PDF/A generator may raise the version,
|
||||||
|
but should not lower it.
|
||||||
|
pdfa_part: The desired PDF/A compliance level, such as ``'2B'``.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
output_file: If successful, the hook should return ``output_file``.
|
||||||
|
"""
|
||||||
@@ -1,4 +1,4 @@
|
|||||||
# © 2016 James R. Barlow: github.com/jbarlow83
|
# © 2020 James R. Barlow: github.com/jbarlow83
|
||||||
#
|
#
|
||||||
# This file is part of OCRmyPDF.
|
# This file is part of OCRmyPDF.
|
||||||
#
|
#
|
||||||
@@ -23,30 +23,22 @@ import re
|
|||||||
import shutil
|
import shutil
|
||||||
import sys
|
import sys
|
||||||
from collections.abc import Mapping
|
from collections.abc import Mapping
|
||||||
|
from contextlib import suppress
|
||||||
from distutils.version import LooseVersion
|
from distutils.version import LooseVersion
|
||||||
from functools import lru_cache
|
from functools import lru_cache
|
||||||
|
from pathlib import Path
|
||||||
from subprocess import PIPE, STDOUT, CalledProcessError
|
from subprocess import PIPE, STDOUT, CalledProcessError
|
||||||
from subprocess import run as subprocess_run
|
from subprocess import run as subprocess_run
|
||||||
|
|
||||||
from ..exceptions import ExitCode, MissingDependencyError
|
from ocrmypdf.exceptions import MissingDependencyError
|
||||||
|
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
def _get_program(args, env=None):
|
|
||||||
program = args[0]
|
|
||||||
test_path = env.get('_OCRMYPDF_TEST_PATH', '')
|
|
||||||
if test_path:
|
|
||||||
program = shutil.which(program, path=test_path)
|
|
||||||
return program
|
|
||||||
|
|
||||||
|
|
||||||
def run(args, *, env=None, **kwargs):
|
def run(args, *, env=None, **kwargs):
|
||||||
"""Wrapper around subprocess.run()
|
"""Wrapper around subprocess.run()
|
||||||
|
|
||||||
The main purpose of this wrapper is to allow us to substitute the main program
|
The main purpose of this wrapper is to log subprocess output.
|
||||||
for a spoof in the test suite. The hidden variable _OCRMYPDF_TEST_PATH replaces
|
|
||||||
the main PATH as a location to check for programs to run.
|
|
||||||
|
|
||||||
Secondly we have to account for behavioral differences in Windows in particular.
|
Secondly we have to account for behavioral differences in Windows in particular.
|
||||||
Creating symbolic links in Windows requires administrator privileges and
|
Creating symbolic links in Windows requires administrator privileges and
|
||||||
@@ -61,40 +53,59 @@ def run(args, *, env=None, **kwargs):
|
|||||||
env = os.environ
|
env = os.environ
|
||||||
|
|
||||||
# Search in spoof path if necessary
|
# Search in spoof path if necessary
|
||||||
program = _get_program(args, env)
|
program = args[0]
|
||||||
|
|
||||||
# If we are running a .py on Windows, ensure we call it with this Python
|
|
||||||
# (to support test suite shims)
|
|
||||||
if os.name == 'nt' and program.lower().endswith('.py'):
|
|
||||||
args = [sys.executable, program] + args[1:]
|
|
||||||
else:
|
|
||||||
args = [program] + args[1:]
|
|
||||||
|
|
||||||
if os.name == 'nt':
|
if os.name == 'nt':
|
||||||
paths = os.pathsep.join(os.get_exec_path(env))
|
args = _fix_windows_args(program, args, env)
|
||||||
if not shutil.which(args[0], path=paths):
|
|
||||||
shimmed_path = shim_paths_with_program_files(env)
|
|
||||||
new_args0 = shutil.which(args[0], path=shimmed_path)
|
|
||||||
if new_args0:
|
|
||||||
args[0] = new_args0
|
|
||||||
|
|
||||||
process_log = log.getChild(os.path.basename(program))
|
log.debug("Running: %s", args)
|
||||||
process_log.debug("Running: %s", args)
|
process_log = log.getChild('subprocess.' + os.path.basename(program))
|
||||||
if sys.version_info < (3, 7) and os.name == 'nt':
|
if sys.version_info < (3, 7) and os.name == 'nt':
|
||||||
# Can't use close_fds=True on Windows with Python 3.6 or older
|
# Can't use close_fds=True on Windows with Python 3.6 or older
|
||||||
# https://bugs.python.org/issue19575, etc.
|
# https://bugs.python.org/issue19575, etc.
|
||||||
kwargs['close_fds'] = False
|
kwargs['close_fds'] = False
|
||||||
proc = subprocess_run(args, env=env, **kwargs)
|
|
||||||
if process_log.isEnabledFor(logging.DEBUG):
|
stderr = None
|
||||||
try:
|
try:
|
||||||
stderr = proc.stderr.decode('utf-8', 'replace')
|
proc = subprocess_run(args, env=env, **kwargs)
|
||||||
except AttributeError:
|
except CalledProcessError as e:
|
||||||
stderr = proc.stderr
|
stderr = getattr(e, 'stderr', None)
|
||||||
if stderr:
|
raise
|
||||||
|
else:
|
||||||
|
stderr = getattr(proc, 'stderr', None)
|
||||||
|
finally:
|
||||||
|
if process_log.isEnabledFor(logging.DEBUG) and stderr:
|
||||||
|
with suppress(AttributeError, UnicodeDecodeError):
|
||||||
|
stderr = stderr.decode('utf-8', 'replace')
|
||||||
process_log.debug("stderr = %s", stderr)
|
process_log.debug("stderr = %s", stderr)
|
||||||
return proc
|
return proc
|
||||||
|
|
||||||
|
|
||||||
|
def _fix_windows_args(program, args, env):
|
||||||
|
"""Adjust our desired program and command line arguments for use on Windows"""
|
||||||
|
|
||||||
|
if sys.version_info < (3, 8):
|
||||||
|
# bpo-33617 - Windows needs manual Path -> str conversion
|
||||||
|
args = [os.fspath(arg) for arg in args]
|
||||||
|
program = os.fspath(program)
|
||||||
|
|
||||||
|
# If we are running a .py on Windows, ensure we call it with this Python
|
||||||
|
# (to support test suite shims)
|
||||||
|
if program.lower().endswith('.py'):
|
||||||
|
args = [sys.executable] + args
|
||||||
|
|
||||||
|
paths = os.pathsep.join(os.get_exec_path(env))
|
||||||
|
if not shutil.which(args[0], path=paths):
|
||||||
|
# If the program we want is not on the PATH, add some interesting
|
||||||
|
# locations in %PROGRAMFILES% to the PATH and try again
|
||||||
|
shimmed_path = shim_paths_with_program_files(env)
|
||||||
|
new_args0 = shutil.which(args[0], path=shimmed_path)
|
||||||
|
if new_args0:
|
||||||
|
args[0] = new_args0
|
||||||
|
return args
|
||||||
|
|
||||||
|
|
||||||
|
@lru_cache(maxsize=None)
|
||||||
def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None):
|
def get_version(program, *, version_arg='--version', regex=r'(\d+(\.\d+)*)', env=None):
|
||||||
"""Get the version of the specified program"""
|
"""Get the version of the specified program"""
|
||||||
args_prog = [program, version_arg]
|
args_prog = [program, version_arg]
|
||||||
@@ -138,24 +149,25 @@ def shim_paths_with_program_files(env=None):
|
|||||||
program_files = env.get('PROGRAMFILES', '')
|
program_files = env.get('PROGRAMFILES', '')
|
||||||
if not program_files:
|
if not program_files:
|
||||||
return env.get('PATH', '')
|
return env.get('PATH', '')
|
||||||
paths = []
|
|
||||||
try:
|
def path_walker():
|
||||||
for dirname in os.listdir(program_files):
|
for path in Path(program_files).iterdir():
|
||||||
if dirname.lower() == 'tesseract-ocr':
|
if not path.is_dir():
|
||||||
paths.append(os.path.join(program_files, dirname))
|
continue
|
||||||
elif dirname.lower() == 'gs':
|
if path.name.lower() == 'tesseract-ocr':
|
||||||
try:
|
yield path
|
||||||
latest_gs = max(
|
elif path.name.lower() == 'gs':
|
||||||
os.listdir(os.path.join(program_files, dirname)),
|
yield from (p for p in path.glob('**/bin') if p.is_dir())
|
||||||
key=lambda d: float(d[2:]),
|
|
||||||
)
|
paths = sorted(
|
||||||
except (FileNotFoundError, NotADirectoryError):
|
(p for p in path_walker()), key=lambda p: (p.name, p.parent.name), reverse=True
|
||||||
continue
|
)
|
||||||
paths.append(os.path.join(program_files, dirname, latest_gs, 'bin'))
|
paths.extend(
|
||||||
except EnvironmentError:
|
Path(str_path)
|
||||||
pass
|
for str_path in os.get_exec_path(env)
|
||||||
paths.extend(path for path in os.get_exec_path(env) if path not in set(paths))
|
if Path(str_path) not in set(paths)
|
||||||
return os.pathsep.join(paths)
|
)
|
||||||
|
return os.pathsep.join(str(p) for p in paths)
|
||||||
|
|
||||||
|
|
||||||
missing_program = '''
|
missing_program = '''
|
||||||
@@ -233,10 +245,10 @@ def _error_trailer(program, package, **kwargs):
|
|||||||
|
|
||||||
|
|
||||||
def _error_missing_program(program, package, required_for, recommended):
|
def _error_missing_program(program, package, required_for, recommended):
|
||||||
if required_for:
|
if recommended:
|
||||||
|
log.warning(missing_recommend_program.format(**locals()))
|
||||||
|
elif required_for:
|
||||||
log.error(missing_optional_program.format(**locals()))
|
log.error(missing_optional_program.format(**locals()))
|
||||||
elif recommended:
|
|
||||||
log.info(missing_recommend_program.format(**locals()))
|
|
||||||
else:
|
else:
|
||||||
log.error(missing_program.format(**locals()))
|
log.error(missing_program.format(**locals()))
|
||||||
_error_trailer(**locals())
|
_error_trailer(**locals())
|
||||||
@@ -258,13 +270,12 @@ def check_external_program(
|
|||||||
need_version,
|
need_version,
|
||||||
required_for=None,
|
required_for=None,
|
||||||
recommended=False,
|
recommended=False,
|
||||||
**kwargs, # To consume log parameter
|
|
||||||
):
|
):
|
||||||
if kwargs:
|
|
||||||
if not 'log' in kwargs:
|
|
||||||
log.warning('check_external_program(log=...) is deprecated')
|
|
||||||
try:
|
try:
|
||||||
found_version = version_checker()
|
if callable(version_checker):
|
||||||
|
found_version = version_checker()
|
||||||
|
else:
|
||||||
|
found_version = version_checker
|
||||||
except (CalledProcessError, FileNotFoundError, MissingDependencyError):
|
except (CalledProcessError, FileNotFoundError, MissingDependencyError):
|
||||||
_error_missing_program(program, package, required_for, recommended)
|
_error_missing_program(program, package, required_for, recommended)
|
||||||
if not recommended:
|
if not recommended:
|
||||||
@@ -279,7 +290,7 @@ def check_external_program(
|
|||||||
found_version = remove_leading_v(found_version)
|
found_version = remove_leading_v(found_version)
|
||||||
need_version = remove_leading_v(need_version)
|
need_version = remove_leading_v(need_version)
|
||||||
|
|
||||||
if LooseVersion(found_version) < LooseVersion(need_version):
|
if found_version and LooseVersion(found_version) < LooseVersion(need_version):
|
||||||
_error_old_version(program, package, need_version, found_version, required_for)
|
_error_old_version(program, package, need_version, found_version, required_for)
|
||||||
if not recommended:
|
if not recommended:
|
||||||
raise MissingDependencyError()
|
raise MissingDependencyError()
|
||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.1.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
+5
-5
@@ -4,12 +4,12 @@
|
|||||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||||
<head>
|
<head>
|
||||||
<title></title>
|
<title></title>
|
||||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8" />
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||||
<meta name='ocr-system' content='tesseract 4.0.0' />
|
<meta name='ocr-system' content='tesseract 4.1.1' />
|
||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.xam82ph5/000001_ocr.png"; bbox 0 0 1000 800; ppageno 0'>
|
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.mrqsewbu/000001_ocr.png"; bbox 0 0 1000 800; ppageno 0'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 296 96 704 504">
|
<div class='ocr_carea' id='block_1_1' title="bbox 296 96 704 504">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 296 96 704 504">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 296 96 704 504">
|
||||||
<span class='ocr_line' id='line_1_1' title="bbox 296 96 704 504; baseline 0 296; x_size 169.33333; x_descenders 42.333332; x_ascenders 42.333336">
|
<span class='ocr_line' id='line_1_1' title="bbox 296 96 704 504; baseline 0 296; x_size 169.33333; x_descenders 42.333332; x_ascenders 42.333336">
|
||||||
|
|||||||
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
+158
-158
@@ -4,19 +4,19 @@
|
|||||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||||
<head>
|
<head>
|
||||||
<title></title>
|
<title></title>
|
||||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8" />
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||||
<meta name='ocr-system' content='tesseract 4.0.0' />
|
<meta name='ocr-system' content='tesseract 4.1.1' />
|
||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.6a60f5yy/000001_ocr.png"; bbox 0 0 2550 3300; ppageno 0'>
|
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.ces85e5u/000001_ocr.png"; bbox 0 0 2550 3300; ppageno 0'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
||||||
<span class='ocr_line' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
||||||
<span class='ocrx_word' id='word_1_1' title='bbox 882 132 1036 202; x_wconf 93'>The</span>
|
<span class='ocrx_word' id='word_1_1' title='bbox 882 132 1036 202; x_wconf 93'>The</span>
|
||||||
<span class='ocrx_word' id='word_1_2' title='bbox 1061 131 1657 217; x_wconf 91'>LinnSequencer</span>
|
<span class='ocrx_word' id='word_1_2' title='bbox 1061 131 1657 217; x_wconf 91'>LinnSequencer</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_2' title="bbox 582 215 1968 303; baseline 0 -17; x_size 87; x_descenders 16; x_ascenders 21">
|
<span class='ocr_header' id='line_1_2' title="bbox 582 215 1968 303; baseline 0 -17; x_size 87; x_descenders 16; x_ascenders 21">
|
||||||
<span class='ocrx_word' id='word_1_3' title='bbox 582 215 674 286; x_wconf 96'>32</span>
|
<span class='ocrx_word' id='word_1_3' title='bbox 582 215 674 286; x_wconf 96'>32</span>
|
||||||
<span class='ocrx_word' id='word_1_4' title='bbox 697 218 923 288; x_wconf 95'>Track</span>
|
<span class='ocrx_word' id='word_1_4' title='bbox 697 218 923 288; x_wconf 95'>Track</span>
|
||||||
<span class='ocrx_word' id='word_1_5' title='bbox 948 218 1181 287; x_wconf 96'>MIDI</span>
|
<span class='ocrx_word' id='word_1_5' title='bbox 948 218 1181 287; x_wconf 96'>MIDI</span>
|
||||||
@@ -27,27 +27,27 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_2' title="bbox 347 380 2188 574">
|
<div class='ocr_carea' id='block_1_2' title="bbox 347 380 2188 574">
|
||||||
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 347 380 2188 423">
|
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 347 380 2188 423">
|
||||||
<span class='ocr_line' id='line_1_3' title="bbox 347 380 2188 423; baseline -0.001 -12; x_size 38; x_descenders 8; x_ascenders 10">
|
<span class='ocr_header' id='line_1_3' title="bbox 347 380 2188 423; baseline -0.001 -12; x_size 38; x_descenders 8; x_ascenders 10">
|
||||||
<span class='ocrx_word' id='word_1_8' title='bbox 347 380 412 410; x_wconf 93'>The</span>
|
<span class='ocrx_word' id='word_1_8' title='bbox 347 380 412 410; x_wconf 93'>The</span>
|
||||||
<span class='ocrx_word' id='word_1_9' title='bbox 424 380 661 417; x_wconf 92'>LinnSequencer</span>
|
<span class='ocrx_word' id='word_1_9' title='bbox 424 380 676 417; x_wconf 92'>LinnSequencer</span>
|
||||||
<span class='ocrx_word' id='word_1_10' title='bbox 663 380 712 411; x_wconf 96'>is</span>
|
<span class='ocrx_word' id='word_1_10' title='bbox 688 380 712 411; x_wconf 96'>is</span>
|
||||||
<span class='ocrx_word' id='word_1_11' title='bbox 724 390 743 411; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_11' title='bbox 724 390 743 411; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_12' title='bbox 754 381 1005 423; x_wconf 96'>state-of-the-art</span>
|
<span class='ocrx_word' id='word_1_12' title='bbox 754 381 1005 423; x_wconf 96'>state-of-the-art</span>
|
||||||
<span class='ocrx_word' id='word_1_13' title='bbox 1017 380 1226 418; x_wconf 96'>composition</span>
|
<span class='ocrx_word' id='word_1_13' title='bbox 1017 380 1226 418; x_wconf 96'>composition</span>
|
||||||
<span class='ocrx_word' id='word_1_14' title='bbox 1238 381 1299 411; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_14' title='bbox 1238 381 1299 411; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_15' title='bbox 1311 380 1507 418; x_wconf 96'>performance</span>
|
<span class='ocrx_word' id='word_1_15' title='bbox 1311 380 1525 418; x_wconf 96'>performance</span>
|
||||||
<span class='ocrx_word' id='word_1_16' title='bbox 1509 385 1591 411; x_wconf 96'>tool</span>
|
<span class='ocrx_word' id='word_1_16' title='bbox 1536 380 1602 411; x_wconf 96'>tool</span>
|
||||||
<span class='ocrx_word' id='word_1_17' title='bbox 1593 380 1663 411; x_wconf 97'>for</span>
|
<span class='ocrx_word' id='word_1_17' title='bbox 1615 380 1663 411; x_wconf 97'>for</span>
|
||||||
<span class='ocrx_word' id='word_1_18' title='bbox 1674 381 1725 410; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_18' title='bbox 1674 381 1725 410; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_19' title='bbox 1737 380 1940 417; x_wconf 95'>professional</span>
|
<span class='ocrx_word' id='word_1_19' title='bbox 1737 380 1940 417; x_wconf 95'>professional</span>
|
||||||
<span class='ocrx_word' id='word_1_20' title='bbox 1952 380 2103 411; x_wconf 96'>musician.</span>
|
<span class='ocrx_word' id='word_1_20' title='bbox 1952 380 2112 411; x_wconf 96'>musician.</span>
|
||||||
<span class='ocrx_word' id='word_1_21' title='bbox 2106 381 2152 410; x_wconf 96'>It</span>
|
<span class='ocrx_word' id='word_1_21' title='bbox 2127 381 2152 410; x_wconf 96'>It</span>
|
||||||
<span class='ocrx_word' id='word_1_22' title='bbox 2164 380 2188 410; x_wconf 96'>is</span>
|
<span class='ocrx_word' id='word_1_22' title='bbox 2164 380 2188 410; x_wconf 96'>is</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
|
|
||||||
<p class='ocr_par' id='par_1_3' lang='eng' title="bbox 347 430 1988 468">
|
<p class='ocr_par' id='par_1_3' lang='eng' title="bbox 347 430 1988 468">
|
||||||
<span class='ocr_line' id='line_1_4' title="bbox 347 430 1988 468; baseline 0 -8; x_size 37; x_descenders 7; x_ascenders 9">
|
<span class='ocr_header' id='line_1_4' title="bbox 347 430 1988 468; baseline 0 -8; x_size 37; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_23' title='bbox 347 430 507 467; x_wconf 96'>extremely</span>
|
<span class='ocrx_word' id='word_1_23' title='bbox 347 430 507 467; x_wconf 96'>extremely</span>
|
||||||
<span class='ocrx_word' id='word_1_24' title='bbox 518 430 677 467; x_wconf 96'>powerful,</span>
|
<span class='ocrx_word' id='word_1_24' title='bbox 518 430 677 467; x_wconf 96'>powerful,</span>
|
||||||
<span class='ocrx_word' id='word_1_25' title='bbox 691 435 739 467; x_wconf 96'>yet</span>
|
<span class='ocrx_word' id='word_1_25' title='bbox 691 435 739 467; x_wconf 96'>yet</span>
|
||||||
@@ -66,7 +66,7 @@
|
|||||||
</p>
|
</p>
|
||||||
|
|
||||||
<p class='ocr_par' id='par_1_4' lang='eng' title="bbox 350 482 2093 574">
|
<p class='ocr_par' id='par_1_4' lang='eng' title="bbox 350 482 2093 574">
|
||||||
<span class='ocr_line' id='line_1_5' title="bbox 350 482 2093 527; baseline 0 -9; x_size 43; x_descenders 7; x_ascenders 11">
|
<span class='ocr_header' id='line_1_5' title="bbox 350 482 2093 527; baseline 0 -9; x_size 43; x_descenders 7; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_37' title='bbox 350 490 368 508; x_wconf 73'>¢</span>
|
<span class='ocrx_word' id='word_1_37' title='bbox 350 490 368 508; x_wconf 73'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_38' title='bbox 383 482 585 526; x_wconf 95'>Operation</span>
|
<span class='ocrx_word' id='word_1_38' title='bbox 383 482 585 526; x_wconf 95'>Operation</span>
|
||||||
<span class='ocrx_word' id='word_1_39' title='bbox 598 482 627 518; x_wconf 96'>is</span>
|
<span class='ocrx_word' id='word_1_39' title='bbox 598 482 627 518; x_wconf 96'>is</span>
|
||||||
@@ -81,18 +81,18 @@
|
|||||||
<span class='ocrx_word' id='word_1_48' title='bbox 1741 483 1957 525; x_wconf 96'>RECORD,</span>
|
<span class='ocrx_word' id='word_1_48' title='bbox 1741 483 1957 525; x_wconf 96'>RECORD,</span>
|
||||||
<span class='ocrx_word' id='word_1_49' title='bbox 1974 483 2093 518; x_wconf 96'>FAST</span>
|
<span class='ocrx_word' id='word_1_49' title='bbox 1974 483 2093 518; x_wconf 96'>FAST</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_6' title="bbox 383 532 1345 574; baseline 0.001 -7; x_size 43; x_descenders 7; x_ascenders 11">
|
<span class='ocr_header' id='line_1_6' title="bbox 383 532 1345 574; baseline 0.001 -7; x_size 43; x_descenders 7; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_50' title='bbox 383 532 635 574; x_wconf 96'>FORWARD,</span>
|
<span class='ocrx_word' id='word_1_50' title='bbox 383 532 635 574; x_wconf 96'>FORWARD,</span>
|
||||||
<span class='ocrx_word' id='word_1_51' title='bbox 652 532 865 574; x_wconf 95'>REWIND,</span>
|
<span class='ocrx_word' id='word_1_51' title='bbox 652 532 865 574; x_wconf 95'>REWIND,</span>
|
||||||
<span class='ocrx_word' id='word_1_52' title='bbox 882 532 956 568; x_wconf 95'>and</span>
|
<span class='ocrx_word' id='word_1_52' title='bbox 882 532 956 568; x_wconf 95'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_53' title='bbox 971 532 1131 568; x_wconf 95'>LOCATE</span>
|
<span class='ocrx_word' id='word_1_53' title='bbox 971 532 1163 568; x_wconf 95'>LOCATE</span>
|
||||||
<span class='ocrx_word' id='word_1_54' title='bbox 1132 532 1345 568; x_wconf 95'>controls.</span>
|
<span class='ocrx_word' id='word_1_54' title='bbox 1177 532 1345 568; x_wconf 95'>controls.</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_3' title="bbox 349 589 2136 685">
|
<div class='ocr_carea' id='block_1_3' title="bbox 349 589 2136 685">
|
||||||
<p class='ocr_par' id='par_1_5' lang='eng' title="bbox 349 589 2136 685">
|
<p class='ocr_par' id='par_1_5' lang='eng' title="bbox 349 589 2136 685">
|
||||||
<span class='ocr_line' id='line_1_7' title="bbox 349 589 2136 634; baseline 0.001 -9; x_size 44; x_descenders 8; x_ascenders 11">
|
<span class='ocr_header' id='line_1_7' title="bbox 349 589 2136 634; baseline 0.001 -9; x_size 44; x_descenders 8; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_55' title='bbox 349 597 368 615; x_wconf 59'>e</span>
|
<span class='ocrx_word' id='word_1_55' title='bbox 349 597 368 615; x_wconf 59'>e</span>
|
||||||
<span class='ocrx_word' id='word_1_56' title='bbox 383 590 482 625; x_wconf 96'>Each</span>
|
<span class='ocrx_word' id='word_1_56' title='bbox 383 590 482 625; x_wconf 96'>Each</span>
|
||||||
<span class='ocrx_word' id='word_1_57' title='bbox 496 589 539 625; x_wconf 96'>of</span>
|
<span class='ocrx_word' id='word_1_57' title='bbox 496 589 539 625; x_wconf 96'>of</span>
|
||||||
@@ -108,7 +108,7 @@
|
|||||||
<span class='ocrx_word' id='word_1_67' title='bbox 1934 590 2035 626; x_wconf 96'>track</span>
|
<span class='ocrx_word' id='word_1_67' title='bbox 1934 590 2035 626; x_wconf 96'>track</span>
|
||||||
<span class='ocrx_word' id='word_1_68' title='bbox 2050 600 2136 634; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_68' title='bbox 2050 600 2136 634; x_wconf 96'>may</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_8' title="bbox 383 639 2022 685; baseline 0.001 -10; x_size 44; x_descenders 8; x_ascenders 11">
|
<span class='ocr_header' id='line_1_8' title="bbox 383 639 2022 685; baseline 0.001 -10; x_size 44; x_descenders 8; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_69' title='bbox 383 639 428 675; x_wconf 95'>be</span>
|
<span class='ocrx_word' id='word_1_69' title='bbox 383 639 428 675; x_wconf 95'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_70' title='bbox 442 639 607 684; x_wconf 95'>assigned</span>
|
<span class='ocrx_word' id='word_1_70' title='bbox 442 639 607 684; x_wconf 95'>assigned</span>
|
||||||
<span class='ocrx_word' id='word_1_71' title='bbox 621 645 659 676; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_71' title='bbox 621 645 659 676; x_wconf 96'>to</span>
|
||||||
@@ -135,7 +135,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_5' title="bbox 349 748 2117 793">
|
<div class='ocr_carea' id='block_1_5' title="bbox 349 748 2117 793">
|
||||||
<p class='ocr_par' id='par_1_7' lang='eng' title="bbox 349 748 2117 793">
|
<p class='ocr_par' id='par_1_7' lang='eng' title="bbox 349 748 2117 793">
|
||||||
<span class='ocr_line' id='line_1_10' title="bbox 349 748 2117 793; baseline 0 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
<span class='ocr_header' id='line_1_10' title="bbox 349 748 2117 793; baseline 0 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_84' title='bbox 349 755 367 774; x_wconf 58'>¢</span>
|
<span class='ocrx_word' id='word_1_84' title='bbox 349 755 367 774; x_wconf 58'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_85' title='bbox 383 748 573 784; x_wconf 91'>Ultra-fast</span>
|
<span class='ocrx_word' id='word_1_85' title='bbox 383 748 573 784; x_wconf 91'>Ultra-fast</span>
|
||||||
<span class='ocrx_word' id='word_1_86' title='bbox 588 749 677 784; x_wconf 22'>3%”</span>
|
<span class='ocrx_word' id='word_1_86' title='bbox 588 749 677 784; x_wconf 22'>3%”</span>
|
||||||
@@ -147,8 +147,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_92' title='bbox 1330 748 1368 784; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_92' title='bbox 1330 748 1368 784; x_wconf 96'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_93' title='bbox 1382 748 1535 785; x_wconf 96'>seconds</span>
|
<span class='ocrx_word' id='word_1_93' title='bbox 1382 748 1535 785; x_wconf 96'>seconds</span>
|
||||||
<span class='ocrx_word' id='word_1_94' title='bbox 1550 748 1624 784; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_94' title='bbox 1550 748 1624 784; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_95' title='bbox 1638 748 1728 784; x_wconf 96'>holds</span>
|
<span class='ocrx_word' id='word_1_95' title='bbox 1638 748 1746 784; x_wconf 96'>holds</span>
|
||||||
<span class='ocrx_word' id='word_1_96' title='bbox 1730 759 1844 784; x_wconf 96'>over</span>
|
<span class='ocrx_word' id='word_1_96' title='bbox 1761 759 1844 784; x_wconf 96'>over</span>
|
||||||
<span class='ocrx_word' id='word_1_97' title='bbox 1859 749 2000 791; x_wconf 96'>110,000</span>
|
<span class='ocrx_word' id='word_1_97' title='bbox 1859 749 2000 791; x_wconf 96'>110,000</span>
|
||||||
<span class='ocrx_word' id='word_1_98' title='bbox 2013 753 2117 784; x_wconf 96'>notes</span>
|
<span class='ocrx_word' id='word_1_98' title='bbox 2013 753 2117 784; x_wconf 96'>notes</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -164,23 +164,23 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_7' title="bbox 349 855 2030 1016">
|
<div class='ocr_carea' id='block_1_7' title="bbox 349 855 2030 1016">
|
||||||
<p class='ocr_par' id='par_1_9' lang='eng' title="bbox 349 855 2030 1016">
|
<p class='ocr_par' id='par_1_9' lang='eng' title="bbox 349 855 2030 1016">
|
||||||
<span class='ocr_line' id='line_1_12' title="bbox 350 855 1638 900; baseline 0.001 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
<span class='ocr_header' id='line_1_12' title="bbox 350 855 1638 900; baseline 0.001 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_101' title='bbox 350 863 367 881; x_wconf 45'>¢</span>
|
<span class='ocrx_word' id='word_1_101' title='bbox 350 863 367 881; x_wconf 45'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_102' title='bbox 383 856 444 891; x_wconf 95'>One</span>
|
<span class='ocrx_word' id='word_1_102' title='bbox 383 856 464 891; x_wconf 95'>One</span>
|
||||||
<span class='ocrx_word' id='word_1_103' title='bbox 445 866 502 891; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_103' title='bbox 478 866 520 891; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_104' title='bbox 503 855 580 891; x_wconf 96'>all</span>
|
<span class='ocrx_word' id='word_1_104' title='bbox 534 855 580 891; x_wconf 96'>all</span>
|
||||||
<span class='ocrx_word' id='word_1_105' title='bbox 594 856 712 892; x_wconf 95'>tracks</span>
|
<span class='ocrx_word' id='word_1_105' title='bbox 594 856 712 892; x_wconf 95'>tracks</span>
|
||||||
<span class='ocrx_word' id='word_1_106' title='bbox 726 867 811 900; x_wconf 95'>may</span>
|
<span class='ocrx_word' id='word_1_106' title='bbox 726 867 811 900; x_wconf 95'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_107' title='bbox 823 856 869 892; x_wconf 81'>be</span>
|
<span class='ocrx_word' id='word_1_107' title='bbox 823 856 869 892; x_wconf 81'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_108' title='bbox 882 856 1175 892; x_wconf 96'>TRANSPOSED</span>
|
<span class='ocrx_word' id='word_1_108' title='bbox 882 856 1212 892; x_wconf 96'>TRANSPOSED</span>
|
||||||
<span class='ocrx_word' id='word_1_109' title='bbox 1178 857 1264 892; x_wconf 96'>at</span>
|
<span class='ocrx_word' id='word_1_109' title='bbox 1227 861 1264 892; x_wconf 96'>at</span>
|
||||||
<span class='ocrx_word' id='word_1_110' title='bbox 1277 856 1338 892; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_110' title='bbox 1277 856 1338 892; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_111' title='bbox 1351 856 1463 892; x_wconf 96'>touch</span>
|
<span class='ocrx_word' id='word_1_111' title='bbox 1351 856 1463 892; x_wconf 96'>touch</span>
|
||||||
<span class='ocrx_word' id='word_1_112' title='bbox 1477 867 1501 892; x_wconf 96'>of</span>
|
<span class='ocrx_word' id='word_1_112' title='bbox 1477 856 1520 892; x_wconf 96'>of</span>
|
||||||
<span class='ocrx_word' id='word_1_113' title='bbox 1502 856 1554 892; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_113' title='bbox 1531 867 1554 892; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_114' title='bbox 1568 856 1638 900; x_wconf 96'>key.</span>
|
<span class='ocrx_word' id='word_1_114' title='bbox 1568 856 1638 900; x_wconf 96'>key.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_13' title="bbox 350 913 1535 958; baseline 0.001 -9; x_size 44; x_descenders 8; x_ascenders 11">
|
<span class='ocr_header' id='line_1_13' title="bbox 350 913 1535 958; baseline 0.001 -9; x_size 44; x_descenders 8; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_115' title='bbox 350 921 367 939; x_wconf 45'>e</span>
|
<span class='ocrx_word' id='word_1_115' title='bbox 350 921 367 939; x_wconf 45'>e</span>
|
||||||
<span class='ocrx_word' id='word_1_116' title='bbox 383 913 568 950; x_wconf 96'>Exclusive</span>
|
<span class='ocrx_word' id='word_1_116' title='bbox 383 913 568 950; x_wconf 96'>Exclusive</span>
|
||||||
<span class='ocrx_word' id='word_1_117' title='bbox 581 913 756 950; x_wconf 96'>real-time</span>
|
<span class='ocrx_word' id='word_1_117' title='bbox 581 913 756 950; x_wconf 96'>real-time</span>
|
||||||
@@ -190,7 +190,7 @@
|
|||||||
<span class='ocrx_word' id='word_1_121' title='bbox 1266 914 1400 958; x_wconf 96'>editing</span>
|
<span class='ocrx_word' id='word_1_121' title='bbox 1266 914 1400 958; x_wconf 96'>editing</span>
|
||||||
<span class='ocrx_word' id='word_1_122' title='bbox 1414 915 1535 950; x_wconf 95'>FAST.</span>
|
<span class='ocrx_word' id='word_1_122' title='bbox 1414 915 1535 950; x_wconf 95'>FAST.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_14' title="bbox 349 971 2030 1016; baseline 0.001 -10; x_size 44; x_descenders 8; x_ascenders 11">
|
<span class='ocr_header' id='line_1_14' title="bbox 349 971 2030 1016; baseline 0.001 -10; x_size 44; x_descenders 8; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_123' title='bbox 349 979 367 997; x_wconf 0'>*</span>
|
<span class='ocrx_word' id='word_1_123' title='bbox 349 979 367 997; x_wconf 0'>*</span>
|
||||||
<span class='ocrx_word' id='word_1_124' title='bbox 382 971 568 1007; x_wconf 95'>Exclusive</span>
|
<span class='ocrx_word' id='word_1_124' title='bbox 382 971 568 1007; x_wconf 95'>Exclusive</span>
|
||||||
<span class='ocrx_word' id='word_1_125' title='bbox 582 972 773 1007; x_wconf 96'>REPEAT</span>
|
<span class='ocrx_word' id='word_1_125' title='bbox 582 972 773 1007; x_wconf 96'>REPEAT</span>
|
||||||
@@ -216,11 +216,11 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_9' title="bbox 349 1080 2174 1125">
|
<div class='ocr_carea' id='block_1_9' title="bbox 349 1080 2174 1125">
|
||||||
<p class='ocr_par' id='par_1_11' lang='eng' title="bbox 349 1080 2174 1125">
|
<p class='ocr_par' id='par_1_11' lang='eng' title="bbox 349 1080 2174 1125">
|
||||||
<span class='ocr_line' id='line_1_16' title="bbox 349 1080 2174 1125; baseline 0.001 -11; x_size 45; x_descenders 9; x_ascenders 11">
|
<span class='ocr_header' id='line_1_16' title="bbox 349 1080 2174 1125; baseline 0.001 -11; x_size 45; x_descenders 9; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_137' title='bbox 349 1087 367 1105; x_wconf 80'>¢</span>
|
<span class='ocrx_word' id='word_1_137' title='bbox 349 1087 367 1105; x_wconf 80'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_138' title='bbox 382 1080 567 1115; x_wconf 95'>TIMING</span>
|
<span class='ocrx_word' id='word_1_138' title='bbox 382 1080 567 1115; x_wconf 95'>TIMING</span>
|
||||||
<span class='ocrx_word' id='word_1_139' title='bbox 582 1080 870 1116; x_wconf 95'>CORRECTION</span>
|
<span class='ocrx_word' id='word_1_139' title='bbox 582 1080 908 1116; x_wconf 95'>CORRECTION</span>
|
||||||
<span class='ocrx_word' id='word_1_140' title='bbox 872 1080 1041 1116; x_wconf 96'>works</span>
|
<span class='ocrx_word' id='word_1_140' title='bbox 921 1080 1041 1116; x_wconf 96'>works</span>
|
||||||
<span class='ocrx_word' id='word_1_141' title='bbox 1056 1080 1186 1124; x_wconf 96'>during</span>
|
<span class='ocrx_word' id='word_1_141' title='bbox 1056 1080 1186 1124; x_wconf 96'>during</span>
|
||||||
<span class='ocrx_word' id='word_1_142' title='bbox 1199 1080 1378 1124; x_wconf 96'>playback</span>
|
<span class='ocrx_word' id='word_1_142' title='bbox 1199 1080 1378 1124; x_wconf 96'>playback</span>
|
||||||
<span class='ocrx_word' id='word_1_143' title='bbox 1392 1080 1466 1116; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_143' title='bbox 1392 1080 1466 1116; x_wconf 96'>and</span>
|
||||||
@@ -233,11 +233,11 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_10' title="bbox 349 1137 1287 1182">
|
<div class='ocr_carea' id='block_1_10' title="bbox 349 1137 1287 1182">
|
||||||
<p class='ocr_par' id='par_1_12' lang='eng' title="bbox 349 1137 1287 1182">
|
<p class='ocr_par' id='par_1_12' lang='eng' title="bbox 349 1137 1287 1182">
|
||||||
<span class='ocr_line' id='line_1_17' title="bbox 349 1137 1287 1182; baseline 0.001 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
<span class='ocr_textfloat' id='line_1_17' title="bbox 349 1137 1287 1182; baseline 0.001 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_148' title='bbox 349 1145 367 1163; x_wconf 80'>¢</span>
|
<span class='ocrx_word' id='word_1_148' title='bbox 349 1145 367 1163; x_wconf 80'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_149' title='bbox 382 1137 560 1182; x_wconf 95'>Optional</span>
|
<span class='ocrx_word' id='word_1_149' title='bbox 382 1137 560 1182; x_wconf 95'>Optional</span>
|
||||||
<span class='ocrx_word' id='word_1_150' title='bbox 575 1138 708 1174; x_wconf 96'>SMPTE</span>
|
<span class='ocrx_word' id='word_1_150' title='bbox 575 1138 739 1174; x_wconf 96'>SMPTE</span>
|
||||||
<span class='ocrx_word' id='word_1_151' title='bbox 709 1138 839 1174; x_wconf 96'>time</span>
|
<span class='ocrx_word' id='word_1_151' title='bbox 752 1138 839 1174; x_wconf 96'>time</span>
|
||||||
<span class='ocrx_word' id='word_1_152' title='bbox 853 1138 945 1174; x_wconf 95'>code</span>
|
<span class='ocrx_word' id='word_1_152' title='bbox 853 1138 945 1174; x_wconf 95'>code</span>
|
||||||
<span class='ocrx_word' id='word_1_153' title='bbox 959 1138 1287 1182; x_wconf 96'>synchronization.</span>
|
<span class='ocrx_word' id='word_1_153' title='bbox 959 1138 1287 1182; x_wconf 96'>synchronization.</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -278,9 +278,9 @@
|
|||||||
<span class='ocrx_word' id='word_1_170' title='bbox 346 1379 411 1406; x_wconf 96'>then</span>
|
<span class='ocrx_word' id='word_1_170' title='bbox 346 1379 411 1406; x_wconf 96'>then</span>
|
||||||
<span class='ocrx_word' id='word_1_171' title='bbox 422 1378 483 1412; x_wconf 96'>play</span>
|
<span class='ocrx_word' id='word_1_171' title='bbox 422 1378 483 1412; x_wconf 96'>play</span>
|
||||||
<span class='ocrx_word' id='word_1_172' title='bbox 493 1387 562 1412; x_wconf 96'>your</span>
|
<span class='ocrx_word' id='word_1_172' title='bbox 493 1387 562 1412; x_wconf 96'>your</span>
|
||||||
<span class='ocrx_word' id='word_1_173' title='bbox 572 1379 646 1405; x_wconf 96'>MIDI</span>
|
<span class='ocrx_word' id='word_1_173' title='bbox 572 1379 659 1405; x_wconf 96'>MIDI</span>
|
||||||
<span class='ocrx_word' id='word_1_174' title='bbox 649 1379 792 1412; x_wconf 96'>keyboard</span>
|
<span class='ocrx_word' id='word_1_174' title='bbox 671 1379 810 1412; x_wconf 96'>keyboard</span>
|
||||||
<span class='ocrx_word' id='word_1_175' title='bbox 792 1379 848 1406; x_wconf 95'>in</span>
|
<span class='ocrx_word' id='word_1_175' title='bbox 821 1379 848 1406; x_wconf 95'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_176' title='bbox 858 1379 923 1406; x_wconf 95'>time</span>
|
<span class='ocrx_word' id='word_1_176' title='bbox 858 1379 923 1406; x_wconf 95'>time</span>
|
||||||
<span class='ocrx_word' id='word_1_177' title='bbox 934 1384 963 1406; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_177' title='bbox 934 1384 963 1406; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_178' title='bbox 974 1379 1019 1406; x_wconf 93'>the</span>
|
<span class='ocrx_word' id='word_1_178' title='bbox 974 1379 1019 1406; x_wconf 93'>the</span>
|
||||||
@@ -294,14 +294,14 @@
|
|||||||
<span class='ocrx_word' id='word_1_184' title='bbox 676 1426 810 1452; x_wconf 96'>sequence</span>
|
<span class='ocrx_word' id='word_1_184' title='bbox 676 1426 810 1452; x_wconf 96'>sequence</span>
|
||||||
<span class='ocrx_word' id='word_1_185' title='bbox 821 1419 901 1452; x_wconf 96'>loops</span>
|
<span class='ocrx_word' id='word_1_185' title='bbox 821 1419 901 1452; x_wconf 96'>loops</span>
|
||||||
<span class='ocrx_word' id='word_1_186' title='bbox 912 1419 983 1446; x_wconf 96'>back</span>
|
<span class='ocrx_word' id='word_1_186' title='bbox 912 1419 983 1446; x_wconf 96'>back</span>
|
||||||
<span class='ocrx_word' id='word_1_187' title='bbox 995 1427 1082 1446; x_wconf 96'>around</span>
|
<span class='ocrx_word' id='word_1_187' title='bbox 995 1419 1101 1446; x_wconf 96'>around</span>
|
||||||
<span class='ocrx_word' id='word_1_188' title='bbox 1084 1419 1141 1446; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_188' title='bbox 1112 1423 1141 1446; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_189' title='bbox 1152 1419 1189 1446; x_wconf 96'>bar</span>
|
<span class='ocrx_word' id='word_1_189' title='bbox 1152 1419 1201 1446; x_wconf 96'>bar</span>
|
||||||
<span class='ocrx_word' id='word_1_190' title='bbox 1189 1419 1232 1450; x_wconf 74'>1,</span>
|
<span class='ocrx_word' id='word_1_190' title='bbox 1213 1419 1232 1450; x_wconf 74'>1,</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_23' title="bbox 346 1457 1223 1491; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_23' title="bbox 346 1457 1223 1491; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_191' title='bbox 346 1465 400 1490; x_wconf 14'>you’</span>
|
<span class='ocrx_word' id='word_1_191' title='bbox 346 1457 430 1490; x_wconf 14'>you’</span>
|
||||||
<span class='ocrx_word' id='word_1_192' title='bbox 404 1457 430 1484; x_wconf 14'>ll</span>
|
<span class='ocrx_word' id='word_1_192' title='bbox 406 1453 436 1496; x_wconf 14'>ll</span>
|
||||||
<span class='ocrx_word' id='word_1_193' title='bbox 441 1457 506 1485; x_wconf 96'>hear</span>
|
<span class='ocrx_word' id='word_1_193' title='bbox 441 1457 506 1485; x_wconf 96'>hear</span>
|
||||||
<span class='ocrx_word' id='word_1_194' title='bbox 517 1458 590 1485; x_wconf 96'>what</span>
|
<span class='ocrx_word' id='word_1_194' title='bbox 517 1458 590 1485; x_wconf 96'>what</span>
|
||||||
<span class='ocrx_word' id='word_1_195' title='bbox 600 1466 654 1491; x_wconf 93'>you</span>
|
<span class='ocrx_word' id='word_1_195' title='bbox 600 1466 654 1491; x_wconf 93'>you</span>
|
||||||
@@ -316,7 +316,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_13' title="bbox 346 1497 1245 1531">
|
<div class='ocr_carea' id='block_1_13' title="bbox 346 1497 1245 1531">
|
||||||
<p class='ocr_par' id='par_1_16' lang='eng' title="bbox 346 1497 1245 1531">
|
<p class='ocr_par' id='par_1_16' lang='eng' title="bbox 346 1497 1245 1531">
|
||||||
<span class='ocr_line' id='line_1_24' title="bbox 346 1497 1245 1531; baseline 0.001 -7; x_size 35; x_descenders 7; x_ascenders 9">
|
<span class='ocr_textfloat' id='line_1_24' title="bbox 346 1497 1245 1531; baseline 0.001 -7; x_size 35; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_202' title='bbox 346 1497 494 1524; x_wconf 96'>corrected!</span>
|
<span class='ocrx_word' id='word_1_202' title='bbox 346 1497 494 1524; x_wconf 96'>corrected!</span>
|
||||||
<span class='ocrx_word' id='word_1_203' title='bbox 508 1497 628 1530; x_wconf 96'>(Timing</span>
|
<span class='ocrx_word' id='word_1_203' title='bbox 508 1497 628 1530; x_wconf 96'>(Timing</span>
|
||||||
<span class='ocrx_word' id='word_1_204' title='bbox 638 1497 791 1525; x_wconf 95'>correction</span>
|
<span class='ocrx_word' id='word_1_204' title='bbox 638 1497 791 1525; x_wconf 95'>correction</span>
|
||||||
@@ -343,8 +343,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_219' title='bbox 1111 1537 1186 1564; x_wconf 96'>track</span>
|
<span class='ocrx_word' id='word_1_219' title='bbox 1111 1537 1186 1564; x_wconf 96'>track</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_26' title="bbox 347 1575 1052 1610; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_26' title="bbox 347 1575 1052 1610; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_220' title='bbox 347 1591 381 1594; x_wconf 0'>—</span>
|
<span class='ocrx_word' id='word_1_220' title='bbox 347 1591 372 1594; x_wconf 0'>—</span>
|
||||||
<span class='ocrx_word' id='word_1_221' title='bbox 382 1575 495 1609; x_wconf 0'>existing</span>
|
<span class='ocrx_word' id='word_1_221' title='bbox 371 1575 495 1609; x_wconf 0'>existing</span>
|
||||||
<span class='ocrx_word' id='word_1_222' title='bbox 505 1580 582 1603; x_wconf 96'>notes</span>
|
<span class='ocrx_word' id='word_1_222' title='bbox 505 1580 582 1603; x_wconf 96'>notes</span>
|
||||||
<span class='ocrx_word' id='word_1_223' title='bbox 593 1584 637 1603; x_wconf 96'>are</span>
|
<span class='ocrx_word' id='word_1_223' title='bbox 593 1584 637 1603; x_wconf 96'>are</span>
|
||||||
<span class='ocrx_word' id='word_1_224' title='bbox 648 1580 696 1603; x_wconf 97'>not</span>
|
<span class='ocrx_word' id='word_1_224' title='bbox 648 1580 696 1603; x_wconf 97'>not</span>
|
||||||
@@ -358,10 +358,10 @@
|
|||||||
<span class='ocr_line' id='line_1_27' title="bbox 384 1616 1199 1648; baseline 0.001 -6; x_size 32; x_descenders 5; x_ascenders 8">
|
<span class='ocr_line' id='line_1_27' title="bbox 384 1616 1199 1648; baseline 0.001 -6; x_size 32; x_descenders 5; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_228' title='bbox 384 1616 471 1642; x_wconf 96'>FAST</span>
|
<span class='ocrx_word' id='word_1_228' title='bbox 384 1616 471 1642; x_wconf 96'>FAST</span>
|
||||||
<span class='ocrx_word' id='word_1_229' title='bbox 481 1616 671 1648; x_wconf 96'>FORWARD,</span>
|
<span class='ocrx_word' id='word_1_229' title='bbox 481 1616 671 1648; x_wconf 96'>FORWARD,</span>
|
||||||
<span class='ocrx_word' id='word_1_230' title='bbox 684 1617 838 1643; x_wconf 95'>REWIND,</span>
|
<span class='ocrx_word' id='word_1_230' title='bbox 684 1617 844 1648; x_wconf 95'>REWIND,</span>
|
||||||
<span class='ocrx_word' id='word_1_231' title='bbox 839 1616 912 1648; x_wconf 95'>and</span>
|
<span class='ocrx_word' id='word_1_231' title='bbox 857 1616 912 1643; x_wconf 95'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_232' title='bbox 924 1616 1045 1643; x_wconf 96'>LOCATE</span>
|
<span class='ocrx_word' id='word_1_232' title='bbox 924 1616 1068 1643; x_wconf 96'>LOCATE</span>
|
||||||
<span class='ocrx_word' id='word_1_233' title='bbox 1046 1616 1199 1643; x_wconf 95'>controls</span>
|
<span class='ocrx_word' id='word_1_233' title='bbox 1079 1616 1199 1643; x_wconf 95'>controls</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_28' title="bbox 346 1655 1202 1689; baseline 0 -7; x_size 34; x_descenders 6; x_ascenders 9">
|
<span class='ocr_line' id='line_1_28' title="bbox 346 1655 1202 1689; baseline 0 -7; x_size 34; x_descenders 6; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_234' title='bbox 346 1663 409 1688; x_wconf 92'>may</span>
|
<span class='ocrx_word' id='word_1_234' title='bbox 346 1663 409 1688; x_wconf 92'>may</span>
|
||||||
@@ -385,8 +385,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_250' title='bbox 860 1696 897 1722; x_wconf 93'>To</span>
|
<span class='ocrx_word' id='word_1_250' title='bbox 860 1696 897 1722; x_wconf 93'>To</span>
|
||||||
<span class='ocrx_word' id='word_1_251' title='bbox 908 1695 1028 1722; x_wconf 93'>overdub</span>
|
<span class='ocrx_word' id='word_1_251' title='bbox 908 1695 1028 1722; x_wconf 93'>overdub</span>
|
||||||
<span class='ocrx_word' id='word_1_252' title='bbox 1039 1703 1056 1722; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_252' title='bbox 1039 1703 1056 1722; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_253' title='bbox 1066 1703 1099 1722; x_wconf 96'>new</span>
|
<span class='ocrx_word' id='word_1_253' title='bbox 1066 1703 1125 1722; x_wconf 96'>new</span>
|
||||||
<span class='ocrx_word' id='word_1_254' title='bbox 1100 1699 1204 1728; x_wconf 96'>part,</span>
|
<span class='ocrx_word' id='word_1_254' title='bbox 1135 1699 1204 1728; x_wconf 96'>part,</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_30' title="bbox 347 1733 1150 1768; baseline 0.001 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
<span class='ocr_line' id='line_1_30' title="bbox 347 1733 1150 1768; baseline 0.001 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_255' title='bbox 347 1733 426 1761; x_wconf 97'>select</span>
|
<span class='ocrx_word' id='word_1_255' title='bbox 347 1733 426 1761; x_wconf 97'>select</span>
|
||||||
@@ -401,8 +401,8 @@
|
|||||||
<span class='ocr_line' id='line_1_31' title="bbox 346 1773 1203 1808; baseline 0.001 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_31' title="bbox 346 1773 1203 1808; baseline 0.001 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_263' title='bbox 346 1774 448 1806; x_wconf 96'>record,</span>
|
<span class='ocrx_word' id='word_1_263' title='bbox 346 1774 448 1806; x_wconf 96'>record,</span>
|
||||||
<span class='ocrx_word' id='word_1_264' title='bbox 460 1774 506 1801; x_wconf 97'>the</span>
|
<span class='ocrx_word' id='word_1_264' title='bbox 460 1774 506 1801; x_wconf 97'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_265' title='bbox 517 1773 577 1801; x_wconf 96'>first</span>
|
<span class='ocrx_word' id='word_1_265' title='bbox 503 1769 577 1812; x_wconf 96'>first</span>
|
||||||
<span class='ocrx_word' id='word_1_266' title='bbox 581 1774 662 1801; x_wconf 96'>track</span>
|
<span class='ocrx_word' id='word_1_266' title='bbox 581 1774 658 1801; x_wconf 96'>track</span>
|
||||||
<span class='ocrx_word' id='word_1_267' title='bbox 673 1774 726 1801; x_wconf 96'>will</span>
|
<span class='ocrx_word' id='word_1_267' title='bbox 673 1774 726 1801; x_wconf 96'>will</span>
|
||||||
<span class='ocrx_word' id='word_1_268' title='bbox 736 1774 799 1807; x_wconf 96'>play</span>
|
<span class='ocrx_word' id='word_1_268' title='bbox 736 1774 799 1807; x_wconf 96'>play</span>
|
||||||
<span class='ocrx_word' id='word_1_269' title='bbox 809 1774 836 1801; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_269' title='bbox 809 1774 836 1801; x_wconf 96'>in</span>
|
||||||
@@ -412,12 +412,12 @@
|
|||||||
<span class='ocrx_word' id='word_1_273' title='bbox 1148 1782 1203 1807; x_wconf 96'>you</span>
|
<span class='ocrx_word' id='word_1_273' title='bbox 1148 1782 1203 1807; x_wconf 96'>you</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_32' title="bbox 346 1813 1191 1847; baseline 0.002 -8; x_size 33; x_descenders 5; x_ascenders 9">
|
<span class='ocr_line' id='line_1_32' title="bbox 346 1813 1191 1847; baseline 0.002 -8; x_size 33; x_descenders 5; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_274' title='bbox 346 1813 431 1840; x_wconf 95'>MUTE</span>
|
<span class='ocrx_word' id='word_1_274' title='bbox 346 1813 454 1840; x_wconf 95'>MUTE</span>
|
||||||
<span class='ocrx_word' id='word_1_275' title='bbox 431 1813 492 1845; x_wconf 95'>it,</span>
|
<span class='ocrx_word' id='word_1_275' title='bbox 464 1813 492 1845; x_wconf 95'>it,</span>
|
||||||
<span class='ocrx_word' id='word_1_276' title='bbox 505 1821 537 1840; x_wconf 95'>or</span>
|
<span class='ocrx_word' id='word_1_276' title='bbox 505 1821 537 1840; x_wconf 95'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_277' title='bbox 547 1813 642 1840; x_wconf 95'>SOLO</span>
|
<span class='ocrx_word' id='word_1_277' title='bbox 547 1813 642 1840; x_wconf 95'>SOLO</span>
|
||||||
<span class='ocrx_word' id='word_1_278' title='bbox 653 1814 756 1841; x_wconf 95'>another</span>
|
<span class='ocrx_word' id='word_1_278' title='bbox 653 1814 769 1841; x_wconf 95'>another</span>
|
||||||
<span class='ocrx_word' id='word_1_279' title='bbox 757 1814 875 1847; x_wconf 94'>track).</span>
|
<span class='ocrx_word' id='word_1_279' title='bbox 779 1814 875 1847; x_wconf 94'>track).</span>
|
||||||
<span class='ocrx_word' id='word_1_280' title='bbox 889 1814 920 1840; x_wconf 96'>In</span>
|
<span class='ocrx_word' id='word_1_280' title='bbox 889 1814 920 1840; x_wconf 96'>In</span>
|
||||||
<span class='ocrx_word' id='word_1_281' title='bbox 930 1813 984 1841; x_wconf 96'>this</span>
|
<span class='ocrx_word' id='word_1_281' title='bbox 930 1813 984 1841; x_wconf 96'>this</span>
|
||||||
<span class='ocrx_word' id='word_1_282' title='bbox 995 1822 1057 1847; x_wconf 96'>way,</span>
|
<span class='ocrx_word' id='word_1_282' title='bbox 995 1822 1057 1847; x_wconf 96'>way,</span>
|
||||||
@@ -431,8 +431,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_288' title='bbox 518 1853 552 1879; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_288' title='bbox 518 1853 552 1879; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_289' title='bbox 562 1853 748 1880; x_wconf 96'>overdubbed!</span>
|
<span class='ocrx_word' id='word_1_289' title='bbox 562 1853 748 1880; x_wconf 96'>overdubbed!</span>
|
||||||
<span class='ocrx_word' id='word_1_290' title='bbox 761 1853 808 1880; x_wconf 94'>All</span>
|
<span class='ocrx_word' id='word_1_290' title='bbox 761 1853 808 1880; x_wconf 94'>All</span>
|
||||||
<span class='ocrx_word' id='word_1_291' title='bbox 819 1854 892 1880; x_wconf 96'>MIDI</span>
|
<span class='ocrx_word' id='word_1_291' title='bbox 819 1854 905 1880; x_wconf 96'>MIDI</span>
|
||||||
<span class='ocrx_word' id='word_1_292' title='bbox 895 1853 1011 1880; x_wconf 96'>effects</span>
|
<span class='ocrx_word' id='word_1_292' title='bbox 917 1853 1011 1880; x_wconf 96'>effects</span>
|
||||||
<span class='ocrx_word' id='word_1_293' title='bbox 1022 1861 1067 1880; x_wconf 96'>are</span>
|
<span class='ocrx_word' id='word_1_293' title='bbox 1022 1861 1067 1880; x_wconf 96'>are</span>
|
||||||
<span class='ocrx_word' id='word_1_294' title='bbox 1076 1853 1205 1880; x_wconf 96'>recorded</span>
|
<span class='ocrx_word' id='word_1_294' title='bbox 1076 1853 1205 1880; x_wconf 96'>recorded</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -440,8 +440,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_295' title='bbox 346 1891 485 1925; x_wconf 96'>including</span>
|
<span class='ocrx_word' id='word_1_295' title='bbox 346 1891 485 1925; x_wconf 96'>including</span>
|
||||||
<span class='ocrx_word' id='word_1_296' title='bbox 495 1891 570 1925; x_wconf 96'>pitch</span>
|
<span class='ocrx_word' id='word_1_296' title='bbox 495 1891 570 1925; x_wconf 96'>pitch</span>
|
||||||
<span class='ocrx_word' id='word_1_297' title='bbox 580 1892 663 1924; x_wconf 96'>bend,</span>
|
<span class='ocrx_word' id='word_1_297' title='bbox 580 1892 663 1924; x_wconf 96'>bend,</span>
|
||||||
<span class='ocrx_word' id='word_1_298' title='bbox 675 1892 850 1920; x_wconf 96'>modulation,</span>
|
<span class='ocrx_word' id='word_1_298' title='bbox 675 1892 859 1924; x_wconf 96'>modulation,</span>
|
||||||
<span class='ocrx_word' id='word_1_299' title='bbox 854 1892 991 1926; x_wconf 93'>velocity,</span>
|
<span class='ocrx_word' id='word_1_299' title='bbox 872 1892 991 1926; x_wconf 93'>velocity,</span>
|
||||||
<span class='ocrx_word' id='word_1_300' title='bbox 1004 1892 1168 1924; x_wconf 92'>aftertouch,</span>
|
<span class='ocrx_word' id='word_1_300' title='bbox 1004 1892 1168 1924; x_wconf 92'>aftertouch,</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_35' title="bbox 346 1931 895 1965; baseline 0.002 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_35' title="bbox 346 1931 895 1965; baseline 0.002 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
@@ -465,13 +465,13 @@
|
|||||||
<span class='ocrx_word' id='word_1_307' title='bbox 383 2050 419 2076; x_wconf 96'>To</span>
|
<span class='ocrx_word' id='word_1_307' title='bbox 383 2050 419 2076; x_wconf 96'>To</span>
|
||||||
<span class='ocrx_word' id='word_1_308' title='bbox 430 2057 503 2076; x_wconf 95'>erase</span>
|
<span class='ocrx_word' id='word_1_308' title='bbox 430 2057 503 2076; x_wconf 95'>erase</span>
|
||||||
<span class='ocrx_word' id='word_1_309' title='bbox 514 2058 530 2077; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_309' title='bbox 514 2058 530 2077; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_310' title='bbox 540 2058 617 2077; x_wconf 96'>wrong</span>
|
<span class='ocrx_word' id='word_1_310' title='bbox 540 2058 634 2083; x_wconf 96'>wrong</span>
|
||||||
<span class='ocrx_word' id='word_1_311' title='bbox 618 2054 717 2083; x_wconf 96'>note,</span>
|
<span class='ocrx_word' id='word_1_311' title='bbox 644 2054 717 2082; x_wconf 96'>note,</span>
|
||||||
<span class='ocrx_word' id='word_1_312' title='bbox 729 2050 829 2083; x_wconf 96'>simply</span>
|
<span class='ocrx_word' id='word_1_312' title='bbox 729 2050 829 2083; x_wconf 96'>simply</span>
|
||||||
<span class='ocrx_word' id='word_1_313' title='bbox 839 2050 905 2077; x_wconf 96'>hold</span>
|
<span class='ocrx_word' id='word_1_313' title='bbox 839 2050 905 2077; x_wconf 96'>hold</span>
|
||||||
<span class='ocrx_word' id='word_1_314' title='bbox 916 2051 1013 2077; x_wconf 96'>ERASE</span>
|
<span class='ocrx_word' id='word_1_314' title='bbox 916 2051 1037 2077; x_wconf 96'>ERASE</span>
|
||||||
<span class='ocrx_word' id='word_1_315' title='bbox 1015 2051 1083 2077; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_315' title='bbox 1048 2050 1103 2077; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_316' title='bbox 1085 2050 1186 2084; x_wconf 96'>press</span>
|
<span class='ocrx_word' id='word_1_316' title='bbox 1113 2059 1186 2084; x_wconf 96'>press</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_38' title="bbox 346 2089 1212 2124; baseline 0.002 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_38' title="bbox 346 2089 1212 2124; baseline 0.002 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_317' title='bbox 346 2089 391 2116; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_317' title='bbox 346 2089 391 2116; x_wconf 96'>the</span>
|
||||||
@@ -480,16 +480,16 @@
|
|||||||
<span class='ocrx_word' id='word_1_320' title='bbox 515 2090 549 2117; x_wconf 97'>be</span>
|
<span class='ocrx_word' id='word_1_320' title='bbox 515 2090 549 2117; x_wconf 97'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_321' title='bbox 559 2090 652 2117; x_wconf 96'>erased</span>
|
<span class='ocrx_word' id='word_1_321' title='bbox 559 2090 652 2117; x_wconf 96'>erased</span>
|
||||||
<span class='ocrx_word' id='word_1_322' title='bbox 661 2090 718 2123; x_wconf 96'>just</span>
|
<span class='ocrx_word' id='word_1_322' title='bbox 661 2090 718 2123; x_wconf 96'>just</span>
|
||||||
<span class='ocrx_word' id='word_1_323' title='bbox 729 2090 808 2117; x_wconf 96'>before</span>
|
<span class='ocrx_word' id='word_1_323' title='bbox 729 2090 822 2117; x_wconf 96'>before</span>
|
||||||
<span class='ocrx_word' id='word_1_324' title='bbox 808 2090 852 2117; x_wconf 96'>it</span>
|
<span class='ocrx_word' id='word_1_324' title='bbox 833 2090 852 2117; x_wconf 96'>it</span>
|
||||||
<span class='ocrx_word' id='word_1_325' title='bbox 862 2090 937 2124; x_wconf 96'>plays</span>
|
<span class='ocrx_word' id='word_1_325' title='bbox 862 2090 937 2124; x_wconf 96'>plays</span>
|
||||||
<span class='ocrx_word' id='word_1_326' title='bbox 947 2090 975 2117; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_326' title='bbox 947 2090 975 2117; x_wconf 96'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_327' title='bbox 986 2090 1032 2118; x_wconf 93'>the</span>
|
<span class='ocrx_word' id='word_1_327' title='bbox 986 2090 1032 2118; x_wconf 93'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_328' title='bbox 1043 2098 1212 2124; x_wconf 88'>sequence—</span>
|
<span class='ocrx_word' id='word_1_328' title='bbox 1043 2098 1212 2124; x_wconf 88'>sequence—</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_39' title="bbox 346 2129 1134 2163; baseline 0.003 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_39' title="bbox 346 2129 1134 2163; baseline 0.003 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_329' title='bbox 346 2129 406 2156; x_wconf 96'>when</span>
|
<span class='ocrx_word' id='word_1_329' title='bbox 346 2129 425 2156; x_wconf 96'>when</span>
|
||||||
<span class='ocrx_word' id='word_1_330' title='bbox 407 2129 531 2162; x_wconf 96'>played</span>
|
<span class='ocrx_word' id='word_1_330' title='bbox 435 2129 531 2162; x_wconf 96'>played</span>
|
||||||
<span class='ocrx_word' id='word_1_331' title='bbox 542 2129 621 2161; x_wconf 96'>back,</span>
|
<span class='ocrx_word' id='word_1_331' title='bbox 542 2129 621 2161; x_wconf 96'>back,</span>
|
||||||
<span class='ocrx_word' id='word_1_332' title='bbox 633 2129 652 2156; x_wconf 96'>it</span>
|
<span class='ocrx_word' id='word_1_332' title='bbox 633 2129 652 2156; x_wconf 96'>it</span>
|
||||||
<span class='ocrx_word' id='word_1_333' title='bbox 663 2129 716 2156; x_wconf 96'>will</span>
|
<span class='ocrx_word' id='word_1_333' title='bbox 663 2129 716 2156; x_wconf 96'>will</span>
|
||||||
@@ -512,15 +512,15 @@
|
|||||||
<span class='ocrx_word' id='word_1_344' title='bbox 749 2169 829 2203; x_wconf 96'>using</span>
|
<span class='ocrx_word' id='word_1_344' title='bbox 749 2169 829 2203; x_wconf 96'>using</span>
|
||||||
<span class='ocrx_word' id='word_1_345' title='bbox 839 2169 885 2196; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_345' title='bbox 839 2169 885 2196; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_346' title='bbox 896 2170 1031 2196; x_wconf 96'>SINGLE</span>
|
<span class='ocrx_word' id='word_1_346' title='bbox 896 2170 1031 2196; x_wconf 96'>SINGLE</span>
|
||||||
<span class='ocrx_word' id='word_1_347' title='bbox 1042 2170 1107 2196; x_wconf 91'>STEP</span>
|
<span class='ocrx_word' id='word_1_347' title='bbox 1042 2170 1131 2196; x_wconf 91'>STEP</span>
|
||||||
<span class='ocrx_word' id='word_1_348' title='bbox 1109 2169 1220 2196; x_wconf 91'>func-</span>
|
<span class='ocrx_word' id='word_1_348' title='bbox 1143 2169 1220 2196; x_wconf 91'>func-</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_41' title="bbox 345 2207 1228 2242; baseline 0.002 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
<span class='ocr_line' id='line_1_41' title="bbox 345 2207 1228 2242; baseline 0.002 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_349' title='bbox 345 2207 412 2234; x_wconf 96'>tion.</span>
|
<span class='ocrx_word' id='word_1_349' title='bbox 345 2207 412 2234; x_wconf 96'>tion.</span>
|
||||||
<span class='ocrx_word' id='word_1_350' title='bbox 424 2208 461 2235; x_wconf 93'>To</span>
|
<span class='ocrx_word' id='word_1_350' title='bbox 424 2208 461 2235; x_wconf 93'>To</span>
|
||||||
<span class='ocrx_word' id='word_1_351' title='bbox 472 2208 592 2235; x_wconf 91'>overdub</span>
|
<span class='ocrx_word' id='word_1_351' title='bbox 472 2208 592 2235; x_wconf 91'>overdub</span>
|
||||||
<span class='ocrx_word' id='word_1_352' title='bbox 603 2212 667 2235; x_wconf 96'>notes</span>
|
<span class='ocrx_word' id='word_1_352' title='bbox 603 2212 680 2235; x_wconf 96'>notes</span>
|
||||||
<span class='ocrx_word' id='word_1_353' title='bbox 668 2212 718 2235; x_wconf 96'>at</span>
|
<span class='ocrx_word' id='word_1_353' title='bbox 691 2212 718 2235; x_wconf 96'>at</span>
|
||||||
<span class='ocrx_word' id='word_1_354' title='bbox 729 2208 841 2242; x_wconf 96'>specific</span>
|
<span class='ocrx_word' id='word_1_354' title='bbox 729 2208 841 2242; x_wconf 96'>specific</span>
|
||||||
<span class='ocrx_word' id='word_1_355' title='bbox 851 2209 943 2242; x_wconf 97'>points</span>
|
<span class='ocrx_word' id='word_1_355' title='bbox 851 2209 943 2242; x_wconf 97'>points</span>
|
||||||
<span class='ocrx_word' id='word_1_356' title='bbox 955 2208 1049 2236; x_wconf 96'>within</span>
|
<span class='ocrx_word' id='word_1_356' title='bbox 955 2208 1049 2236; x_wconf 96'>within</span>
|
||||||
@@ -544,10 +544,10 @@
|
|||||||
<span class='ocrx_word' id='word_1_362' title='bbox 1404 1297 1452 1316; x_wconf 96'>use</span>
|
<span class='ocrx_word' id='word_1_362' title='bbox 1404 1297 1452 1316; x_wconf 96'>use</span>
|
||||||
<span class='ocrx_word' id='word_1_363' title='bbox 1463 1290 1615 1321; x_wconf 96'>LOCATE,</span>
|
<span class='ocrx_word' id='word_1_363' title='bbox 1463 1290 1615 1321; x_wconf 96'>LOCATE,</span>
|
||||||
<span class='ocrx_word' id='word_1_364' title='bbox 1628 1290 1716 1316; x_wconf 96'>FAST</span>
|
<span class='ocrx_word' id='word_1_364' title='bbox 1628 1290 1716 1316; x_wconf 96'>FAST</span>
|
||||||
<span class='ocrx_word' id='word_1_365' title='bbox 1726 1289 1910 1316; x_wconf 95'>FORWARD,</span>
|
<span class='ocrx_word' id='word_1_365' title='bbox 1726 1289 1917 1321; x_wconf 95'>FORWARD,</span>
|
||||||
<span class='ocrx_word' id='word_1_366' title='bbox 1911 1297 1960 1321; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_366' title='bbox 1929 1297 1960 1317; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_367' title='bbox 1972 1290 2098 1316; x_wconf 95'>REWIND</span>
|
<span class='ocrx_word' id='word_1_367' title='bbox 1972 1290 2126 1316; x_wconf 95'>REWIND</span>
|
||||||
<span class='ocrx_word' id='word_1_368' title='bbox 2100 1290 2165 1317; x_wconf 95'>to</span>
|
<span class='ocrx_word' id='word_1_368' title='bbox 2136 1294 2165 1317; x_wconf 95'>to</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_44' title="bbox 1297 1328 2033 1362; baseline 0.001 -7; x_size 32; x_descenders 5; x_ascenders 8">
|
<span class='ocr_line' id='line_1_44' title="bbox 1297 1328 2033 1362; baseline 0.001 -7; x_size 32; x_descenders 5; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_369' title='bbox 1297 1328 1356 1355; x_wconf 96'>find</span>
|
<span class='ocrx_word' id='word_1_369' title='bbox 1297 1328 1356 1355; x_wconf 96'>find</span>
|
||||||
@@ -564,8 +564,8 @@
|
|||||||
<p class='ocr_par' id='par_1_24' lang='eng' title="bbox 1295 1368 2160 1593">
|
<p class='ocr_par' id='par_1_24' lang='eng' title="bbox 1295 1368 2160 1593">
|
||||||
<span class='ocr_line' id='line_1_45' title="bbox 1332 1368 2160 1402; baseline 0.002 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_45' title="bbox 1332 1368 2160 1402; baseline 0.002 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_377' title='bbox 1332 1368 1391 1395; x_wconf 93'>The</span>
|
<span class='ocrx_word' id='word_1_377' title='bbox 1332 1368 1391 1395; x_wconf 93'>The</span>
|
||||||
<span class='ocrx_word' id='word_1_378' title='bbox 1402 1368 1626 1395; x_wconf 91'>INSERT/COPY</span>
|
<span class='ocrx_word' id='word_1_378' title='bbox 1402 1368 1650 1395; x_wconf 91'>INSERT/COPY</span>
|
||||||
<span class='ocrx_word' id='word_1_379' title='bbox 1626 1368 1789 1395; x_wconf 96'>function</span>
|
<span class='ocrx_word' id='word_1_379' title='bbox 1662 1368 1789 1395; x_wconf 96'>function</span>
|
||||||
<span class='ocrx_word' id='word_1_380' title='bbox 1800 1368 1893 1395; x_wconf 96'>allows</span>
|
<span class='ocrx_word' id='word_1_380' title='bbox 1800 1368 1893 1395; x_wconf 96'>allows</span>
|
||||||
<span class='ocrx_word' id='word_1_381' title='bbox 1904 1376 1958 1402; x_wconf 96'>you</span>
|
<span class='ocrx_word' id='word_1_381' title='bbox 1904 1376 1958 1402; x_wconf 96'>you</span>
|
||||||
<span class='ocrx_word' id='word_1_382' title='bbox 1968 1373 1997 1395; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_382' title='bbox 1968 1373 1997 1395; x_wconf 96'>to</span>
|
||||||
@@ -580,8 +580,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_389' title='bbox 1616 1407 1796 1435; x_wconf 91'>another—in</span>
|
<span class='ocrx_word' id='word_1_389' title='bbox 1616 1407 1796 1435; x_wconf 91'>another—in</span>
|
||||||
<span class='ocrx_word' id='word_1_390' title='bbox 1806 1408 1852 1435; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_390' title='bbox 1806 1408 1852 1435; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_391' title='bbox 1863 1416 1937 1435; x_wconf 96'>same</span>
|
<span class='ocrx_word' id='word_1_391' title='bbox 1863 1416 1937 1435; x_wconf 96'>same</span>
|
||||||
<span class='ocrx_word' id='word_1_392' title='bbox 1948 1416 2067 1441; x_wconf 96'>sequence</span>
|
<span class='ocrx_word' id='word_1_392' title='bbox 1948 1416 2083 1441; x_wconf 96'>sequence</span>
|
||||||
<span class='ocrx_word' id='word_1_393' title='bbox 2068 1416 2125 1435; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_393' title='bbox 2093 1416 2125 1435; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_394' title='bbox 2135 1416 2151 1435; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_394' title='bbox 2135 1416 2151 1435; x_wconf 96'>a</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_47' title="bbox 1296 1447 2160 1481; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_47' title="bbox 1296 1447 2160 1481; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
@@ -603,15 +603,15 @@
|
|||||||
<span class='ocrx_word' id='word_1_408' title='bbox 1449 1487 1571 1514; x_wconf 96'>between</span>
|
<span class='ocrx_word' id='word_1_408' title='bbox 1449 1487 1571 1514; x_wconf 96'>between</span>
|
||||||
<span class='ocrx_word' id='word_1_409' title='bbox 1582 1487 1628 1514; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_409' title='bbox 1582 1487 1628 1514; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_410' title='bbox 1638 1487 1740 1514; x_wconf 96'>second</span>
|
<span class='ocrx_word' id='word_1_410' title='bbox 1638 1487 1740 1514; x_wconf 96'>second</span>
|
||||||
<span class='ocrx_word' id='word_1_411' title='bbox 1751 1487 1838 1514; x_wconf 96'>chorus</span>
|
<span class='ocrx_word' id='word_1_411' title='bbox 1751 1487 1852 1514; x_wconf 96'>chorus</span>
|
||||||
<span class='ocrx_word' id='word_1_412' title='bbox 1841 1495 1899 1514; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_412' title='bbox 1863 1487 1919 1514; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_413' title='bbox 1901 1487 1975 1514; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_413' title='bbox 1929 1487 1975 1514; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_414' title='bbox 1985 1487 2087 1521; x_wconf 96'>bridge.</span>
|
<span class='ocrx_word' id='word_1_414' title='bbox 1985 1487 2087 1521; x_wconf 96'>bridge.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_49' title="bbox 1296 1527 2047 1560; baseline 0.003 -8; x_size 32; x_descenders 6; x_ascenders 7">
|
<span class='ocr_line' id='line_1_49' title="bbox 1296 1527 2047 1560; baseline 0.003 -8; x_size 32; x_descenders 6; x_ascenders 7">
|
||||||
<span class='ocrx_word' id='word_1_415' title='bbox 1296 1527 1441 1553; x_wconf 95'>DELETE</span>
|
<span class='ocrx_word' id='word_1_415' title='bbox 1296 1527 1441 1553; x_wconf 95'>DELETE</span>
|
||||||
<span class='ocrx_word' id='word_1_416' title='bbox 1453 1527 1527 1553; x_wconf 96'>BARS</span>
|
<span class='ocrx_word' id='word_1_416' title='bbox 1453 1527 1546 1553; x_wconf 96'>BARS</span>
|
||||||
<span class='ocrx_word' id='word_1_417' title='bbox 1529 1527 1681 1559; x_wconf 96'>operates</span>
|
<span class='ocrx_word' id='word_1_417' title='bbox 1557 1531 1681 1559; x_wconf 96'>operates</span>
|
||||||
<span class='ocrx_word' id='word_1_418' title='bbox 1691 1527 1737 1553; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_418' title='bbox 1691 1527 1737 1553; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_419' title='bbox 1748 1535 1823 1554; x_wconf 96'>same</span>
|
<span class='ocrx_word' id='word_1_419' title='bbox 1748 1535 1823 1554; x_wconf 96'>same</span>
|
||||||
<span class='ocrx_word' id='word_1_420' title='bbox 1833 1535 1891 1560; x_wconf 96'>way</span>
|
<span class='ocrx_word' id='word_1_420' title='bbox 1833 1535 1891 1560; x_wconf 96'>way</span>
|
||||||
@@ -619,8 +619,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_422' title='bbox 1940 1535 2047 1554; x_wconf 96'>remove</span>
|
<span class='ocrx_word' id='word_1_422' title='bbox 1940 1535 2047 1554; x_wconf 96'>remove</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_50' title="bbox 1295 1565 1577 1593; baseline 0.004 -1; x_size 34.748871; x_descenders 6.7488689; x_ascenders 9">
|
<span class='ocr_line' id='line_1_50' title="bbox 1295 1565 1577 1593; baseline 0.004 -1; x_size 34.748871; x_descenders 6.7488689; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_423' title='bbox 1295 1569 1422 1592; x_wconf 96'>unwanted</span>
|
<span class='ocrx_word' id='word_1_423' title='bbox 1295 1565 1441 1592; x_wconf 96'>unwanted</span>
|
||||||
<span class='ocrx_word' id='word_1_424' title='bbox 1423 1565 1577 1593; x_wconf 95'>sections,</span>
|
<span class='ocrx_word' id='word_1_424' title='bbox 1452 1565 1577 1593; x_wconf 95'>sections,</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
@@ -636,12 +636,12 @@
|
|||||||
<p class='ocr_par' id='par_1_26' lang='eng' title="bbox 1295 1690 2215 1960">
|
<p class='ocr_par' id='par_1_26' lang='eng' title="bbox 1295 1690 2215 1960">
|
||||||
<span class='ocr_line' id='line_1_52' title="bbox 1333 1690 2146 1723; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_52' title="bbox 1333 1690 2146 1723; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_428' title='bbox 1333 1690 1394 1717; x_wconf 96'>One</span>
|
<span class='ocrx_word' id='word_1_428' title='bbox 1333 1690 1394 1717; x_wconf 96'>One</span>
|
||||||
<span class='ocrx_word' id='word_1_429' title='bbox 1404 1698 1446 1717; x_wconf 96'>way</span>
|
<span class='ocrx_word' id='word_1_429' title='bbox 1404 1698 1462 1723; x_wconf 96'>way</span>
|
||||||
<span class='ocrx_word' id='word_1_430' title='bbox 1445 1694 1500 1723; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_430' title='bbox 1472 1694 1500 1717; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_431' title='bbox 1511 1694 1598 1717; x_wconf 96'>create</span>
|
<span class='ocrx_word' id='word_1_431' title='bbox 1511 1694 1598 1717; x_wconf 96'>create</span>
|
||||||
<span class='ocrx_word' id='word_1_432' title='bbox 1608 1698 1625 1717; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_432' title='bbox 1608 1698 1625 1717; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_433' title='bbox 1635 1698 1687 1717; x_wconf 95'>song</span>
|
<span class='ocrx_word' id='word_1_433' title='bbox 1635 1698 1704 1723; x_wconf 95'>song</span>
|
||||||
<span class='ocrx_word' id='word_1_434' title='bbox 1688 1690 1736 1723; x_wconf 95'>is</span>
|
<span class='ocrx_word' id='word_1_434' title='bbox 1715 1690 1736 1717; x_wconf 95'>is</span>
|
||||||
<span class='ocrx_word' id='word_1_435' title='bbox 1747 1694 1776 1717; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_435' title='bbox 1747 1694 1776 1717; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_436' title='bbox 1787 1690 1880 1717; x_wconf 96'>record</span>
|
<span class='ocrx_word' id='word_1_436' title='bbox 1787 1690 1880 1717; x_wconf 96'>record</span>
|
||||||
<span class='ocrx_word' id='word_1_437' title='bbox 1891 1690 1958 1717; x_wconf 96'>each</span>
|
<span class='ocrx_word' id='word_1_437' title='bbox 1891 1690 1958 1717; x_wconf 96'>each</span>
|
||||||
@@ -657,8 +657,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_445' title='bbox 1592 1730 1644 1756; x_wconf 96'>999</span>
|
<span class='ocrx_word' id='word_1_445' title='bbox 1592 1730 1644 1756; x_wconf 96'>999</span>
|
||||||
<span class='ocrx_word' id='word_1_446' title='bbox 1654 1729 1738 1762; x_wconf 96'>bars).</span>
|
<span class='ocrx_word' id='word_1_446' title='bbox 1654 1729 1738 1762; x_wconf 96'>bars).</span>
|
||||||
<span class='ocrx_word' id='word_1_447' title='bbox 1751 1729 1878 1757; x_wconf 96'>Another</span>
|
<span class='ocrx_word' id='word_1_447' title='bbox 1751 1729 1878 1757; x_wconf 96'>Another</span>
|
||||||
<span class='ocrx_word' id='word_1_448' title='bbox 1888 1737 1930 1757; x_wconf 96'>way</span>
|
<span class='ocrx_word' id='word_1_448' title='bbox 1888 1737 1945 1762; x_wconf 96'>way</span>
|
||||||
<span class='ocrx_word' id='word_1_449' title='bbox 1929 1729 1977 1762; x_wconf 96'>is</span>
|
<span class='ocrx_word' id='word_1_449' title='bbox 1956 1729 1977 1757; x_wconf 96'>is</span>
|
||||||
<span class='ocrx_word' id='word_1_450' title='bbox 1987 1733 2016 1757; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_450' title='bbox 1987 1733 2016 1757; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_451' title='bbox 2027 1729 2121 1757; x_wconf 96'>record</span>
|
<span class='ocrx_word' id='word_1_451' title='bbox 2027 1729 2121 1757; x_wconf 96'>record</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -667,8 +667,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_453' title='bbox 1373 1768 1448 1796; x_wconf 96'>basic</span>
|
<span class='ocrx_word' id='word_1_453' title='bbox 1373 1768 1448 1796; x_wconf 96'>basic</span>
|
||||||
<span class='ocrx_word' id='word_1_454' title='bbox 1458 1768 1562 1796; x_wconf 96'>section</span>
|
<span class='ocrx_word' id='word_1_454' title='bbox 1458 1768 1562 1796; x_wconf 96'>section</span>
|
||||||
<span class='ocrx_word' id='word_1_455' title='bbox 1574 1769 1666 1802; x_wconf 96'>(verse,</span>
|
<span class='ocrx_word' id='word_1_455' title='bbox 1574 1769 1666 1802; x_wconf 96'>(verse,</span>
|
||||||
<span class='ocrx_word' id='word_1_456' title='bbox 1679 1769 1779 1796; x_wconf 96'>chorus,</span>
|
<span class='ocrx_word' id='word_1_456' title='bbox 1679 1769 1788 1801; x_wconf 96'>chorus,</span>
|
||||||
<span class='ocrx_word' id='word_1_457' title='bbox 1782 1769 1865 1802; x_wconf 96'>etc.)</span>
|
<span class='ocrx_word' id='word_1_457' title='bbox 1800 1769 1865 1802; x_wconf 96'>etc.)</span>
|
||||||
<span class='ocrx_word' id='word_1_458' title='bbox 1876 1768 1904 1795; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_458' title='bbox 1876 1768 1904 1795; x_wconf 96'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_459' title='bbox 1914 1768 2066 1796; x_wconf 96'>individual</span>
|
<span class='ocrx_word' id='word_1_459' title='bbox 1914 1768 2066 1796; x_wconf 96'>individual</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -678,8 +678,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_462' title='bbox 1538 1816 1587 1835; x_wconf 96'>use</span>
|
<span class='ocrx_word' id='word_1_462' title='bbox 1538 1816 1587 1835; x_wconf 96'>use</span>
|
||||||
<span class='ocrx_word' id='word_1_463' title='bbox 1597 1808 1643 1835; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_463' title='bbox 1597 1808 1643 1835; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_464' title='bbox 1653 1809 1799 1835; x_wconf 96'>CREATE</span>
|
<span class='ocrx_word' id='word_1_464' title='bbox 1653 1809 1799 1835; x_wconf 96'>CREATE</span>
|
||||||
<span class='ocrx_word' id='word_1_465' title='bbox 1810 1808 1883 1835; x_wconf 96'>SONG</span>
|
<span class='ocrx_word' id='word_1_465' title='bbox 1810 1808 1911 1835; x_wconf 96'>SONG</span>
|
||||||
<span class='ocrx_word' id='word_1_466' title='bbox 1885 1808 2050 1836; x_wconf 96'>function</span>
|
<span class='ocrx_word' id='word_1_466' title='bbox 1923 1808 2050 1836; x_wconf 96'>function</span>
|
||||||
<span class='ocrx_word' id='word_1_467' title='bbox 2060 1812 2089 1835; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_467' title='bbox 2060 1812 2089 1835; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_468' title='bbox 2103 1808 2215 1836; x_wconf 93'>“chain”</span>
|
<span class='ocrx_word' id='word_1_468' title='bbox 2103 1808 2215 1836; x_wconf 93'>“chain”</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -687,14 +687,14 @@
|
|||||||
<span class='ocrx_word' id='word_1_469' title='bbox 1295 1848 1370 1874; x_wconf 96'>them</span>
|
<span class='ocrx_word' id='word_1_469' title='bbox 1295 1848 1370 1874; x_wconf 96'>them</span>
|
||||||
<span class='ocrx_word' id='word_1_470' title='bbox 1381 1848 1508 1881; x_wconf 95'>together.</span>
|
<span class='ocrx_word' id='word_1_470' title='bbox 1381 1848 1508 1881; x_wconf 95'>together.</span>
|
||||||
<span class='ocrx_word' id='word_1_471' title='bbox 1521 1848 1667 1875; x_wconf 96'>CREATE</span>
|
<span class='ocrx_word' id='word_1_471' title='bbox 1521 1848 1667 1875; x_wconf 96'>CREATE</span>
|
||||||
<span class='ocrx_word' id='word_1_472' title='bbox 1678 1848 1751 1875; x_wconf 96'>SONG</span>
|
<span class='ocrx_word' id='word_1_472' title='bbox 1678 1848 1779 1875; x_wconf 96'>SONG</span>
|
||||||
<span class='ocrx_word' id='word_1_473' title='bbox 1753 1847 1842 1874; x_wconf 96'>will</span>
|
<span class='ocrx_word' id='word_1_473' title='bbox 1789 1847 1842 1874; x_wconf 96'>will</span>
|
||||||
<span class='ocrx_word' id='word_1_474' title='bbox 1853 1848 1918 1875; x_wconf 96'>then</span>
|
<span class='ocrx_word' id='word_1_474' title='bbox 1853 1848 1918 1875; x_wconf 96'>then</span>
|
||||||
<span class='ocrx_word' id='word_1_475' title='bbox 1929 1848 2135 1881; x_wconf 96'>automatically</span>
|
<span class='ocrx_word' id='word_1_475' title='bbox 1929 1848 2135 1881; x_wconf 96'>automatically</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_57' title="bbox 1296 1887 2162 1920; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_57' title="bbox 1296 1887 2162 1920; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_476' title='bbox 1296 1895 1349 1920; x_wconf 96'>copy</span>
|
<span class='ocrx_word' id='word_1_476' title='bbox 1296 1895 1366 1920; x_wconf 96'>copy</span>
|
||||||
<span class='ocrx_word' id='word_1_477' title='bbox 1349 1887 1412 1920; x_wconf 96'>all</span>
|
<span class='ocrx_word' id='word_1_477' title='bbox 1377 1887 1412 1914; x_wconf 96'>all</span>
|
||||||
<span class='ocrx_word' id='word_1_478' title='bbox 1422 1887 1468 1914; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_478' title='bbox 1422 1887 1468 1914; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_479' title='bbox 1478 1891 1552 1920; x_wconf 96'>parts</span>
|
<span class='ocrx_word' id='word_1_479' title='bbox 1478 1891 1552 1920; x_wconf 96'>parts</span>
|
||||||
<span class='ocrx_word' id='word_1_480' title='bbox 1563 1887 1621 1914; x_wconf 95'>into</span>
|
<span class='ocrx_word' id='word_1_480' title='bbox 1563 1887 1621 1914; x_wconf 95'>into</span>
|
||||||
@@ -714,8 +714,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_492' title='bbox 1540 1926 1590 1953; x_wconf 96'>few</span>
|
<span class='ocrx_word' id='word_1_492' title='bbox 1540 1926 1590 1953; x_wconf 96'>few</span>
|
||||||
<span class='ocrx_word' id='word_1_493' title='bbox 1601 1927 1664 1953; x_wconf 96'>bars</span>
|
<span class='ocrx_word' id='word_1_493' title='bbox 1601 1927 1664 1953; x_wconf 96'>bars</span>
|
||||||
<span class='ocrx_word' id='word_1_494' title='bbox 1675 1931 1704 1954; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_494' title='bbox 1675 1931 1704 1954; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_495' title='bbox 1715 1935 1795 1960; x_wconf 96'>repeat</span>
|
<span class='ocrx_word' id='word_1_495' title='bbox 1715 1931 1806 1960; x_wconf 96'>repeat</span>
|
||||||
<span class='ocrx_word' id='word_1_496' title='bbox 1795 1926 1955 1960; x_wconf 96'>infinitely,</span>
|
<span class='ocrx_word' id='word_1_496' title='bbox 1816 1926 1955 1960; x_wconf 96'>infinitely,</span>
|
||||||
<span class='ocrx_word' id='word_1_497' title='bbox 1968 1926 2011 1954; x_wconf 95'>for</span>
|
<span class='ocrx_word' id='word_1_497' title='bbox 1968 1926 2011 1954; x_wconf 95'>for</span>
|
||||||
<span class='ocrx_word' id='word_1_498' title='bbox 2022 1935 2038 1954; x_wconf 93'>a</span>
|
<span class='ocrx_word' id='word_1_498' title='bbox 2022 1935 2038 1954; x_wconf 93'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_499' title='bbox 2049 1927 2169 1954; x_wconf 92'>fadeout.</span>
|
<span class='ocrx_word' id='word_1_499' title='bbox 2049 1927 2169 1954; x_wconf 92'>fadeout.</span>
|
||||||
@@ -757,8 +757,8 @@
|
|||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_62' title="bbox 1293 2130 2156 2164; baseline 0 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_62' title="bbox 1293 2130 2156 2164; baseline 0 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_522' title='bbox 1293 2130 1339 2157; x_wconf 93'>the</span>
|
<span class='ocrx_word' id='word_1_522' title='bbox 1293 2130 1339 2157; x_wconf 93'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_523' title='bbox 1350 2130 1562 2164; x_wconf 90'>LinnSequencer</span>
|
<span class='ocrx_word' id='word_1_523' title='bbox 1350 2130 1576 2164; x_wconf 90'>LinnSequencer</span>
|
||||||
<span class='ocrx_word' id='word_1_524' title='bbox 1564 2130 1607 2157; x_wconf 97'>is</span>
|
<span class='ocrx_word' id='word_1_524' title='bbox 1586 2130 1607 2157; x_wconf 97'>is</span>
|
||||||
<span class='ocrx_word' id='word_1_525' title='bbox 1619 2130 1747 2164; x_wconf 96'>designed</span>
|
<span class='ocrx_word' id='word_1_525' title='bbox 1619 2130 1747 2164; x_wconf 96'>designed</span>
|
||||||
<span class='ocrx_word' id='word_1_526' title='bbox 1758 2134 1787 2157; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_526' title='bbox 1758 2134 1787 2157; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_527' title='bbox 1798 2130 1834 2157; x_wconf 96'>let</span>
|
<span class='ocrx_word' id='word_1_527' title='bbox 1798 2130 1834 2157; x_wconf 96'>let</span>
|
||||||
@@ -767,8 +767,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_530' title='bbox 2063 2130 2156 2157; x_wconf 96'>record</span>
|
<span class='ocrx_word' id='word_1_530' title='bbox 2063 2130 2156 2157; x_wconf 96'>record</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_63' title="bbox 1294 2169 2145 2203; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_63' title="bbox 1294 2169 2145 2203; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_531' title='bbox 1294 2178 1330 2197; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_531' title='bbox 1294 2170 1349 2197; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_532' title='bbox 1331 2169 1414 2197; x_wconf 96'>edit</span>
|
<span class='ocrx_word' id='word_1_532' title='bbox 1360 2169 1414 2197; x_wconf 96'>edit</span>
|
||||||
<span class='ocrx_word' id='word_1_533' title='bbox 1425 2170 1503 2197; x_wconf 96'>while</span>
|
<span class='ocrx_word' id='word_1_533' title='bbox 1425 2170 1503 2197; x_wconf 96'>while</span>
|
||||||
<span class='ocrx_word' id='word_1_534' title='bbox 1514 2170 1643 2203; x_wconf 96'>devoting</span>
|
<span class='ocrx_word' id='word_1_534' title='bbox 1514 2170 1643 2203; x_wconf 96'>devoting</span>
|
||||||
<span class='ocrx_word' id='word_1_535' title='bbox 1653 2178 1721 2203; x_wconf 96'>your</span>
|
<span class='ocrx_word' id='word_1_535' title='bbox 1653 2178 1721 2203; x_wconf 96'>your</span>
|
||||||
@@ -792,7 +792,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_21' title="bbox 347 2343 2171 2378">
|
<div class='ocr_carea' id='block_1_21' title="bbox 347 2343 2171 2378">
|
||||||
<p class='ocr_par' id='par_1_29' lang='eng' title="bbox 347 2343 2171 2378">
|
<p class='ocr_par' id='par_1_29' lang='eng' title="bbox 347 2343 2171 2378">
|
||||||
<span class='ocr_line' id='line_1_65' title="bbox 347 2343 2171 2378; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_header' id='line_1_65' title="bbox 347 2343 2171 2378; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_549' title='bbox 347 2350 361 2363; x_wconf 58'>*</span>
|
<span class='ocrx_word' id='word_1_549' title='bbox 347 2350 361 2363; x_wconf 58'>*</span>
|
||||||
<span class='ocrx_word' id='word_1_550' title='bbox 373 2343 483 2377; x_wconf 96'>Simple,</span>
|
<span class='ocrx_word' id='word_1_550' title='bbox 373 2343 483 2377; x_wconf 96'>Simple,</span>
|
||||||
<span class='ocrx_word' id='word_1_551' title='bbox 495 2352 559 2377; x_wconf 96'>easy</span>
|
<span class='ocrx_word' id='word_1_551' title='bbox 495 2352 559 2377; x_wconf 96'>easy</span>
|
||||||
@@ -806,8 +806,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_559' title='bbox 1326 2345 1424 2378; x_wconf 97'>clearly</span>
|
<span class='ocrx_word' id='word_1_559' title='bbox 1326 2345 1424 2378; x_wconf 97'>clearly</span>
|
||||||
<span class='ocrx_word' id='word_1_560' title='bbox 1434 2345 1528 2378; x_wconf 96'>guides</span>
|
<span class='ocrx_word' id='word_1_560' title='bbox 1434 2345 1528 2378; x_wconf 96'>guides</span>
|
||||||
<span class='ocrx_word' id='word_1_561' title='bbox 1539 2353 1594 2378; x_wconf 97'>you</span>
|
<span class='ocrx_word' id='word_1_561' title='bbox 1539 2353 1594 2378; x_wconf 97'>you</span>
|
||||||
<span class='ocrx_word' id='word_1_562' title='bbox 1604 2345 1705 2378; x_wconf 96'>through</span>
|
<span class='ocrx_word' id='word_1_562' title='bbox 1604 2345 1724 2378; x_wconf 96'>through</span>
|
||||||
<span class='ocrx_word' id='word_1_563' title='bbox 1706 2344 1770 2371; x_wconf 96'>all</span>
|
<span class='ocrx_word' id='word_1_563' title='bbox 1735 2344 1770 2371; x_wconf 96'>all</span>
|
||||||
<span class='ocrx_word' id='word_1_564' title='bbox 1781 2344 1947 2377; x_wconf 96'>operations.</span>
|
<span class='ocrx_word' id='word_1_564' title='bbox 1781 2344 1947 2377; x_wconf 96'>operations.</span>
|
||||||
<span class='ocrx_word' id='word_1_565' title='bbox 1961 2344 1989 2371; x_wconf 96'>If</span>
|
<span class='ocrx_word' id='word_1_565' title='bbox 1961 2344 1989 2371; x_wconf 96'>If</span>
|
||||||
<span class='ocrx_word' id='word_1_566' title='bbox 1997 2344 2112 2376; x_wconf 96'>needed,</span>
|
<span class='ocrx_word' id='word_1_566' title='bbox 1997 2344 2112 2376; x_wconf 96'>needed,</span>
|
||||||
@@ -818,8 +818,8 @@
|
|||||||
<div class='ocr_carea' id='block_1_22' title="bbox 373 2381 1083 2415">
|
<div class='ocr_carea' id='block_1_22' title="bbox 373 2381 1083 2415">
|
||||||
<p class='ocr_par' id='par_1_30' lang='eng' title="bbox 373 2381 1083 2415">
|
<p class='ocr_par' id='par_1_30' lang='eng' title="bbox 373 2381 1083 2415">
|
||||||
<span class='ocr_line' id='line_1_66' title="bbox 373 2381 1083 2415; baseline 0.003 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_66' title="bbox 373 2381 1083 2415; baseline 0.003 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_568' title='bbox 373 2381 448 2407; x_wconf 96'>HELP</span>
|
<span class='ocrx_word' id='word_1_568' title='bbox 373 2381 472 2407; x_wconf 96'>HELP</span>
|
||||||
<span class='ocrx_word' id='word_1_569' title='bbox 450 2381 583 2408; x_wconf 96'>button</span>
|
<span class='ocrx_word' id='word_1_569' title='bbox 483 2381 583 2408; x_wconf 96'>button</span>
|
||||||
<span class='ocrx_word' id='word_1_570' title='bbox 594 2381 711 2415; x_wconf 96'>displays</span>
|
<span class='ocrx_word' id='word_1_570' title='bbox 594 2381 711 2415; x_wconf 96'>displays</span>
|
||||||
<span class='ocrx_word' id='word_1_571' title='bbox 722 2382 875 2409; x_wconf 96'>additional</span>
|
<span class='ocrx_word' id='word_1_571' title='bbox 722 2382 875 2409; x_wconf 96'>additional</span>
|
||||||
<span class='ocrx_word' id='word_1_572' title='bbox 886 2382 1083 2415; x_wconf 96'>explanations.</span>
|
<span class='ocrx_word' id='word_1_572' title='bbox 886 2382 1083 2415; x_wconf 96'>explanations.</span>
|
||||||
@@ -828,7 +828,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_23' title="bbox 347 2427 2145 2507">
|
<div class='ocr_carea' id='block_1_23' title="bbox 347 2427 2145 2507">
|
||||||
<p class='ocr_par' id='par_1_31' lang='eng' title="bbox 347 2427 2145 2507">
|
<p class='ocr_par' id='par_1_31' lang='eng' title="bbox 347 2427 2145 2507">
|
||||||
<span class='ocr_line' id='line_1_67' title="bbox 347 2427 1468 2461; baseline 0.002 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_header' id='line_1_67' title="bbox 347 2427 1468 2461; baseline 0.002 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_573' title='bbox 347 2432 361 2446; x_wconf 70'>*</span>
|
<span class='ocrx_word' id='word_1_573' title='bbox 347 2432 361 2446; x_wconf 70'>*</span>
|
||||||
<span class='ocrx_word' id='word_1_574' title='bbox 373 2427 612 2454; x_wconf 91'>Non-destructive</span>
|
<span class='ocrx_word' id='word_1_574' title='bbox 373 2427 612 2454; x_wconf 91'>Non-destructive</span>
|
||||||
<span class='ocrx_word' id='word_1_575' title='bbox 622 2427 914 2461; x_wconf 89'>recording—existing</span>
|
<span class='ocrx_word' id='word_1_575' title='bbox 622 2427 914 2461; x_wconf 89'>recording—existing</span>
|
||||||
@@ -839,14 +839,14 @@
|
|||||||
<span class='ocrx_word' id='word_1_580' title='bbox 1231 2428 1309 2455; x_wconf 96'>while</span>
|
<span class='ocrx_word' id='word_1_580' title='bbox 1231 2428 1309 2455; x_wconf 96'>while</span>
|
||||||
<span class='ocrx_word' id='word_1_581' title='bbox 1319 2428 1468 2461; x_wconf 92'>recording.</span>
|
<span class='ocrx_word' id='word_1_581' title='bbox 1319 2428 1468 2461; x_wconf 92'>recording.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_68' title="bbox 347 2473 2145 2507; baseline 0.001 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
<span class='ocr_header' id='line_1_68' title="bbox 347 2473 2145 2507; baseline 0.001 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_582' title='bbox 347 2478 361 2492; x_wconf 70'>¢</span>
|
<span class='ocrx_word' id='word_1_582' title='bbox 347 2478 361 2492; x_wconf 70'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_583' title='bbox 372 2473 433 2500; x_wconf 93'>Two</span>
|
<span class='ocrx_word' id='word_1_583' title='bbox 372 2473 433 2500; x_wconf 93'>Two</span>
|
||||||
<span class='ocrx_word' id='word_1_584' title='bbox 444 2473 689 2500; x_wconf 90'>FOOTSWITCH</span>
|
<span class='ocrx_word' id='word_1_584' title='bbox 444 2473 689 2500; x_wconf 90'>FOOTSWITCH</span>
|
||||||
<span class='ocrx_word' id='word_1_585' title='bbox 701 2474 818 2500; x_wconf 95'>INPUTS</span>
|
<span class='ocrx_word' id='word_1_585' title='bbox 701 2474 837 2501; x_wconf 95'>INPUTS</span>
|
||||||
<span class='ocrx_word' id='word_1_586' title='bbox 819 2474 894 2501; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_586' title='bbox 848 2481 910 2507; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_587' title='bbox 893 2473 939 2507; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_587' title='bbox 921 2473 955 2500; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_588' title='bbox 941 2473 1091 2507; x_wconf 96'>assigned</span>
|
<span class='ocrx_word' id='word_1_588' title='bbox 966 2473 1091 2507; x_wconf 96'>assigned</span>
|
||||||
<span class='ocrx_word' id='word_1_589' title='bbox 1101 2478 1130 2501; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_589' title='bbox 1101 2478 1130 2501; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_590' title='bbox 1141 2474 1271 2507; x_wconf 96'>remotely</span>
|
<span class='ocrx_word' id='word_1_590' title='bbox 1141 2474 1271 2507; x_wconf 96'>remotely</span>
|
||||||
<span class='ocrx_word' id='word_1_591' title='bbox 1281 2473 1387 2501; x_wconf 96'>control</span>
|
<span class='ocrx_word' id='word_1_591' title='bbox 1281 2473 1387 2501; x_wconf 96'>control</span>
|
||||||
@@ -873,12 +873,12 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_25' title="bbox 347 2556 1768 2590">
|
<div class='ocr_carea' id='block_1_25' title="bbox 347 2556 1768 2590">
|
||||||
<p class='ocr_par' id='par_1_33' lang='eng' title="bbox 347 2556 1768 2590">
|
<p class='ocr_par' id='par_1_33' lang='eng' title="bbox 347 2556 1768 2590">
|
||||||
<span class='ocr_line' id='line_1_70' title="bbox 347 2556 1768 2590; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_header' id='line_1_70' title="bbox 347 2556 1768 2590; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_604' title='bbox 347 2561 361 2575; x_wconf 86'>¢</span>
|
<span class='ocrx_word' id='word_1_604' title='bbox 347 2561 361 2575; x_wconf 86'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_605' title='bbox 372 2556 433 2583; x_wconf 85'>Iwo</span>
|
<span class='ocrx_word' id='word_1_605' title='bbox 372 2556 433 2583; x_wconf 85'>Iwo</span>
|
||||||
<span class='ocrx_word' id='word_1_606' title='bbox 443 2556 612 2583; x_wconf 96'>TRIGGER</span>
|
<span class='ocrx_word' id='word_1_606' title='bbox 443 2556 612 2583; x_wconf 96'>TRIGGER</span>
|
||||||
<span class='ocrx_word' id='word_1_607' title='bbox 623 2556 778 2584; x_wconf 96'>OUTPUTS</span>
|
<span class='ocrx_word' id='word_1_607' title='bbox 623 2556 797 2584; x_wconf 96'>OUTPUTS</span>
|
||||||
<span class='ocrx_word' id='word_1_608' title='bbox 780 2557 871 2590; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_608' title='bbox 808 2565 871 2590; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_609' title='bbox 881 2557 915 2584; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_609' title='bbox 881 2557 915 2584; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_610' title='bbox 925 2557 1119 2590; x_wconf 96'>programmed</span>
|
<span class='ocrx_word' id='word_1_610' title='bbox 925 2557 1119 2590; x_wconf 96'>programmed</span>
|
||||||
<span class='ocrx_word' id='word_1_611' title='bbox 1129 2561 1158 2584; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_611' title='bbox 1129 2561 1158 2584; x_wconf 96'>to</span>
|
||||||
@@ -904,14 +904,14 @@
|
|||||||
<span class='ocrx_word' id='word_1_625' title='bbox 875 2610 907 2629; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_625' title='bbox 875 2610 907 2629; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_626' title='bbox 918 2602 989 2629; x_wconf 96'>Linn</span>
|
<span class='ocrx_word' id='word_1_626' title='bbox 918 2602 989 2629; x_wconf 96'>Linn</span>
|
||||||
<span class='ocrx_word' id='word_1_627' title='bbox 1000 2603 1069 2629; x_wconf 95'>9000</span>
|
<span class='ocrx_word' id='word_1_627' title='bbox 1000 2603 1069 2629; x_wconf 95'>9000</span>
|
||||||
<span class='ocrx_word' id='word_1_628' title='bbox 1080 2610 1129 2635; x_wconf 96'>sync</span>
|
<span class='ocrx_word' id='word_1_628' title='bbox 1080 2610 1145 2635; x_wconf 96'>sync</span>
|
||||||
<span class='ocrx_word' id='word_1_629' title='bbox 1131 2607 1226 2630; x_wconf 96'>tone.</span>
|
<span class='ocrx_word' id='word_1_629' title='bbox 1155 2607 1226 2630; x_wconf 96'>tone.</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_27' title="bbox 347 2648 2100 2727">
|
<div class='ocr_carea' id='block_1_27' title="bbox 347 2648 2100 2727">
|
||||||
<p class='ocr_par' id='par_1_35' lang='eng' title="bbox 347 2648 2100 2727">
|
<p class='ocr_par' id='par_1_35' lang='eng' title="bbox 347 2648 2100 2727">
|
||||||
<span class='ocr_line' id='line_1_72' title="bbox 347 2648 1664 2682; baseline 0.001 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_header' id='line_1_72' title="bbox 347 2648 1664 2682; baseline 0.001 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_630' title='bbox 347 2654 360 2667; x_wconf 45'>©</span>
|
<span class='ocrx_word' id='word_1_630' title='bbox 347 2654 360 2667; x_wconf 45'>©</span>
|
||||||
<span class='ocrx_word' id='word_1_631' title='bbox 372 2648 483 2675; x_wconf 95'>Utilizes</span>
|
<span class='ocrx_word' id='word_1_631' title='bbox 372 2648 483 2675; x_wconf 95'>Utilizes</span>
|
||||||
<span class='ocrx_word' id='word_1_632' title='bbox 493 2648 564 2680; x_wconf 96'>ultra</span>
|
<span class='ocrx_word' id='word_1_632' title='bbox 493 2648 564 2680; x_wconf 96'>ultra</span>
|
||||||
@@ -927,17 +927,17 @@
|
|||||||
<span class='ocrx_word' id='word_1_642' title='bbox 1414 2649 1502 2676; x_wconf 96'>FAST</span>
|
<span class='ocrx_word' id='word_1_642' title='bbox 1414 2649 1502 2676; x_wconf 96'>FAST</span>
|
||||||
<span class='ocrx_word' id='word_1_643' title='bbox 1512 2648 1664 2682; x_wconf 96'>operation.</span>
|
<span class='ocrx_word' id='word_1_643' title='bbox 1512 2648 1664 2682; x_wconf 96'>operation.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_73' title="bbox 347 2694 2100 2727; baseline 0.001 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_header' id='line_1_73' title="bbox 347 2694 2100 2727; baseline 0.001 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_644' title='bbox 347 2699 361 2713; x_wconf 31'>*</span>
|
<span class='ocrx_word' id='word_1_644' title='bbox 347 2699 361 2713; x_wconf 31'>*</span>
|
||||||
<span class='ocrx_word' id='word_1_645' title='bbox 372 2694 476 2720; x_wconf 96'>TEMPO</span>
|
<span class='ocrx_word' id='word_1_645' title='bbox 372 2694 504 2721; x_wconf 96'>TEMPO</span>
|
||||||
<span class='ocrx_word' id='word_1_646' title='bbox 478 2694 562 2721; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_646' title='bbox 515 2702 578 2727; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_647' title='bbox 561 2694 606 2727; x_wconf 95'>be</span>
|
<span class='ocrx_word' id='word_1_647' title='bbox 589 2694 623 2721; x_wconf 95'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_648' title='bbox 608 2694 764 2727; x_wconf 95'>specified</span>
|
<span class='ocrx_word' id='word_1_648' title='bbox 633 2694 764 2727; x_wconf 95'>specified</span>
|
||||||
<span class='ocrx_word' id='word_1_649' title='bbox 774 2694 802 2721; x_wconf 93'>in</span>
|
<span class='ocrx_word' id='word_1_649' title='bbox 774 2694 802 2721; x_wconf 93'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_650' title='bbox 814 2695 1172 2722; x_wconf 91'>BEATS-PER-MINUTE</span>
|
<span class='ocrx_word' id='word_1_650' title='bbox 814 2695 1172 2722; x_wconf 91'>BEATS-PER-MINUTE</span>
|
||||||
<span class='ocrx_word' id='word_1_651' title='bbox 1183 2703 1215 2722; x_wconf 93'>or</span>
|
<span class='ocrx_word' id='word_1_651' title='bbox 1183 2703 1215 2722; x_wconf 93'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_652' title='bbox 1225 2695 1547 2722; x_wconf 91'>FRAMES-PER-BEAT</span>
|
<span class='ocrx_word' id='word_1_652' title='bbox 1225 2695 1567 2722; x_wconf 91'>FRAMES-PER-BEAT</span>
|
||||||
<span class='ocrx_word' id='word_1_653' title='bbox 1543 2695 1605 2721; x_wconf 96'>at</span>
|
<span class='ocrx_word' id='word_1_653' title='bbox 1577 2698 1605 2721; x_wconf 96'>at</span>
|
||||||
<span class='ocrx_word' id='word_1_654' title='bbox 1616 2695 1659 2726; x_wconf 96'>24,</span>
|
<span class='ocrx_word' id='word_1_654' title='bbox 1616 2695 1659 2726; x_wconf 96'>24,</span>
|
||||||
<span class='ocrx_word' id='word_1_655' title='bbox 1672 2695 1716 2726; x_wconf 96'>25,</span>
|
<span class='ocrx_word' id='word_1_655' title='bbox 1672 2695 1716 2726; x_wconf 96'>25,</span>
|
||||||
<span class='ocrx_word' id='word_1_656' title='bbox 1728 2702 1760 2721; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_656' title='bbox 1728 2702 1760 2721; x_wconf 96'>or</span>
|
||||||
@@ -959,11 +959,11 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_29' title="bbox 347 2777 2174 2811">
|
<div class='ocr_carea' id='block_1_29' title="bbox 347 2777 2174 2811">
|
||||||
<p class='ocr_par' id='par_1_37' lang='eng' title="bbox 347 2777 2174 2811">
|
<p class='ocr_par' id='par_1_37' lang='eng' title="bbox 347 2777 2174 2811">
|
||||||
<span class='ocr_line' id='line_1_75' title="bbox 347 2777 2174 2811; baseline 0.001 -8; x_size 33; x_descenders 5; x_ascenders 9">
|
<span class='ocr_header' id='line_1_75' title="bbox 347 2777 2174 2811; baseline 0.001 -8; x_size 33; x_descenders 5; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_664' title='bbox 347 2782 360 2796; x_wconf 81'>¢</span>
|
<span class='ocrx_word' id='word_1_664' title='bbox 347 2782 360 2796; x_wconf 81'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_665' title='bbox 372 2777 476 2803; x_wconf 94'>TEMPO</span>
|
<span class='ocrx_word' id='word_1_665' title='bbox 372 2777 504 2804; x_wconf 94'>TEMPO</span>
|
||||||
<span class='ocrx_word' id='word_1_666' title='bbox 478 2777 562 2804; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_666' title='bbox 515 2785 578 2810; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_667' title='bbox 561 2777 622 2810; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_667' title='bbox 588 2777 622 2804; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_668' title='bbox 633 2778 741 2804; x_wconf 95'>entered</span>
|
<span class='ocrx_word' id='word_1_668' title='bbox 633 2778 741 2804; x_wconf 95'>entered</span>
|
||||||
<span class='ocrx_word' id='word_1_669' title='bbox 751 2777 934 2811; x_wconf 96'>numerically,</span>
|
<span class='ocrx_word' id='word_1_669' title='bbox 751 2777 934 2811; x_wconf 96'>numerically,</span>
|
||||||
<span class='ocrx_word' id='word_1_670' title='bbox 946 2777 1101 2811; x_wconf 95'>adjustable</span>
|
<span class='ocrx_word' id='word_1_670' title='bbox 946 2777 1101 2811; x_wconf 95'>adjustable</span>
|
||||||
@@ -987,36 +987,36 @@
|
|||||||
<span class='ocrx_word' id='word_1_682' title='bbox 372 2822 410 2841; x_wconf 96'>on</span>
|
<span class='ocrx_word' id='word_1_682' title='bbox 372 2822 410 2841; x_wconf 96'>on</span>
|
||||||
<span class='ocrx_word' id='word_1_683' title='bbox 420 2815 466 2842; x_wconf 95'>the</span>
|
<span class='ocrx_word' id='word_1_683' title='bbox 420 2815 466 2842; x_wconf 95'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_684' title='bbox 476 2815 545 2841; x_wconf 95'>TAP</span>
|
<span class='ocrx_word' id='word_1_684' title='bbox 476 2815 545 2841; x_wconf 95'>TAP</span>
|
||||||
<span class='ocrx_word' id='word_1_685' title='bbox 556 2816 660 2842; x_wconf 95'>TEMPO</span>
|
<span class='ocrx_word' id='word_1_685' title='bbox 556 2815 689 2842; x_wconf 95'>TEMPO</span>
|
||||||
<span class='ocrx_word' id='word_1_686' title='bbox 662 2815 808 2842; x_wconf 96'>button.</span>
|
<span class='ocrx_word' id='word_1_686' title='bbox 699 2815 808 2842; x_wconf 96'>button.</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_31' title="bbox 347 2861 1792 2940">
|
<div class='ocr_carea' id='block_1_31' title="bbox 347 2861 1792 2940">
|
||||||
<p class='ocr_par' id='par_1_39' lang='eng' title="bbox 347 2861 1792 2940">
|
<p class='ocr_par' id='par_1_39' lang='eng' title="bbox 347 2861 1792 2940">
|
||||||
<span class='ocr_line' id='line_1_77' title="bbox 347 2861 1792 2895; baseline 0.001 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_header' id='line_1_77' title="bbox 347 2861 1792 2895; baseline 0.001 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_687' title='bbox 347 2866 360 2880; x_wconf 43'>¢</span>
|
<span class='ocrx_word' id='word_1_687' title='bbox 347 2866 360 2880; x_wconf 43'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_688' title='bbox 372 2861 504 2887; x_wconf 96'>TEMPO</span>
|
<span class='ocrx_word' id='word_1_688' title='bbox 372 2861 504 2887; x_wconf 96'>TEMPO</span>
|
||||||
<span class='ocrx_word' id='word_1_689' title='bbox 515 2861 677 2888; x_wconf 96'>CHANGES</span>
|
<span class='ocrx_word' id='word_1_689' title='bbox 515 2861 696 2888; x_wconf 96'>CHANGES</span>
|
||||||
<span class='ocrx_word' id='word_1_690' title='bbox 679 2861 771 2894; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_690' title='bbox 707 2869 771 2894; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_691' title='bbox 781 2861 815 2888; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_691' title='bbox 781 2861 815 2888; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_692' title='bbox 825 2869 1000 2894; x_wconf 96'>programmed</span>
|
<span class='ocrx_word' id='word_1_692' title='bbox 825 2861 1019 2894; x_wconf 96'>programmed</span>
|
||||||
<span class='ocrx_word' id='word_1_693' title='bbox 1001 2861 1087 2888; x_wconf 96'>into</span>
|
<span class='ocrx_word' id='word_1_693' title='bbox 1030 2861 1087 2888; x_wconf 96'>into</span>
|
||||||
<span class='ocrx_word' id='word_1_694' title='bbox 1099 2869 1115 2888; x_wconf 95'>a</span>
|
<span class='ocrx_word' id='word_1_694' title='bbox 1099 2869 1115 2888; x_wconf 95'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_695' title='bbox 1126 2870 1268 2895; x_wconf 96'>sequence,</span>
|
<span class='ocrx_word' id='word_1_695' title='bbox 1126 2870 1268 2895; x_wconf 96'>sequence,</span>
|
||||||
<span class='ocrx_word' id='word_1_696' title='bbox 1280 2861 1344 2888; x_wconf 96'>with</span>
|
<span class='ocrx_word' id='word_1_696' title='bbox 1280 2861 1344 2888; x_wconf 96'>with</span>
|
||||||
<span class='ocrx_word' id='word_1_697' title='bbox 1356 2866 1448 2888; x_wconf 95'>smooth</span>
|
<span class='ocrx_word' id='word_1_697' title='bbox 1356 2862 1467 2888; x_wconf 95'>smooth</span>
|
||||||
<span class='ocrx_word' id='word_1_698' title='bbox 1450 2861 1635 2888; x_wconf 96'>transitions</span>
|
<span class='ocrx_word' id='word_1_698' title='bbox 1478 2861 1635 2888; x_wconf 96'>transitions</span>
|
||||||
<span class='ocrx_word' id='word_1_699' title='bbox 1646 2861 1670 2887; x_wconf 96'>if</span>
|
<span class='ocrx_word' id='word_1_699' title='bbox 1646 2861 1670 2887; x_wconf 96'>if</span>
|
||||||
<span class='ocrx_word' id='word_1_700' title='bbox 1679 2861 1792 2888; x_wconf 87'>desired.</span>
|
<span class='ocrx_word' id='word_1_700' title='bbox 1679 2861 1792 2888; x_wconf 87'>desired.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_78' title="bbox 347 2906 1507 2940; baseline 0.002 -8; x_size 33; x_descenders 7; x_ascenders 8">
|
<span class='ocr_header' id='line_1_78' title="bbox 347 2906 1507 2940; baseline 0.002 -8; x_size 33; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_701' title='bbox 347 2911 360 2925; x_wconf 69'>¢</span>
|
<span class='ocrx_word' id='word_1_701' title='bbox 347 2911 360 2925; x_wconf 69'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_702' title='bbox 371 2906 434 2938; x_wconf 96'>Any</span>
|
<span class='ocrx_word' id='word_1_702' title='bbox 371 2906 434 2938; x_wconf 96'>Any</span>
|
||||||
<span class='ocrx_word' id='word_1_703' title='bbox 444 2906 539 2932; x_wconf 96'>TIME</span>
|
<span class='ocrx_word' id='word_1_703' title='bbox 444 2906 539 2932; x_wconf 96'>TIME</span>
|
||||||
<span class='ocrx_word' id='word_1_704' title='bbox 550 2906 739 2933; x_wconf 96'>SIGNATURE</span>
|
<span class='ocrx_word' id='word_1_704' title='bbox 550 2906 763 2933; x_wconf 96'>SIGNATURE</span>
|
||||||
<span class='ocrx_word' id='word_1_705' title='bbox 740 2907 820 2933; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_705' title='bbox 773 2915 836 2939; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_706' title='bbox 819 2907 880 2939; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_706' title='bbox 846 2907 880 2933; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_707' title='bbox 891 2907 968 2938; x_wconf 96'>used,</span>
|
<span class='ocrx_word' id='word_1_707' title='bbox 891 2907 968 2938; x_wconf 96'>used,</span>
|
||||||
<span class='ocrx_word' id='word_1_708' title='bbox 980 2907 1036 2934; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_708' title='bbox 980 2907 1036 2934; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_709' title='bbox 1046 2915 1109 2940; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_709' title='bbox 1046 2915 1109 2940; x_wconf 96'>may</span>
|
||||||
|
|||||||
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
+892
-892
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
+3
-3
@@ -103,7 +103,7 @@ ERASE, REPEAT, PLAY/STOP, or LOCATE.
|
|||||||
|
|
||||||
© Will sync to standard LinnDrum or Linn 9000 sync tone.
|
© Will sync to standard LinnDrum or Linn 9000 sync tone.
|
||||||
|
|
||||||
© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation.
|
® Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation.
|
||||||
* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second,
|
* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second,
|
||||||
|
|
||||||
(even drop frame!)
|
(even drop frame!)
|
||||||
@@ -115,9 +115,9 @@ on the TAP TEMPO button.
|
|||||||
¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired.
|
¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired.
|
||||||
¢ Any TIME SIGNATURE may be used, and may be changed within a song.
|
¢ Any TIME SIGNATURE may be used, and may be changed within a song.
|
||||||
|
|
||||||
linn
|
nn
|
||||||
Linn Electronics, Inc.
|
|
||||||
|
|
||||||
|
Linn Electronics, Inc.
|
||||||
18720 Oxnard Street, Tarzana, CA 91356
|
18720 Oxnard Street, Tarzana, CA 91356
|
||||||
(818) 708-8131 TELEX #298949 LINN UR
|
(818) 708-8131 TELEX #298949 LINN UR
|
||||||
|
|
||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
+1011
-981
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
+87
-82
@@ -1,123 +1,128 @@
|
|||||||
The LinnSequencer
|
2A NNI‘I 6F6867# XATALL IE18-80L (818)
|
||||||
32 Track MIDI Sequence Recorder
|
|
||||||
|
|
||||||
The LinnSequencer is a state-of-the-art composition and performance tool for the professional musician. It is
|
9SEI6 VO “BUBZIRY, “J0aNS PIPUXO OZLEI
|
||||||
|
“Uy ‘soTUOMOI,q UUrT
|
||||||
|
|
||||||
extremely powerful, yet amazingly simple to learn and use. It’s many remarkable features include:
|
uut]
|
||||||
|
|
||||||
¢ Operation is similar to multi-track tape recorder with PLAY, STOP, RECORD, FAST
|
“‘SUOS B UIJIM pasueyo oq ABU pue ‘posn oq AWW AYN IVNOIS AWLL AUV
|
||||||
FORWARD, REWIND, and LOCATE controls.
|
“parlsop Jr SUOTIISUBI} YIOOUIS YIM “BoueNbas eB OJUI pourtueIZOId 9q ABU SFONWHO OdINAL e
|
||||||
|
|
||||||
e Each of the 100 sequences contains 32 simultaneous, polyphonic tracks. Each track may
|
‘uonng OdNAL dV L 9) uO
|
||||||
be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic
|
|
||||||
|
|
||||||
synthesizers!
|
sojou Jayienb Suiddy} Aq 10 ‘syUSTIOIOUI oINUTIAI-J8g-Jesg & JO sys} UL ofquisn(pe ‘ATTeouIAUINU paiajus oq ABU OdINALL e
|
||||||
|
|
||||||
¢ Ultra-fast 3%” disk drive stores complex songs in seconds and holds over 110,000 notes
|
(jouer doup u3a9)
|
||||||
|
|
||||||
per disk!
|
“puooes Jed souely O€ 10 “SZ “pz 18 [LVAG-MAd-SHN VU 10 ALOANIWAAd-SLVAd U! patyoeds aq kewl OAL «
|
||||||
|
‘uoTe1odo [SVx JO} Aj[eusoyUT JoyndUIOd 11g 9] 98108 ZHI 8g ‘poeds-ysry Bann soz] e
|
||||||
|
|
||||||
¢ One or all tracks may be TRANSPOSED at the touch of a key.
|
"9U0} DUAS 0006 UUL] Jo wNIqUUr] prepue}s 0} OUAS [ITAA ©
|
||||||
e Exclusive real-time ERASE function makes editing FAST.
|
|
||||||
* Exclusive REPEAT function automatically repeats any held notes at a pre-selected
|
|
||||||
|
|
||||||
rhythmic value.
|
“ONYBA 9}OU poloapes Aue Je sas—nd jndyno 07 pewureigold 3q ACW SL Ad LNO YADONAL OML
|
||||||
|
|
||||||
¢ TIMING CORRECTION works during playback and operates without ‘chopping’ notes.
|
"ALVOOT 10 GOLS/AV 1d ‘LWddad “ASV
|
||||||
|
|
||||||
¢ Optional SMPTE time code synchronization.
|
SUIpNpoUr ‘suOTIOUN] posn A[UOUILUOS 94] JO AUBUT [O1]UOD AJ9]OWIAI 0} PousIsse oq ACUI ST AdNI HOLIMSLOO OME «
|
||||||
|
“SUIPIONAI I[IYM P2sesd JOU Iv $3}OU BUTISIXO—ZUIPIOIA SATON.ASOP-UON
|
||||||
|
|
||||||
© Optional remote control.
|
‘suoneurldxa peuoyippe sdeydsip uowng g1TqH
|
||||||
|
|
||||||
Recording a Sequence
|
oy] ‘pepsau JI ‘suoneiodo [ye yYsnosy] NOA sapins ApIespo Avfdsip QO] Joey Z7¢ 9y3—uoeIodo Urea] 0} Ased ‘aus «
|
||||||
|
|
||||||
To record a sequence, simply press RECORD and PLAY,
|
jUorel]suowtap & IO} Aepol Jayeap uur’] INOA dag ‘dISHUL
|
||||||
then play your MIDI keyboard in time to the Sequencer’s
|
INOA 0} UONUS}]¥ PaplAIPUN INOA SUTJOASp ITY ps pue
|
||||||
click track. When the sequence loops back around to bar 1,
|
p1osai ‘asoduod no Jay 0} pausisap st 1s0uenbesuur’] oy)
|
||||||
you’ ll hear what you played—only all timing errors will be
|
Aum Aposiooid $,Jeu], ‘SS9d0Id SATTBS1D OY} YIM SOIOJIOIUT
|
||||||
|
yey) xo]dwWI0d Os dq JOA9U P[NoUsS osn NOA AZopOuYdE} oy
|
||||||
|
|
||||||
corrected! (Timing correction may be adjusted or defeated).
|
ISTUMOIAUIO?) NOAA UOHISOdWIO)
|
||||||
|
|
||||||
Any additional notes played will be added into the track
|
"NOSpr] B Oy ‘AONUTJUT yada 0} seq Maz Se] BY] Jas UdAd
|
||||||
— existing notes are not erased while recording!
|
uvd NOA ‘palisap JJ ‘souanbes Mou ¥B OVUT sjied ou] [Te Adoo
|
||||||
|
ATesrewO Ne WI) [IM ONOS ALVAAO JeyIe80} wey}
|
||||||
|
,deyd,, 0} UOTOUNJ ONOS ALVA ou] asn usy] ‘saouanbes
|
||||||
|
JENPIAIpUt UI (“949 ‘snJOYD ‘aS1OA) UOTIDIS JIseq Yes
|
||||||
|
Pl0da1 OF ST ABM JOuIOUY “(812g 666 01 dn) ysnory) ABM
|
||||||
|
|
||||||
FAST FORWARD, REWIND, and LOCATE controls
|
dU} [fe YORI] YORs p10991 0} ST SUOS B 9789I9 0} ABM SUG,
|
||||||
may be used at any time to quickly access any location in
|
|
||||||
your sequence for spot-recording. To overdub a new part,
|
|
||||||
select a different track and start recording—while you
|
|
||||||
record, the first track will play in perfect sync (unless you
|
|
||||||
MUTE it, or SOLO another track). In this way, up to 32
|
|
||||||
tracks may be overdubbed! All MIDI effects are recorded
|
|
||||||
including pitch bend, modulation, velocity, aftertouch,
|
|
||||||
sustain pedal, and program changes!
|
|
||||||
|
|
||||||
Editing
|
SUOS & SUTVAID
|
||||||
|
|
||||||
To erase a wrong note, simply hold ERASE and press
|
*suoT}oes poJUBMUN
|
||||||
the note to be erased just before it plays in the sequence—
|
|
||||||
when played back, it will be gone. Notes may also be
|
|
||||||
|
|
||||||
added, erased, or changed using the SINGLE STEP func-
|
SAOUIOI 0} ABM SWS dU} SoyeIodo SUV ALATAaG
|
||||||
tion. To overdub notes at specific points within a sequence,
|
|
||||||
|
|
||||||
Additional Features
|
“OBPLIq dy} PUB SNIOY PUOdAS dT]] Ud9MIAQ SIDA ISI
|
||||||
|
|
||||||
simply use LOCATE, FAST FORWARD, or REWIND to
|
ay) Jo Adoo B JJasuT WYSE NOAA ‘afdwexs 10.f ‘UO JUSIN]JIP
|
||||||
find the desired bar number, then start recording.
|
|
||||||
|
|
||||||
The INSERT/COPY function allows you to move bars
|
B IO aouaNbas sues OY} UI—JOY OUP 0} UOTIEIO] 9UO WOT]
|
||||||
from one location to another—in the same sequence or a
|
$1Bq JAOUI OF NOA sMOTIe WOTIOUNS AdOO/IMASNI OULL
|
||||||
different one. For example, you might insert a copy of the
|
|
||||||
first verse between the second chorus and the bridge.
|
|
||||||
DELETE BARS operates the same way to remove
|
|
||||||
unwanted sections,
|
|
||||||
|
|
||||||
Creating a Song
|
‘SUIPIONAI JIVIS Udy) “OQuINU eq porisop ay} puy
|
||||||
|
|
||||||
One way to create a song is to record each track all the
|
0} CNIMAY 10 ‘CYVM Od LSWA “AEVOOT esn Apduns
|
||||||
way through (up to 999 bars). Another way is to record
|
|
||||||
each basic section (verse, chorus, etc.) in individual
|
|
||||||
sequences, then use the CREATE SONG function to “chain”
|
|
||||||
them together. CREATE SONG will then automatically
|
|
||||||
copy all the parts into a new sequence. If desired, you can
|
|
||||||
even set the last few bars to repeat infinitely, for a fadeout.
|
|
||||||
|
|
||||||
Composition Without Compromise
|
sainjeay [PUOHIPPY
|
||||||
|
|
||||||
The technology you use should never be so complex that
|
‘gouanbas & UTYIIM s]UTOd a1y1dads 3¥ $9100 QnPIOAO OL "UOT}
|
||||||
it interferes with the creative process. That’s precisely why
|
-ouns dALLS ATONIS 24) Suisn pasueyo Jo ‘pasesa ‘pappe
|
||||||
the LinnSequencer is designed to let you compose, record
|
aq osye ABUT S9]ON ‘U0 9q ]IIM 1 “yoeq podeyd uayM
|
||||||
and edit while devoting your undivided attention to your
|
—aouanbas oy] ul skeyd 71 a10J9q Isnf posers oq 0} d]0U ayy
|
||||||
music. See your Linn dealer today for a demonstration!
|
ssaid pue ASvwug ploy Aydunis ‘jou Suomm & aseso OL
|
||||||
|
|
||||||
* Simple, easy to learn operation—the 32 character LCD display clearly guides you through all operations. If needed, the
|
sunipa
|
||||||
|
|
||||||
HELP button displays additional explanations.
|
jsesdueyo ureisoid pue ‘fepod ureysns
|
||||||
|
‘yonoplalje ‘AWOOTOA ‘UOTyeTNpow ‘pusg youd Surpnyour
|
||||||
|
pep10del are $199JJ2 TCTIN [WV iPeqqnpseao aq Aeur syoen
|
||||||
|
Ze 07 dn ‘Kem sie Uy *(foeI} JOyOUR OJOS 10 ALLAN
|
||||||
|
NOA ssofum) duAS yOaysod ul Avy [[IM Yow] ISI 93 “prooar
|
||||||
|
NOA 3[IYM—SUIPIOIA LIBIS PU YORI) TUdIOTJIP B JOaTas
|
||||||
|
*y1ed MOU B QNPIsA0 OL, “SuIps0daJ-jods 10} aouanbes mno0k
|
||||||
|
UI UOHBIO] Aue ssad0e ATYOIND 0} owt} Aue ye pasn aq AvUE
|
||||||
|
SJONUOD FLIVOOT pur ‘ANIMA ‘CYVMaYOd LSVd
|
||||||
|
{SUIPIOSAI {IY posesa JOU se So]OU SuTsTXO—
|
||||||
|
yous} 3U} OUT poppe aq JIM poteyd sajou yeuonippe Auy
|
||||||
|
|
||||||
* Non-destructive recording—existing notes are not erased while recording.
|
*(povesjap 10 poysn{pe oq ABW UOTIIII0D BUTUTT]) j{paqoeLI09
|
||||||
¢ Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including
|
|
||||||
|
|
||||||
ERASE, REPEAT, PLAY/STOP, or LOCATE.
|
2q ][IM S1OLIe Sur [fe ATUO—patey]d nod Jey Jedy ]],NOA
|
||||||
|
|
||||||
¢ Iwo TRIGGER OUTPUTS may be programmed to output pulses at any selected note value.
|
‘] req 0] punose yoeq sdoo] sduanbas ay] Udy AA “YOu Yor
|
||||||
|
|
||||||
© Will sync to standard LinnDrum or Linn 9000 sync tone.
|
§,sa0uaNbas at} O] SUIT) UI preogday [IW] INO Avy usy3
|
||||||
|
AV'1d pue (YOON ssoid Ayduus ‘aousnbes & p1o09es OF,
|
||||||
|
|
||||||
© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation.
|
g0uaNbas & SUIP10I0y]
|
||||||
* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second,
|
|
||||||
|
|
||||||
(even drop frame!)
|
‘JONWOD s}JouNaI TeuONdGO e
|
||||||
|
|
||||||
¢ TEMPO may be entered numerically, adjustable in tenths of a Beat-Per-Minute increments, or by tapping quarter notes
|
"UOTJEZIUOIYUAS OPOS UIT} FLAWS [euondo e
|
||||||
|
|
||||||
on the TAP TEMPO button.
|
‘sou .sulddoys, noyyM sayelodo pue yoegdvyd ZuLINp S¥IOM NOLLOANNYOO ONIWILL e
|
||||||
|
|
||||||
¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired.
|
‘onqea ory AY
|
||||||
¢ Any TIME SIGNATURE may be used, and may be changed within a song.
|
|
||||||
|
|
||||||
linn
|
pojoojes-oid & ye sajou pyoy Aue syeadas ATTeONewWO Ne UOTOUNS [WAdAY OAISNOX e
|
||||||
Linn Electronics, Inc.
|
‘LSVJ SUnIpS soyeu UOTOUN ASV UA OUlN-[eal SAISNIOXY e
|
||||||
|
‘Koy B JO YONO} 941 12 CASOdSNVALL 0g ABU Syde] [Te 10 9UC e
|
||||||
|
|
||||||
18720 Oxnard Street, Tarzana, CA 91356
|
i ASIP Jed
|
||||||
(818) 708-8131 TELEX #298949 LINN UR
|
|
||||||
|
S9}0U OOO‘OTT JOA SpfOy puv SpUOdeS UT SBUOS Xa[AUIOD So10}S DALIP YSIP , 74 € ISCJ-CNIN
|
||||||
|
|
||||||
|
jSIOZISOUJUAS
|
||||||
|
|
||||||
|
stuoydAjod of 0} dn skeyd A[snoourynuls ‘spouueYd [IW 9T JO duo 0} pousisse oq
|
||||||
|
ABUL YORI] YOR ‘syous) oruoydAjod ‘snoouelnurs 7¢ SuTeJUOS ssouUaNbas QO] OY} JO YORA e
|
||||||
|
|
||||||
|
‘SJONUOS ATWOOT pur ‘GNIMAY ‘GaVM OA
|
||||||
|
LSVd ‘GYOOde AOLS ‘AV Td YIM Jopsocas ade} Yowsj-N[NU O} eps st UOTLISdO @
|
||||||
|
LOPNOUT SaINjeoy s[quyIeUlss AUB S.JJ ‘OSN pue UIes] 0} o[duns A[suIzeUe JOA ‘PnJsomod APOUIOITXO
|
||||||
|
St 1] “UeIOIsNUL feUOIssajoid oY} 10 JOO} soUBULIOJIJAd pue UOTIsOduIOS 11e-dY1-JO-9}e)s B SI IONUANbDaguUT] ay
|
||||||
|
|
||||||
|
JOps1odady soUINbIS [GTI YVAL ZE
|
||||||
|
Jgouanbaguury oy
|
||||||
|
|
||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
+999
-983
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
+83
-82
@@ -1,123 +1,124 @@
|
|||||||
The LinnSequencer
|
2A NNI‘I 6F6867# XATALL IE18-80L (818)
|
||||||
32 Track MIDI Sequence Recorder
|
|
||||||
|
|
||||||
The LinnSequencer is a state-of-the-art composition and performance tool for the professional musician. It is
|
9SEI6 VO “BUBZIRY, “J0aNS PIPUXO OZLEI
|
||||||
|
“Uy ‘soTUOMOI,q UUrT
|
||||||
|
|
||||||
extremely powerful, yet amazingly simple to learn and use. It’s many remarkable features include:
|
uu
|
||||||
|
|
||||||
¢ Operation is similar to multi-track tape recorder with PLAY, STOP, RECORD, FAST
|
“‘SUOS B UIJIM pasueyo oq ABU pue ‘posn oq AWW AYN IVNOIS AWLL AUV
|
||||||
FORWARD, REWIND, and LOCATE controls.
|
“parlsop Jr SUOTIISUBI} YIOOUIS YIM “BoueNbas eB OJUI pourtueIZOId 9q ABU SFONWHO OdINAL e
|
||||||
|
|
||||||
e Each of the 100 sequences contains 32 simultaneous, polyphonic tracks. Each track may
|
‘uonng OdNAL dV L 9) uO
|
||||||
be assigned to one of 16 MIDI channels. Simultaneously plays up to 16 polyphonic
|
|
||||||
|
|
||||||
synthesizers!
|
sojou Jayienb Suiddy} Aq 10 ‘syUSTIOIOUI oINUTIAI-J8g-Jesg & JO sys} UL ofquisn(pe ‘ATTeouIAUINU paiajus oq ABU OdINALL e
|
||||||
|
|
||||||
¢ Ultra-fast 3%” disk drive stores complex songs in seconds and holds over 110,000 notes
|
(jouer doup u3a9)
|
||||||
|
|
||||||
per disk!
|
“puooes Jed souely O€ 10 “SZ “pz 18 [LVAG-MAd-SHN VU 10 ALOANIWAAd-SLVAd U! patyoeds aq kewl OAL «
|
||||||
|
‘uoTe1odo [SVx JO} Aj[eusoyUT JoyndUIOd 11g 9] 98108 ZHI 8g ‘poeds-ysry Bann soz] e
|
||||||
|
|
||||||
¢ One or all tracks may be TRANSPOSED at the touch of a key.
|
"9U0} DUAS 0006 UUL] Jo wNIqUUr] prepue}s 0} OUAS [ITAA ©
|
||||||
e Exclusive real-time ERASE function makes editing FAST.
|
|
||||||
* Exclusive REPEAT function automatically repeats any held notes at a pre-selected
|
|
||||||
|
|
||||||
rhythmic value.
|
“ONYBA 9}OU poloapes Aue Je sas—nd jndyno 07 pewureigold 3q ACW SL Ad LNO YADONAL OML
|
||||||
|
|
||||||
¢ TIMING CORRECTION works during playback and operates without ‘chopping’ notes.
|
"ALVOOT 10 GOLS/AV 1d ‘LWddad “ASV
|
||||||
|
|
||||||
¢ Optional SMPTE time code synchronization.
|
SUIpNpoUr ‘suOTIOUN] posn A[UOUILUOS 94] JO AUBUT [O1]UOD AJ9]OWIAI 0} PousIsse oq ACUI ST AdNI HOLIMSLOO OME «
|
||||||
|
“SUIPIONAI I[IYM P2sesd JOU Iv $3}OU BUTISIXO—ZUIPIOIA SATON.ASOP-UON
|
||||||
|
|
||||||
© Optional remote control.
|
‘suoneurldxa peuoyippe sdeydsip uowng g1TqH
|
||||||
|
|
||||||
Recording a Sequence
|
oy] ‘pepsau JI ‘suoneiodo [ye yYsnosy] NOA sapins ApIespo Avfdsip QO] Joey Z7¢ 9y3—uoeIodo Urea] 0} Ased ‘aus «
|
||||||
|
|
||||||
To record a sequence, simply press RECORD and PLAY,
|
jUorel]suowtap & IO} Aepol Jayeap uur’] INOA dag ‘dISHUL
|
||||||
then play your MIDI keyboard in time to the Sequencer’s
|
INOA 0} UONUS}]¥ PaplAIPUN INOA SUTJOASp ITY ps pue
|
||||||
click track. When the sequence loops back around to bar 1,
|
p1osai ‘asoduod no Jay 0} pausisap st 1s0uenbesuur’] oy)
|
||||||
you’ ll hear what you played—only all timing errors will be
|
Aum Aposiooid $,Jeu], ‘SS9d0Id SATTBS1D OY} YIM SOIOJIOIUT
|
||||||
|
yey) xo]dwWI0d Os dq JOA9U P[NoUsS osn NOA AZopOuYdE} oy
|
||||||
|
|
||||||
corrected! (Timing correction may be adjusted or defeated).
|
ISTUMOIAUIO?) NOAA UOHISOdWIO)
|
||||||
|
|
||||||
Any additional notes played will be added into the track
|
"NOSpr] B Oy ‘AONUTJUT yada 0} seq Maz Se] BY] Jas UdAd
|
||||||
— existing notes are not erased while recording!
|
uvd NOA ‘palisap JJ ‘souanbes Mou ¥B OVUT sjied ou] [Te Adoo
|
||||||
|
ATesrewO Ne WI) [IM ONOS ALVAAO JeyIe80} wey}
|
||||||
|
,deyd,, 0} UOTOUNJ ONOS ALVA ou] asn usy] ‘saouanbes
|
||||||
|
JENPIAIpUt UI (“949 ‘snJOYD ‘aS1OA) UOTIDIS JIseq Yes
|
||||||
|
Pl0da1 OF ST ABM JOuIOUY “(812g 666 01 dn) ysnory) ABM
|
||||||
|
|
||||||
FAST FORWARD, REWIND, and LOCATE controls
|
dU} [fe YORI] YORs p10991 0} ST SUOS B 9789I9 0} ABM SUG,
|
||||||
may be used at any time to quickly access any location in
|
|
||||||
your sequence for spot-recording. To overdub a new part,
|
|
||||||
select a different track and start recording—while you
|
|
||||||
record, the first track will play in perfect sync (unless you
|
|
||||||
MUTE it, or SOLO another track). In this way, up to 32
|
|
||||||
tracks may be overdubbed! All MIDI effects are recorded
|
|
||||||
including pitch bend, modulation, velocity, aftertouch,
|
|
||||||
sustain pedal, and program changes!
|
|
||||||
|
|
||||||
Editing
|
SUOS & SUTVAID
|
||||||
|
|
||||||
To erase a wrong note, simply hold ERASE and press
|
*suoT}oes poJUBMUN
|
||||||
the note to be erased just before it plays in the sequence—
|
|
||||||
when played back, it will be gone. Notes may also be
|
|
||||||
|
|
||||||
added, erased, or changed using the SINGLE STEP func-
|
SAOUIOI 0} ABM SWS dU} SoyeIodo SUV ALATAaG
|
||||||
tion. To overdub notes at specific points within a sequence,
|
|
||||||
|
|
||||||
Additional Features
|
“OBPLIq dy} PUB SNIOY PUOdAS dT]] Ud9MIAQ SIDA ISI
|
||||||
|
|
||||||
simply use LOCATE, FAST FORWARD, or REWIND to
|
ay) Jo Adoo B JJasuT WYSE NOAA ‘afdwexs 10.f ‘UO JUSIN]JIP
|
||||||
find the desired bar number, then start recording.
|
|
||||||
|
|
||||||
The INSERT/COPY function allows you to move bars
|
B IO aouaNbas sues OY} UI—JOY OUP 0} UOTIEIO] 9UO WOT]
|
||||||
from one location to another—in the same sequence or a
|
$1Bq JAOUI OF NOA sMOTIe WOTIOUNS AdOO/IMASNI OULL
|
||||||
different one. For example, you might insert a copy of the
|
|
||||||
first verse between the second chorus and the bridge.
|
|
||||||
DELETE BARS operates the same way to remove
|
|
||||||
unwanted sections,
|
|
||||||
|
|
||||||
Creating a Song
|
‘SUIPIONAI JIVIS Udy) “OQuINU eq porisop ay} puy
|
||||||
|
|
||||||
One way to create a song is to record each track all the
|
0} CNIMAY 10 ‘CYVM Od LSWA “AEVOOT esn Apduns
|
||||||
way through (up to 999 bars). Another way is to record
|
|
||||||
each basic section (verse, chorus, etc.) in individual
|
|
||||||
sequences, then use the CREATE SONG function to “chain”
|
|
||||||
them together. CREATE SONG will then automatically
|
|
||||||
copy all the parts into a new sequence. If desired, you can
|
|
||||||
even set the last few bars to repeat infinitely, for a fadeout.
|
|
||||||
|
|
||||||
Composition Without Compromise
|
sainjeay [PUOHIPPY
|
||||||
|
|
||||||
The technology you use should never be so complex that
|
‘gouanbas & UTYIIM s]UTOd a1y1dads 3¥ $9100 QnPIOAO OL "UOT}
|
||||||
it interferes with the creative process. That’s precisely why
|
-ouns dALLS ATONIS 24) Suisn pasueyo Jo ‘pasesa ‘pappe
|
||||||
the LinnSequencer is designed to let you compose, record
|
aq osye ABUT S9]ON ‘U0 9q ]IIM 1 “yoeq podeyd uayM
|
||||||
and edit while devoting your undivided attention to your
|
—aouanbas oy] ul skeyd 71 a10J9q Isnf posers oq 0} d]0U ayy
|
||||||
music. See your Linn dealer today for a demonstration!
|
ssaid pue ASvwug ploy Aydunis ‘jou Suomm & aseso OL
|
||||||
|
|
||||||
* Simple, easy to learn operation—the 32 character LCD display clearly guides you through all operations. If needed, the
|
sunipa
|
||||||
|
|
||||||
HELP button displays additional explanations.
|
jsesdueyo ureisoid pue ‘fepod ureysns
|
||||||
|
‘yonoplalje ‘AWOOTOA ‘UOTyeTNpow ‘pusg youd Surpnyour
|
||||||
|
pep10del are $199JJ2 TCTIN [WV iPeqqnpseao aq Aeur syoen
|
||||||
|
Ze 07 dn ‘Kem sie Uy *(foeI} JOyOUR OJOS 10 ALLAN
|
||||||
|
NOA ssofum) duAS yOaysod ul Avy [[IM Yow] ISI 93 “prooar
|
||||||
|
NOA 3[IYM—SUIPIOIA LIBIS PU YORI) TUdIOTJIP B JOaTas
|
||||||
|
*y1ed MOU B QNPIsA0 OL, “SuIps0daJ-jods 10} aouanbes mno0k
|
||||||
|
UI UOHBIO] Aue ssad0e ATYOIND 0} owt} Aue ye pasn aq AvUE
|
||||||
|
SJONUOD FLIVOOT pur ‘ANIMA ‘CYVMaYOd LSVd
|
||||||
|
{SUIPIOSAI {IY posesa JOU se So]OU SuTsTXO—
|
||||||
|
yous} 3U} OUT poppe aq JIM poteyd sajou yeuonippe Auy
|
||||||
|
*(povesjap 10 poysn{pe oq ABW UOTIIII0D BUTUTT]) j{paqoeLI09
|
||||||
|
2q ][IM S1OLIe Sur [fe ATUO—patey]d nod Jey Jedy ]],NOA
|
||||||
|
‘] req 0] punose yoeq sdoo] sduanbas ay] Udy AA “YOu Yor
|
||||||
|
§,sa0uaNbas at} O] SUIT) UI preogday [IW] INO Avy usy3
|
||||||
|
AV'1d pue (YOON ssoid Ayduus ‘aousnbes & p1o09es OF,
|
||||||
|
|
||||||
* Non-destructive recording—existing notes are not erased while recording.
|
g0uaNbas & SUIP10I0y]
|
||||||
¢ Two FOOTSWITCH INPUTS may be assigned to remotely control many of the commonly used functions, including
|
|
||||||
|
|
||||||
ERASE, REPEAT, PLAY/STOP, or LOCATE.
|
‘JONWOD s}JouNaI TeuONdGO e
|
||||||
|
|
||||||
¢ Iwo TRIGGER OUTPUTS may be programmed to output pulses at any selected note value.
|
"UOTJEZIUOIYUAS OPOS UIT} FLAWS [euondo e
|
||||||
|
|
||||||
© Will sync to standard LinnDrum or Linn 9000 sync tone.
|
‘sou .sulddoys, noyyM sayelodo pue yoegdvyd ZuLINp S¥IOM NOLLOANNYOO ONIWILL e
|
||||||
|
|
||||||
© Utilizes ultra high-speed, 8 MHz 80186 16 bit computer internally for FAST operation.
|
‘onqea ory AY
|
||||||
* TEMPO may be specified in BEATS-PER-MINUTE or FRAMES-PER-BEAT at 24, 25, or 30 frames per second,
|
|
||||||
|
|
||||||
(even drop frame!)
|
pojoojes-oid & ye sajou pyoy Aue syeadas ATTeONewWO Ne UOTOUNS [WAdAY OAISNOX e
|
||||||
|
‘LSVJ SUnIpS soyeu UOTOUN ASV UA OUlN-[eal SAISNIOXY e
|
||||||
|
‘Koy B JO YONO} 941 12 CASOdSNVALL 0g ABU Syde] [Te 10 9UC e
|
||||||
|
|
||||||
¢ TEMPO may be entered numerically, adjustable in tenths of a Beat-Per-Minute increments, or by tapping quarter notes
|
i ASIP Jed
|
||||||
|
|
||||||
on the TAP TEMPO button.
|
S9}0U OOO‘OTT JOA SpfOy puv SpUOdeS UT SBUOS Xa[AUIOD So10}S DALIP YSIP , 74 € ISCJ-CNIN
|
||||||
|
|
||||||
¢ TEMPO CHANGES may be programmed into a sequence, with smooth transitions if desired.
|
jSIOZISOUJUAS
|
||||||
¢ Any TIME SIGNATURE may be used, and may be changed within a song.
|
|
||||||
|
|
||||||
linn
|
stuoydAjod of 0} dn skeyd A[snoourynuls ‘spouueYd [IW 9T JO duo 0} pousisse oq
|
||||||
Linn Electronics, Inc.
|
ABUL YORI] YOR ‘syous) oruoydAjod ‘snoouelnurs 7¢ SuTeJUOS ssouUaNbas QO] OY} JO YORA e
|
||||||
|
|
||||||
18720 Oxnard Street, Tarzana, CA 91356
|
‘SJONUOS ATWOOT pur ‘GNIMAY ‘GaVM OA
|
||||||
(818) 708-8131 TELEX #298949 LINN UR
|
LSVd ‘GYOOde AOLS ‘AV Td YIM Jopsocas ade} Yowsj-N[NU O} eps st UOTLISdO @
|
||||||
|
LOPNOUT SaINjeoy s[quyIeUlss AUB S.JJ ‘OSN pue UIes] 0} o[duns A[suIzeUe JOA ‘PnJsomod APOUIOITXO
|
||||||
|
St 1] “UeIOIsNUL feUOIssajoid oY} 10 JOO} soUBULIOJIJAd pue UOTIsOduIOS 11e-dY1-JO-9}e)s B SI IONUANbDaguUT] ay
|
||||||
|
|
||||||
|
JOps1odady soUINbIS [GTI YVAL ZE
|
||||||
|
Jgouanbaguury oy
|
||||||
|
|
||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
-1
@@ -1 +0,0 @@
|
|||||||
Warning. Invalid resolution 0 dpi. Using 70 instead.
|
|
||||||
|
|||||||
-1
@@ -1 +0,0 @@
|
|||||||
Warning. Invalid resolution 0 dpi. Using 70 instead.
|
|
||||||
|
|||||||
-1
@@ -1 +0,0 @@
|
|||||||
Warning. Invalid resolution 0 dpi. Using 70 instead.
|
|
||||||
|
|||||||
-1
@@ -1 +0,0 @@
|
|||||||
Warning. Invalid resolution 0 dpi. Using 70 instead.
|
|
||||||
|
|||||||
+179
-179
@@ -4,19 +4,19 @@
|
|||||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
||||||
<head>
|
<head>
|
||||||
<title></title>
|
<title></title>
|
||||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8" />
|
<meta http-equiv="Content-Type" content="text/html;charset=utf-8"/>
|
||||||
<meta name='ocr-system' content='tesseract 4.0.0' />
|
<meta name='ocr-system' content='tesseract 4.1.1' />
|
||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
||||||
</head>
|
</head>
|
||||||
<body>
|
<body>
|
||||||
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.y8xsq1f7/000001_ocr.png"; bbox 0 0 2550 3300; ppageno 0'>
|
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.ucmqv4tt/000001_ocr.png"; bbox 0 0 2550 3300; ppageno 0'>
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
<div class='ocr_carea' id='block_1_1' title="bbox 582 131 1968 303">
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 582 131 1968 303">
|
||||||
<span class='ocr_line' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
<span class='ocr_header' id='line_1_1' title="bbox 882 131 1657 217; baseline 0.001 -17; x_size 85; x_descenders 16; x_ascenders 19">
|
||||||
<span class='ocrx_word' id='word_1_1' title='bbox 882 132 1036 202; x_wconf 93'>The</span>
|
<span class='ocrx_word' id='word_1_1' title='bbox 882 132 1036 202; x_wconf 93'>The</span>
|
||||||
<span class='ocrx_word' id='word_1_2' title='bbox 1061 131 1657 217; x_wconf 91'>LinnSequencer</span>
|
<span class='ocrx_word' id='word_1_2' title='bbox 1061 131 1657 217; x_wconf 91'>LinnSequencer</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_2' title="bbox 582 215 1968 303; baseline 0 -17; x_size 87; x_descenders 16; x_ascenders 21">
|
<span class='ocr_header' id='line_1_2' title="bbox 582 215 1968 303; baseline 0 -17; x_size 87; x_descenders 16; x_ascenders 21">
|
||||||
<span class='ocrx_word' id='word_1_3' title='bbox 582 215 674 286; x_wconf 96'>32</span>
|
<span class='ocrx_word' id='word_1_3' title='bbox 582 215 674 286; x_wconf 96'>32</span>
|
||||||
<span class='ocrx_word' id='word_1_4' title='bbox 697 218 923 288; x_wconf 95'>Track</span>
|
<span class='ocrx_word' id='word_1_4' title='bbox 697 218 923 288; x_wconf 95'>Track</span>
|
||||||
<span class='ocrx_word' id='word_1_5' title='bbox 948 218 1181 287; x_wconf 96'>MIDI</span>
|
<span class='ocrx_word' id='word_1_5' title='bbox 948 218 1181 287; x_wconf 96'>MIDI</span>
|
||||||
@@ -27,27 +27,27 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_2' title="bbox 347 380 2188 574">
|
<div class='ocr_carea' id='block_1_2' title="bbox 347 380 2188 574">
|
||||||
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 347 380 2188 423">
|
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 347 380 2188 423">
|
||||||
<span class='ocr_line' id='line_1_3' title="bbox 347 380 2188 423; baseline -0.001 -12; x_size 38; x_descenders 8; x_ascenders 10">
|
<span class='ocr_header' id='line_1_3' title="bbox 347 380 2188 423; baseline -0.001 -12; x_size 38; x_descenders 8; x_ascenders 10">
|
||||||
<span class='ocrx_word' id='word_1_8' title='bbox 347 380 412 410; x_wconf 93'>The</span>
|
<span class='ocrx_word' id='word_1_8' title='bbox 347 380 412 410; x_wconf 93'>The</span>
|
||||||
<span class='ocrx_word' id='word_1_9' title='bbox 424 380 661 417; x_wconf 91'>LinnSequencer</span>
|
<span class='ocrx_word' id='word_1_9' title='bbox 424 380 676 417; x_wconf 92'>LinnSequencer</span>
|
||||||
<span class='ocrx_word' id='word_1_10' title='bbox 663 380 712 411; x_wconf 96'>is</span>
|
<span class='ocrx_word' id='word_1_10' title='bbox 688 380 712 411; x_wconf 96'>is</span>
|
||||||
<span class='ocrx_word' id='word_1_11' title='bbox 724 390 743 411; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_11' title='bbox 724 390 743 411; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_12' title='bbox 754 381 1005 423; x_wconf 96'>state-of-the-art</span>
|
<span class='ocrx_word' id='word_1_12' title='bbox 754 381 1005 423; x_wconf 96'>state-of-the-art</span>
|
||||||
<span class='ocrx_word' id='word_1_13' title='bbox 1017 380 1226 418; x_wconf 96'>composition</span>
|
<span class='ocrx_word' id='word_1_13' title='bbox 1017 380 1226 418; x_wconf 96'>composition</span>
|
||||||
<span class='ocrx_word' id='word_1_14' title='bbox 1238 381 1299 411; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_14' title='bbox 1238 381 1299 411; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_15' title='bbox 1311 380 1507 418; x_wconf 96'>performance</span>
|
<span class='ocrx_word' id='word_1_15' title='bbox 1311 380 1525 418; x_wconf 96'>performance</span>
|
||||||
<span class='ocrx_word' id='word_1_16' title='bbox 1509 385 1591 411; x_wconf 96'>tool</span>
|
<span class='ocrx_word' id='word_1_16' title='bbox 1536 380 1602 411; x_wconf 96'>tool</span>
|
||||||
<span class='ocrx_word' id='word_1_17' title='bbox 1593 380 1663 411; x_wconf 97'>for</span>
|
<span class='ocrx_word' id='word_1_17' title='bbox 1615 380 1663 411; x_wconf 97'>for</span>
|
||||||
<span class='ocrx_word' id='word_1_18' title='bbox 1674 381 1725 410; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_18' title='bbox 1674 381 1725 410; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_19' title='bbox 1737 380 1940 417; x_wconf 96'>professional</span>
|
<span class='ocrx_word' id='word_1_19' title='bbox 1737 380 1940 417; x_wconf 95'>professional</span>
|
||||||
<span class='ocrx_word' id='word_1_20' title='bbox 1952 380 2103 411; x_wconf 96'>musician.</span>
|
<span class='ocrx_word' id='word_1_20' title='bbox 1952 380 2112 411; x_wconf 96'>musician.</span>
|
||||||
<span class='ocrx_word' id='word_1_21' title='bbox 2106 381 2152 410; x_wconf 96'>It</span>
|
<span class='ocrx_word' id='word_1_21' title='bbox 2127 381 2152 410; x_wconf 96'>It</span>
|
||||||
<span class='ocrx_word' id='word_1_22' title='bbox 2164 380 2188 410; x_wconf 96'>is</span>
|
<span class='ocrx_word' id='word_1_22' title='bbox 2164 380 2188 410; x_wconf 96'>is</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
|
|
||||||
<p class='ocr_par' id='par_1_3' lang='eng' title="bbox 347 430 1988 468">
|
<p class='ocr_par' id='par_1_3' lang='eng' title="bbox 347 430 1988 468">
|
||||||
<span class='ocr_line' id='line_1_4' title="bbox 347 430 1988 468; baseline 0 -8; x_size 37; x_descenders 7; x_ascenders 9">
|
<span class='ocr_header' id='line_1_4' title="bbox 347 430 1988 468; baseline 0 -8; x_size 37; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_23' title='bbox 347 430 507 467; x_wconf 96'>extremely</span>
|
<span class='ocrx_word' id='word_1_23' title='bbox 347 430 507 467; x_wconf 96'>extremely</span>
|
||||||
<span class='ocrx_word' id='word_1_24' title='bbox 518 430 677 467; x_wconf 96'>powerful,</span>
|
<span class='ocrx_word' id='word_1_24' title='bbox 518 430 677 467; x_wconf 96'>powerful,</span>
|
||||||
<span class='ocrx_word' id='word_1_25' title='bbox 691 435 739 467; x_wconf 96'>yet</span>
|
<span class='ocrx_word' id='word_1_25' title='bbox 691 435 739 467; x_wconf 96'>yet</span>
|
||||||
@@ -66,7 +66,7 @@
|
|||||||
</p>
|
</p>
|
||||||
|
|
||||||
<p class='ocr_par' id='par_1_4' lang='eng' title="bbox 350 482 2093 574">
|
<p class='ocr_par' id='par_1_4' lang='eng' title="bbox 350 482 2093 574">
|
||||||
<span class='ocr_line' id='line_1_5' title="bbox 350 482 2093 527; baseline 0 -9; x_size 43; x_descenders 7; x_ascenders 11">
|
<span class='ocr_header' id='line_1_5' title="bbox 350 482 2093 527; baseline 0 -9; x_size 43; x_descenders 7; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_37' title='bbox 350 490 368 508; x_wconf 73'>¢</span>
|
<span class='ocrx_word' id='word_1_37' title='bbox 350 490 368 508; x_wconf 73'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_38' title='bbox 383 482 585 526; x_wconf 95'>Operation</span>
|
<span class='ocrx_word' id='word_1_38' title='bbox 383 482 585 526; x_wconf 95'>Operation</span>
|
||||||
<span class='ocrx_word' id='word_1_39' title='bbox 598 482 627 518; x_wconf 96'>is</span>
|
<span class='ocrx_word' id='word_1_39' title='bbox 598 482 627 518; x_wconf 96'>is</span>
|
||||||
@@ -81,18 +81,18 @@
|
|||||||
<span class='ocrx_word' id='word_1_48' title='bbox 1741 483 1957 525; x_wconf 96'>RECORD,</span>
|
<span class='ocrx_word' id='word_1_48' title='bbox 1741 483 1957 525; x_wconf 96'>RECORD,</span>
|
||||||
<span class='ocrx_word' id='word_1_49' title='bbox 1974 483 2093 518; x_wconf 96'>FAST</span>
|
<span class='ocrx_word' id='word_1_49' title='bbox 1974 483 2093 518; x_wconf 96'>FAST</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_6' title="bbox 383 532 1345 574; baseline 0.001 -7; x_size 43; x_descenders 7; x_ascenders 11">
|
<span class='ocr_header' id='line_1_6' title="bbox 383 532 1345 574; baseline 0.001 -7; x_size 43; x_descenders 7; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_50' title='bbox 383 532 635 574; x_wconf 96'>FORWARD,</span>
|
<span class='ocrx_word' id='word_1_50' title='bbox 383 532 635 574; x_wconf 96'>FORWARD,</span>
|
||||||
<span class='ocrx_word' id='word_1_51' title='bbox 652 532 865 574; x_wconf 95'>REWIND,</span>
|
<span class='ocrx_word' id='word_1_51' title='bbox 652 532 865 574; x_wconf 95'>REWIND,</span>
|
||||||
<span class='ocrx_word' id='word_1_52' title='bbox 882 532 956 568; x_wconf 95'>and</span>
|
<span class='ocrx_word' id='word_1_52' title='bbox 882 532 956 568; x_wconf 95'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_53' title='bbox 971 532 1131 568; x_wconf 95'>LOCATE</span>
|
<span class='ocrx_word' id='word_1_53' title='bbox 971 532 1163 568; x_wconf 95'>LOCATE</span>
|
||||||
<span class='ocrx_word' id='word_1_54' title='bbox 1132 532 1345 568; x_wconf 95'>controls.</span>
|
<span class='ocrx_word' id='word_1_54' title='bbox 1177 532 1345 568; x_wconf 95'>controls.</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_3' title="bbox 349 589 2136 685">
|
<div class='ocr_carea' id='block_1_3' title="bbox 349 589 2136 685">
|
||||||
<p class='ocr_par' id='par_1_5' lang='eng' title="bbox 349 589 2136 685">
|
<p class='ocr_par' id='par_1_5' lang='eng' title="bbox 349 589 2136 685">
|
||||||
<span class='ocr_line' id='line_1_7' title="bbox 349 589 2136 634; baseline 0.001 -9; x_size 44; x_descenders 8; x_ascenders 11">
|
<span class='ocr_header' id='line_1_7' title="bbox 349 589 2136 634; baseline 0.001 -9; x_size 44; x_descenders 8; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_55' title='bbox 349 597 368 615; x_wconf 59'>e</span>
|
<span class='ocrx_word' id='word_1_55' title='bbox 349 597 368 615; x_wconf 59'>e</span>
|
||||||
<span class='ocrx_word' id='word_1_56' title='bbox 383 590 482 625; x_wconf 96'>Each</span>
|
<span class='ocrx_word' id='word_1_56' title='bbox 383 590 482 625; x_wconf 96'>Each</span>
|
||||||
<span class='ocrx_word' id='word_1_57' title='bbox 496 589 539 625; x_wconf 96'>of</span>
|
<span class='ocrx_word' id='word_1_57' title='bbox 496 589 539 625; x_wconf 96'>of</span>
|
||||||
@@ -108,7 +108,7 @@
|
|||||||
<span class='ocrx_word' id='word_1_67' title='bbox 1934 590 2035 626; x_wconf 96'>track</span>
|
<span class='ocrx_word' id='word_1_67' title='bbox 1934 590 2035 626; x_wconf 96'>track</span>
|
||||||
<span class='ocrx_word' id='word_1_68' title='bbox 2050 600 2136 634; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_68' title='bbox 2050 600 2136 634; x_wconf 96'>may</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_8' title="bbox 383 639 2022 685; baseline 0.001 -10; x_size 44; x_descenders 8; x_ascenders 11">
|
<span class='ocr_header' id='line_1_8' title="bbox 383 639 2022 685; baseline 0.001 -10; x_size 44; x_descenders 8; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_69' title='bbox 383 639 428 675; x_wconf 95'>be</span>
|
<span class='ocrx_word' id='word_1_69' title='bbox 383 639 428 675; x_wconf 95'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_70' title='bbox 442 639 607 684; x_wconf 95'>assigned</span>
|
<span class='ocrx_word' id='word_1_70' title='bbox 442 639 607 684; x_wconf 95'>assigned</span>
|
||||||
<span class='ocrx_word' id='word_1_71' title='bbox 621 645 659 676; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_71' title='bbox 621 645 659 676; x_wconf 96'>to</span>
|
||||||
@@ -135,7 +135,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_5' title="bbox 349 748 2117 793">
|
<div class='ocr_carea' id='block_1_5' title="bbox 349 748 2117 793">
|
||||||
<p class='ocr_par' id='par_1_7' lang='eng' title="bbox 349 748 2117 793">
|
<p class='ocr_par' id='par_1_7' lang='eng' title="bbox 349 748 2117 793">
|
||||||
<span class='ocr_line' id='line_1_10' title="bbox 349 748 2117 793; baseline 0 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
<span class='ocr_header' id='line_1_10' title="bbox 349 748 2117 793; baseline 0 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_84' title='bbox 349 755 367 774; x_wconf 58'>¢</span>
|
<span class='ocrx_word' id='word_1_84' title='bbox 349 755 367 774; x_wconf 58'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_85' title='bbox 383 748 573 784; x_wconf 91'>Ultra-fast</span>
|
<span class='ocrx_word' id='word_1_85' title='bbox 383 748 573 784; x_wconf 91'>Ultra-fast</span>
|
||||||
<span class='ocrx_word' id='word_1_86' title='bbox 588 749 677 784; x_wconf 22'>3%”</span>
|
<span class='ocrx_word' id='word_1_86' title='bbox 588 749 677 784; x_wconf 22'>3%”</span>
|
||||||
@@ -147,8 +147,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_92' title='bbox 1330 748 1368 784; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_92' title='bbox 1330 748 1368 784; x_wconf 96'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_93' title='bbox 1382 748 1535 785; x_wconf 96'>seconds</span>
|
<span class='ocrx_word' id='word_1_93' title='bbox 1382 748 1535 785; x_wconf 96'>seconds</span>
|
||||||
<span class='ocrx_word' id='word_1_94' title='bbox 1550 748 1624 784; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_94' title='bbox 1550 748 1624 784; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_95' title='bbox 1638 748 1728 784; x_wconf 96'>holds</span>
|
<span class='ocrx_word' id='word_1_95' title='bbox 1638 748 1746 784; x_wconf 96'>holds</span>
|
||||||
<span class='ocrx_word' id='word_1_96' title='bbox 1730 759 1844 784; x_wconf 96'>over</span>
|
<span class='ocrx_word' id='word_1_96' title='bbox 1761 759 1844 784; x_wconf 96'>over</span>
|
||||||
<span class='ocrx_word' id='word_1_97' title='bbox 1859 749 2000 791; x_wconf 96'>110,000</span>
|
<span class='ocrx_word' id='word_1_97' title='bbox 1859 749 2000 791; x_wconf 96'>110,000</span>
|
||||||
<span class='ocrx_word' id='word_1_98' title='bbox 2013 753 2117 784; x_wconf 96'>notes</span>
|
<span class='ocrx_word' id='word_1_98' title='bbox 2013 753 2117 784; x_wconf 96'>notes</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -164,23 +164,23 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_7' title="bbox 349 855 2030 1016">
|
<div class='ocr_carea' id='block_1_7' title="bbox 349 855 2030 1016">
|
||||||
<p class='ocr_par' id='par_1_9' lang='eng' title="bbox 349 855 2030 1016">
|
<p class='ocr_par' id='par_1_9' lang='eng' title="bbox 349 855 2030 1016">
|
||||||
<span class='ocr_line' id='line_1_12' title="bbox 350 855 1638 900; baseline 0.001 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
<span class='ocr_header' id='line_1_12' title="bbox 350 855 1638 900; baseline 0.001 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_101' title='bbox 350 863 367 881; x_wconf 45'>¢</span>
|
<span class='ocrx_word' id='word_1_101' title='bbox 350 863 367 881; x_wconf 45'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_102' title='bbox 383 856 444 891; x_wconf 95'>One</span>
|
<span class='ocrx_word' id='word_1_102' title='bbox 383 856 464 891; x_wconf 95'>One</span>
|
||||||
<span class='ocrx_word' id='word_1_103' title='bbox 445 866 502 891; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_103' title='bbox 478 866 520 891; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_104' title='bbox 503 855 580 891; x_wconf 96'>all</span>
|
<span class='ocrx_word' id='word_1_104' title='bbox 534 855 580 891; x_wconf 96'>all</span>
|
||||||
<span class='ocrx_word' id='word_1_105' title='bbox 594 856 712 892; x_wconf 95'>tracks</span>
|
<span class='ocrx_word' id='word_1_105' title='bbox 594 856 712 892; x_wconf 95'>tracks</span>
|
||||||
<span class='ocrx_word' id='word_1_106' title='bbox 726 867 811 900; x_wconf 95'>may</span>
|
<span class='ocrx_word' id='word_1_106' title='bbox 726 867 811 900; x_wconf 95'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_107' title='bbox 823 856 869 892; x_wconf 81'>be</span>
|
<span class='ocrx_word' id='word_1_107' title='bbox 823 856 869 892; x_wconf 81'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_108' title='bbox 882 856 1175 892; x_wconf 96'>TRANSPOSED</span>
|
<span class='ocrx_word' id='word_1_108' title='bbox 882 856 1212 892; x_wconf 96'>TRANSPOSED</span>
|
||||||
<span class='ocrx_word' id='word_1_109' title='bbox 1178 857 1264 892; x_wconf 96'>at</span>
|
<span class='ocrx_word' id='word_1_109' title='bbox 1227 861 1264 892; x_wconf 96'>at</span>
|
||||||
<span class='ocrx_word' id='word_1_110' title='bbox 1277 856 1338 892; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_110' title='bbox 1277 856 1338 892; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_111' title='bbox 1351 856 1463 892; x_wconf 96'>touch</span>
|
<span class='ocrx_word' id='word_1_111' title='bbox 1351 856 1463 892; x_wconf 96'>touch</span>
|
||||||
<span class='ocrx_word' id='word_1_112' title='bbox 1477 867 1501 892; x_wconf 96'>of</span>
|
<span class='ocrx_word' id='word_1_112' title='bbox 1477 856 1520 892; x_wconf 96'>of</span>
|
||||||
<span class='ocrx_word' id='word_1_113' title='bbox 1502 856 1554 892; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_113' title='bbox 1531 867 1554 892; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_114' title='bbox 1568 856 1638 900; x_wconf 96'>key.</span>
|
<span class='ocrx_word' id='word_1_114' title='bbox 1568 856 1638 900; x_wconf 96'>key.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_13' title="bbox 350 913 1535 958; baseline 0.001 -9; x_size 44; x_descenders 8; x_ascenders 11">
|
<span class='ocr_header' id='line_1_13' title="bbox 350 913 1535 958; baseline 0.001 -9; x_size 44; x_descenders 8; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_115' title='bbox 350 921 367 939; x_wconf 45'>e</span>
|
<span class='ocrx_word' id='word_1_115' title='bbox 350 921 367 939; x_wconf 45'>e</span>
|
||||||
<span class='ocrx_word' id='word_1_116' title='bbox 383 913 568 950; x_wconf 96'>Exclusive</span>
|
<span class='ocrx_word' id='word_1_116' title='bbox 383 913 568 950; x_wconf 96'>Exclusive</span>
|
||||||
<span class='ocrx_word' id='word_1_117' title='bbox 581 913 756 950; x_wconf 96'>real-time</span>
|
<span class='ocrx_word' id='word_1_117' title='bbox 581 913 756 950; x_wconf 96'>real-time</span>
|
||||||
@@ -190,7 +190,7 @@
|
|||||||
<span class='ocrx_word' id='word_1_121' title='bbox 1266 914 1400 958; x_wconf 96'>editing</span>
|
<span class='ocrx_word' id='word_1_121' title='bbox 1266 914 1400 958; x_wconf 96'>editing</span>
|
||||||
<span class='ocrx_word' id='word_1_122' title='bbox 1414 915 1535 950; x_wconf 95'>FAST.</span>
|
<span class='ocrx_word' id='word_1_122' title='bbox 1414 915 1535 950; x_wconf 95'>FAST.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_14' title="bbox 349 971 2030 1016; baseline 0.001 -10; x_size 44; x_descenders 8; x_ascenders 11">
|
<span class='ocr_header' id='line_1_14' title="bbox 349 971 2030 1016; baseline 0.001 -10; x_size 44; x_descenders 8; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_123' title='bbox 349 979 367 997; x_wconf 0'>*</span>
|
<span class='ocrx_word' id='word_1_123' title='bbox 349 979 367 997; x_wconf 0'>*</span>
|
||||||
<span class='ocrx_word' id='word_1_124' title='bbox 382 971 568 1007; x_wconf 95'>Exclusive</span>
|
<span class='ocrx_word' id='word_1_124' title='bbox 382 971 568 1007; x_wconf 95'>Exclusive</span>
|
||||||
<span class='ocrx_word' id='word_1_125' title='bbox 582 972 773 1007; x_wconf 96'>REPEAT</span>
|
<span class='ocrx_word' id='word_1_125' title='bbox 582 972 773 1007; x_wconf 96'>REPEAT</span>
|
||||||
@@ -209,18 +209,18 @@
|
|||||||
<div class='ocr_carea' id='block_1_8' title="bbox 382 1021 689 1065">
|
<div class='ocr_carea' id='block_1_8' title="bbox 382 1021 689 1065">
|
||||||
<p class='ocr_par' id='par_1_10' lang='eng' title="bbox 382 1021 689 1065">
|
<p class='ocr_par' id='par_1_10' lang='eng' title="bbox 382 1021 689 1065">
|
||||||
<span class='ocr_line' id='line_1_15' title="bbox 382 1021 689 1065; baseline 0.003 -8; x_size 45; x_descenders 8; x_ascenders 12">
|
<span class='ocr_line' id='line_1_15' title="bbox 382 1021 689 1065; baseline 0.003 -8; x_size 45; x_descenders 8; x_ascenders 12">
|
||||||
<span class='ocrx_word' id='word_1_135' title='bbox 382 1021 564 1065; x_wconf 95'>rhythmic</span>
|
<span class='ocrx_word' id='word_1_135' title='bbox 382 1021 564 1065; x_wconf 96'>rhythmic</span>
|
||||||
<span class='ocrx_word' id='word_1_136' title='bbox 577 1021 689 1058; x_wconf 96'>value.</span>
|
<span class='ocrx_word' id='word_1_136' title='bbox 577 1021 689 1058; x_wconf 96'>value.</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_9' title="bbox 349 1080 2174 1125">
|
<div class='ocr_carea' id='block_1_9' title="bbox 349 1080 2174 1125">
|
||||||
<p class='ocr_par' id='par_1_11' lang='eng' title="bbox 349 1080 2174 1125">
|
<p class='ocr_par' id='par_1_11' lang='eng' title="bbox 349 1080 2174 1125">
|
||||||
<span class='ocr_line' id='line_1_16' title="bbox 349 1080 2174 1125; baseline 0.001 -11; x_size 45; x_descenders 9; x_ascenders 11">
|
<span class='ocr_header' id='line_1_16' title="bbox 349 1080 2174 1125; baseline 0.001 -11; x_size 45; x_descenders 9; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_137' title='bbox 349 1087 367 1105; x_wconf 80'>¢</span>
|
<span class='ocrx_word' id='word_1_137' title='bbox 349 1087 367 1105; x_wconf 80'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_138' title='bbox 382 1080 567 1115; x_wconf 95'>TIMING</span>
|
<span class='ocrx_word' id='word_1_138' title='bbox 382 1080 567 1115; x_wconf 95'>TIMING</span>
|
||||||
<span class='ocrx_word' id='word_1_139' title='bbox 582 1080 870 1116; x_wconf 95'>CORRECTION</span>
|
<span class='ocrx_word' id='word_1_139' title='bbox 582 1080 908 1116; x_wconf 95'>CORRECTION</span>
|
||||||
<span class='ocrx_word' id='word_1_140' title='bbox 872 1080 1041 1116; x_wconf 96'>works</span>
|
<span class='ocrx_word' id='word_1_140' title='bbox 921 1080 1041 1116; x_wconf 96'>works</span>
|
||||||
<span class='ocrx_word' id='word_1_141' title='bbox 1056 1080 1186 1124; x_wconf 96'>during</span>
|
<span class='ocrx_word' id='word_1_141' title='bbox 1056 1080 1186 1124; x_wconf 96'>during</span>
|
||||||
<span class='ocrx_word' id='word_1_142' title='bbox 1199 1080 1378 1124; x_wconf 96'>playback</span>
|
<span class='ocrx_word' id='word_1_142' title='bbox 1199 1080 1378 1124; x_wconf 96'>playback</span>
|
||||||
<span class='ocrx_word' id='word_1_143' title='bbox 1392 1080 1466 1116; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_143' title='bbox 1392 1080 1466 1116; x_wconf 96'>and</span>
|
||||||
@@ -233,11 +233,11 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_10' title="bbox 349 1137 1287 1182">
|
<div class='ocr_carea' id='block_1_10' title="bbox 349 1137 1287 1182">
|
||||||
<p class='ocr_par' id='par_1_12' lang='eng' title="bbox 349 1137 1287 1182">
|
<p class='ocr_par' id='par_1_12' lang='eng' title="bbox 349 1137 1287 1182">
|
||||||
<span class='ocr_line' id='line_1_17' title="bbox 349 1137 1287 1182; baseline 0.001 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
<span class='ocr_textfloat' id='line_1_17' title="bbox 349 1137 1287 1182; baseline 0.001 -9; x_size 45; x_descenders 9; x_ascenders 11">
|
||||||
<span class='ocrx_word' id='word_1_148' title='bbox 349 1145 367 1163; x_wconf 80'>¢</span>
|
<span class='ocrx_word' id='word_1_148' title='bbox 349 1145 367 1163; x_wconf 80'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_149' title='bbox 382 1137 560 1182; x_wconf 95'>Optional</span>
|
<span class='ocrx_word' id='word_1_149' title='bbox 382 1137 560 1182; x_wconf 95'>Optional</span>
|
||||||
<span class='ocrx_word' id='word_1_150' title='bbox 575 1138 708 1174; x_wconf 96'>SMPTE</span>
|
<span class='ocrx_word' id='word_1_150' title='bbox 575 1138 739 1174; x_wconf 96'>SMPTE</span>
|
||||||
<span class='ocrx_word' id='word_1_151' title='bbox 709 1138 839 1174; x_wconf 96'>time</span>
|
<span class='ocrx_word' id='word_1_151' title='bbox 752 1138 839 1174; x_wconf 96'>time</span>
|
||||||
<span class='ocrx_word' id='word_1_152' title='bbox 853 1138 945 1174; x_wconf 95'>code</span>
|
<span class='ocrx_word' id='word_1_152' title='bbox 853 1138 945 1174; x_wconf 95'>code</span>
|
||||||
<span class='ocrx_word' id='word_1_153' title='bbox 959 1138 1287 1182; x_wconf 96'>synchronization.</span>
|
<span class='ocrx_word' id='word_1_153' title='bbox 959 1138 1287 1182; x_wconf 96'>synchronization.</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -246,7 +246,7 @@
|
|||||||
<div class='ocr_carea' id='block_1_11' title="bbox 349 1195 874 1240">
|
<div class='ocr_carea' id='block_1_11' title="bbox 349 1195 874 1240">
|
||||||
<p class='ocr_par' id='par_1_13' lang='eng' title="bbox 349 1195 874 1240">
|
<p class='ocr_par' id='par_1_13' lang='eng' title="bbox 349 1195 874 1240">
|
||||||
<span class='ocr_line' id='line_1_18' title="bbox 349 1195 874 1240; baseline 0 -8; x_size 45; x_descenders 8; x_ascenders 12">
|
<span class='ocr_line' id='line_1_18' title="bbox 349 1195 874 1240; baseline 0 -8; x_size 45; x_descenders 8; x_ascenders 12">
|
||||||
<span class='ocrx_word' id='word_1_154' title='bbox 349 1203 367 1222; x_wconf 70'>©</span>
|
<span class='ocrx_word' id='word_1_154' title='bbox 349 1203 367 1222; x_wconf 74'>©</span>
|
||||||
<span class='ocrx_word' id='word_1_155' title='bbox 382 1195 560 1240; x_wconf 96'>Optional</span>
|
<span class='ocrx_word' id='word_1_155' title='bbox 382 1195 560 1240; x_wconf 96'>Optional</span>
|
||||||
<span class='ocrx_word' id='word_1_156' title='bbox 573 1201 709 1233; x_wconf 96'>remote</span>
|
<span class='ocrx_word' id='word_1_156' title='bbox 573 1201 709 1233; x_wconf 96'>remote</span>
|
||||||
<span class='ocrx_word' id='word_1_157' title='bbox 723 1196 874 1233; x_wconf 95'>control.</span>
|
<span class='ocrx_word' id='word_1_157' title='bbox 723 1196 874 1233; x_wconf 95'>control.</span>
|
||||||
@@ -278,9 +278,9 @@
|
|||||||
<span class='ocrx_word' id='word_1_170' title='bbox 346 1379 411 1406; x_wconf 96'>then</span>
|
<span class='ocrx_word' id='word_1_170' title='bbox 346 1379 411 1406; x_wconf 96'>then</span>
|
||||||
<span class='ocrx_word' id='word_1_171' title='bbox 422 1378 483 1412; x_wconf 96'>play</span>
|
<span class='ocrx_word' id='word_1_171' title='bbox 422 1378 483 1412; x_wconf 96'>play</span>
|
||||||
<span class='ocrx_word' id='word_1_172' title='bbox 493 1387 562 1412; x_wconf 96'>your</span>
|
<span class='ocrx_word' id='word_1_172' title='bbox 493 1387 562 1412; x_wconf 96'>your</span>
|
||||||
<span class='ocrx_word' id='word_1_173' title='bbox 572 1379 646 1405; x_wconf 96'>MIDI</span>
|
<span class='ocrx_word' id='word_1_173' title='bbox 572 1379 659 1405; x_wconf 96'>MIDI</span>
|
||||||
<span class='ocrx_word' id='word_1_174' title='bbox 649 1379 792 1412; x_wconf 96'>keyboard</span>
|
<span class='ocrx_word' id='word_1_174' title='bbox 671 1379 810 1412; x_wconf 96'>keyboard</span>
|
||||||
<span class='ocrx_word' id='word_1_175' title='bbox 792 1379 848 1406; x_wconf 95'>in</span>
|
<span class='ocrx_word' id='word_1_175' title='bbox 821 1379 848 1406; x_wconf 95'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_176' title='bbox 858 1379 923 1406; x_wconf 95'>time</span>
|
<span class='ocrx_word' id='word_1_176' title='bbox 858 1379 923 1406; x_wconf 95'>time</span>
|
||||||
<span class='ocrx_word' id='word_1_177' title='bbox 934 1384 963 1406; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_177' title='bbox 934 1384 963 1406; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_178' title='bbox 974 1379 1019 1406; x_wconf 93'>the</span>
|
<span class='ocrx_word' id='word_1_178' title='bbox 974 1379 1019 1406; x_wconf 93'>the</span>
|
||||||
@@ -294,14 +294,14 @@
|
|||||||
<span class='ocrx_word' id='word_1_184' title='bbox 676 1426 810 1452; x_wconf 96'>sequence</span>
|
<span class='ocrx_word' id='word_1_184' title='bbox 676 1426 810 1452; x_wconf 96'>sequence</span>
|
||||||
<span class='ocrx_word' id='word_1_185' title='bbox 821 1419 901 1452; x_wconf 96'>loops</span>
|
<span class='ocrx_word' id='word_1_185' title='bbox 821 1419 901 1452; x_wconf 96'>loops</span>
|
||||||
<span class='ocrx_word' id='word_1_186' title='bbox 912 1419 983 1446; x_wconf 96'>back</span>
|
<span class='ocrx_word' id='word_1_186' title='bbox 912 1419 983 1446; x_wconf 96'>back</span>
|
||||||
<span class='ocrx_word' id='word_1_187' title='bbox 995 1427 1082 1446; x_wconf 96'>around</span>
|
<span class='ocrx_word' id='word_1_187' title='bbox 995 1419 1101 1446; x_wconf 96'>around</span>
|
||||||
<span class='ocrx_word' id='word_1_188' title='bbox 1084 1419 1141 1446; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_188' title='bbox 1112 1423 1141 1446; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_189' title='bbox 1152 1419 1189 1446; x_wconf 96'>bar</span>
|
<span class='ocrx_word' id='word_1_189' title='bbox 1152 1419 1201 1446; x_wconf 96'>bar</span>
|
||||||
<span class='ocrx_word' id='word_1_190' title='bbox 1189 1419 1232 1450; x_wconf 74'>1,</span>
|
<span class='ocrx_word' id='word_1_190' title='bbox 1213 1419 1232 1450; x_wconf 74'>1,</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_23' title="bbox 346 1457 1223 1491; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_23' title="bbox 346 1457 1223 1491; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_191' title='bbox 346 1465 400 1490; x_wconf 15'>you’</span>
|
<span class='ocrx_word' id='word_1_191' title='bbox 346 1457 430 1490; x_wconf 14'>you’</span>
|
||||||
<span class='ocrx_word' id='word_1_192' title='bbox 404 1457 430 1484; x_wconf 15'>ll</span>
|
<span class='ocrx_word' id='word_1_192' title='bbox 406 1453 436 1496; x_wconf 14'>ll</span>
|
||||||
<span class='ocrx_word' id='word_1_193' title='bbox 441 1457 506 1485; x_wconf 96'>hear</span>
|
<span class='ocrx_word' id='word_1_193' title='bbox 441 1457 506 1485; x_wconf 96'>hear</span>
|
||||||
<span class='ocrx_word' id='word_1_194' title='bbox 517 1458 590 1485; x_wconf 96'>what</span>
|
<span class='ocrx_word' id='word_1_194' title='bbox 517 1458 590 1485; x_wconf 96'>what</span>
|
||||||
<span class='ocrx_word' id='word_1_195' title='bbox 600 1466 654 1491; x_wconf 93'>you</span>
|
<span class='ocrx_word' id='word_1_195' title='bbox 600 1466 654 1491; x_wconf 93'>you</span>
|
||||||
@@ -316,7 +316,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_13' title="bbox 346 1497 1245 1531">
|
<div class='ocr_carea' id='block_1_13' title="bbox 346 1497 1245 1531">
|
||||||
<p class='ocr_par' id='par_1_16' lang='eng' title="bbox 346 1497 1245 1531">
|
<p class='ocr_par' id='par_1_16' lang='eng' title="bbox 346 1497 1245 1531">
|
||||||
<span class='ocr_line' id='line_1_24' title="bbox 346 1497 1245 1531; baseline 0.001 -7; x_size 35; x_descenders 7; x_ascenders 9">
|
<span class='ocr_textfloat' id='line_1_24' title="bbox 346 1497 1245 1531; baseline 0.001 -7; x_size 35; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_202' title='bbox 346 1497 494 1524; x_wconf 96'>corrected!</span>
|
<span class='ocrx_word' id='word_1_202' title='bbox 346 1497 494 1524; x_wconf 96'>corrected!</span>
|
||||||
<span class='ocrx_word' id='word_1_203' title='bbox 508 1497 628 1530; x_wconf 96'>(Timing</span>
|
<span class='ocrx_word' id='word_1_203' title='bbox 508 1497 628 1530; x_wconf 96'>(Timing</span>
|
||||||
<span class='ocrx_word' id='word_1_204' title='bbox 638 1497 791 1525; x_wconf 95'>correction</span>
|
<span class='ocrx_word' id='word_1_204' title='bbox 638 1497 791 1525; x_wconf 95'>correction</span>
|
||||||
@@ -343,8 +343,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_219' title='bbox 1111 1537 1186 1564; x_wconf 96'>track</span>
|
<span class='ocrx_word' id='word_1_219' title='bbox 1111 1537 1186 1564; x_wconf 96'>track</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_26' title="bbox 347 1575 1052 1610; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_26' title="bbox 347 1575 1052 1610; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_220' title='bbox 347 1591 381 1594; x_wconf 0'>—</span>
|
<span class='ocrx_word' id='word_1_220' title='bbox 347 1591 372 1594; x_wconf 0'>—</span>
|
||||||
<span class='ocrx_word' id='word_1_221' title='bbox 382 1575 495 1609; x_wconf 0'>existing</span>
|
<span class='ocrx_word' id='word_1_221' title='bbox 371 1575 495 1609; x_wconf 0'>existing</span>
|
||||||
<span class='ocrx_word' id='word_1_222' title='bbox 505 1580 582 1603; x_wconf 96'>notes</span>
|
<span class='ocrx_word' id='word_1_222' title='bbox 505 1580 582 1603; x_wconf 96'>notes</span>
|
||||||
<span class='ocrx_word' id='word_1_223' title='bbox 593 1584 637 1603; x_wconf 96'>are</span>
|
<span class='ocrx_word' id='word_1_223' title='bbox 593 1584 637 1603; x_wconf 96'>are</span>
|
||||||
<span class='ocrx_word' id='word_1_224' title='bbox 648 1580 696 1603; x_wconf 97'>not</span>
|
<span class='ocrx_word' id='word_1_224' title='bbox 648 1580 696 1603; x_wconf 97'>not</span>
|
||||||
@@ -358,10 +358,10 @@
|
|||||||
<span class='ocr_line' id='line_1_27' title="bbox 384 1616 1199 1648; baseline 0.001 -6; x_size 32; x_descenders 5; x_ascenders 8">
|
<span class='ocr_line' id='line_1_27' title="bbox 384 1616 1199 1648; baseline 0.001 -6; x_size 32; x_descenders 5; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_228' title='bbox 384 1616 471 1642; x_wconf 96'>FAST</span>
|
<span class='ocrx_word' id='word_1_228' title='bbox 384 1616 471 1642; x_wconf 96'>FAST</span>
|
||||||
<span class='ocrx_word' id='word_1_229' title='bbox 481 1616 671 1648; x_wconf 96'>FORWARD,</span>
|
<span class='ocrx_word' id='word_1_229' title='bbox 481 1616 671 1648; x_wconf 96'>FORWARD,</span>
|
||||||
<span class='ocrx_word' id='word_1_230' title='bbox 684 1617 838 1643; x_wconf 95'>REWIND,</span>
|
<span class='ocrx_word' id='word_1_230' title='bbox 684 1617 844 1648; x_wconf 95'>REWIND,</span>
|
||||||
<span class='ocrx_word' id='word_1_231' title='bbox 839 1616 912 1648; x_wconf 95'>and</span>
|
<span class='ocrx_word' id='word_1_231' title='bbox 857 1616 912 1643; x_wconf 95'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_232' title='bbox 924 1616 1045 1643; x_wconf 96'>LOCATE</span>
|
<span class='ocrx_word' id='word_1_232' title='bbox 924 1616 1068 1643; x_wconf 96'>LOCATE</span>
|
||||||
<span class='ocrx_word' id='word_1_233' title='bbox 1046 1616 1199 1643; x_wconf 96'>controls</span>
|
<span class='ocrx_word' id='word_1_233' title='bbox 1079 1616 1199 1643; x_wconf 95'>controls</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_28' title="bbox 346 1655 1202 1689; baseline 0 -7; x_size 34; x_descenders 6; x_ascenders 9">
|
<span class='ocr_line' id='line_1_28' title="bbox 346 1655 1202 1689; baseline 0 -7; x_size 34; x_descenders 6; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_234' title='bbox 346 1663 409 1688; x_wconf 92'>may</span>
|
<span class='ocrx_word' id='word_1_234' title='bbox 346 1663 409 1688; x_wconf 92'>may</span>
|
||||||
@@ -385,8 +385,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_250' title='bbox 860 1696 897 1722; x_wconf 93'>To</span>
|
<span class='ocrx_word' id='word_1_250' title='bbox 860 1696 897 1722; x_wconf 93'>To</span>
|
||||||
<span class='ocrx_word' id='word_1_251' title='bbox 908 1695 1028 1722; x_wconf 93'>overdub</span>
|
<span class='ocrx_word' id='word_1_251' title='bbox 908 1695 1028 1722; x_wconf 93'>overdub</span>
|
||||||
<span class='ocrx_word' id='word_1_252' title='bbox 1039 1703 1056 1722; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_252' title='bbox 1039 1703 1056 1722; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_253' title='bbox 1066 1703 1099 1722; x_wconf 96'>new</span>
|
<span class='ocrx_word' id='word_1_253' title='bbox 1066 1703 1125 1722; x_wconf 96'>new</span>
|
||||||
<span class='ocrx_word' id='word_1_254' title='bbox 1100 1699 1204 1728; x_wconf 96'>part,</span>
|
<span class='ocrx_word' id='word_1_254' title='bbox 1135 1699 1204 1728; x_wconf 96'>part,</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_30' title="bbox 347 1733 1150 1768; baseline 0.001 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
<span class='ocr_line' id='line_1_30' title="bbox 347 1733 1150 1768; baseline 0.001 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_255' title='bbox 347 1733 426 1761; x_wconf 97'>select</span>
|
<span class='ocrx_word' id='word_1_255' title='bbox 347 1733 426 1761; x_wconf 97'>select</span>
|
||||||
@@ -400,9 +400,9 @@
|
|||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_31' title="bbox 346 1773 1203 1808; baseline 0.001 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_31' title="bbox 346 1773 1203 1808; baseline 0.001 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_263' title='bbox 346 1774 448 1806; x_wconf 96'>record,</span>
|
<span class='ocrx_word' id='word_1_263' title='bbox 346 1774 448 1806; x_wconf 96'>record,</span>
|
||||||
<span class='ocrx_word' id='word_1_264' title='bbox 460 1774 506 1801; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_264' title='bbox 460 1774 506 1801; x_wconf 97'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_265' title='bbox 517 1773 577 1801; x_wconf 96'>first</span>
|
<span class='ocrx_word' id='word_1_265' title='bbox 503 1769 577 1812; x_wconf 96'>first</span>
|
||||||
<span class='ocrx_word' id='word_1_266' title='bbox 587 1774 662 1801; x_wconf 96'>track</span>
|
<span class='ocrx_word' id='word_1_266' title='bbox 581 1774 658 1801; x_wconf 96'>track</span>
|
||||||
<span class='ocrx_word' id='word_1_267' title='bbox 673 1774 726 1801; x_wconf 96'>will</span>
|
<span class='ocrx_word' id='word_1_267' title='bbox 673 1774 726 1801; x_wconf 96'>will</span>
|
||||||
<span class='ocrx_word' id='word_1_268' title='bbox 736 1774 799 1807; x_wconf 96'>play</span>
|
<span class='ocrx_word' id='word_1_268' title='bbox 736 1774 799 1807; x_wconf 96'>play</span>
|
||||||
<span class='ocrx_word' id='word_1_269' title='bbox 809 1774 836 1801; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_269' title='bbox 809 1774 836 1801; x_wconf 96'>in</span>
|
||||||
@@ -412,12 +412,12 @@
|
|||||||
<span class='ocrx_word' id='word_1_273' title='bbox 1148 1782 1203 1807; x_wconf 96'>you</span>
|
<span class='ocrx_word' id='word_1_273' title='bbox 1148 1782 1203 1807; x_wconf 96'>you</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_32' title="bbox 346 1813 1191 1847; baseline 0.002 -8; x_size 33; x_descenders 5; x_ascenders 9">
|
<span class='ocr_line' id='line_1_32' title="bbox 346 1813 1191 1847; baseline 0.002 -8; x_size 33; x_descenders 5; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_274' title='bbox 346 1813 431 1840; x_wconf 94'>MUTE</span>
|
<span class='ocrx_word' id='word_1_274' title='bbox 346 1813 454 1840; x_wconf 95'>MUTE</span>
|
||||||
<span class='ocrx_word' id='word_1_275' title='bbox 431 1813 492 1845; x_wconf 94'>it,</span>
|
<span class='ocrx_word' id='word_1_275' title='bbox 464 1813 492 1845; x_wconf 95'>it,</span>
|
||||||
<span class='ocrx_word' id='word_1_276' title='bbox 505 1821 537 1840; x_wconf 95'>or</span>
|
<span class='ocrx_word' id='word_1_276' title='bbox 505 1821 537 1840; x_wconf 95'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_277' title='bbox 547 1813 642 1840; x_wconf 95'>SOLO</span>
|
<span class='ocrx_word' id='word_1_277' title='bbox 547 1813 642 1840; x_wconf 95'>SOLO</span>
|
||||||
<span class='ocrx_word' id='word_1_278' title='bbox 653 1814 756 1841; x_wconf 96'>another</span>
|
<span class='ocrx_word' id='word_1_278' title='bbox 653 1814 769 1841; x_wconf 95'>another</span>
|
||||||
<span class='ocrx_word' id='word_1_279' title='bbox 757 1814 875 1847; x_wconf 94'>track).</span>
|
<span class='ocrx_word' id='word_1_279' title='bbox 779 1814 875 1847; x_wconf 94'>track).</span>
|
||||||
<span class='ocrx_word' id='word_1_280' title='bbox 889 1814 920 1840; x_wconf 96'>In</span>
|
<span class='ocrx_word' id='word_1_280' title='bbox 889 1814 920 1840; x_wconf 96'>In</span>
|
||||||
<span class='ocrx_word' id='word_1_281' title='bbox 930 1813 984 1841; x_wconf 96'>this</span>
|
<span class='ocrx_word' id='word_1_281' title='bbox 930 1813 984 1841; x_wconf 96'>this</span>
|
||||||
<span class='ocrx_word' id='word_1_282' title='bbox 995 1822 1057 1847; x_wconf 96'>way,</span>
|
<span class='ocrx_word' id='word_1_282' title='bbox 995 1822 1057 1847; x_wconf 96'>way,</span>
|
||||||
@@ -431,8 +431,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_288' title='bbox 518 1853 552 1879; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_288' title='bbox 518 1853 552 1879; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_289' title='bbox 562 1853 748 1880; x_wconf 96'>overdubbed!</span>
|
<span class='ocrx_word' id='word_1_289' title='bbox 562 1853 748 1880; x_wconf 96'>overdubbed!</span>
|
||||||
<span class='ocrx_word' id='word_1_290' title='bbox 761 1853 808 1880; x_wconf 94'>All</span>
|
<span class='ocrx_word' id='word_1_290' title='bbox 761 1853 808 1880; x_wconf 94'>All</span>
|
||||||
<span class='ocrx_word' id='word_1_291' title='bbox 819 1854 892 1880; x_wconf 96'>MIDI</span>
|
<span class='ocrx_word' id='word_1_291' title='bbox 819 1854 905 1880; x_wconf 96'>MIDI</span>
|
||||||
<span class='ocrx_word' id='word_1_292' title='bbox 895 1853 1011 1880; x_wconf 96'>effects</span>
|
<span class='ocrx_word' id='word_1_292' title='bbox 917 1853 1011 1880; x_wconf 96'>effects</span>
|
||||||
<span class='ocrx_word' id='word_1_293' title='bbox 1022 1861 1067 1880; x_wconf 96'>are</span>
|
<span class='ocrx_word' id='word_1_293' title='bbox 1022 1861 1067 1880; x_wconf 96'>are</span>
|
||||||
<span class='ocrx_word' id='word_1_294' title='bbox 1076 1853 1205 1880; x_wconf 96'>recorded</span>
|
<span class='ocrx_word' id='word_1_294' title='bbox 1076 1853 1205 1880; x_wconf 96'>recorded</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -440,8 +440,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_295' title='bbox 346 1891 485 1925; x_wconf 96'>including</span>
|
<span class='ocrx_word' id='word_1_295' title='bbox 346 1891 485 1925; x_wconf 96'>including</span>
|
||||||
<span class='ocrx_word' id='word_1_296' title='bbox 495 1891 570 1925; x_wconf 96'>pitch</span>
|
<span class='ocrx_word' id='word_1_296' title='bbox 495 1891 570 1925; x_wconf 96'>pitch</span>
|
||||||
<span class='ocrx_word' id='word_1_297' title='bbox 580 1892 663 1924; x_wconf 96'>bend,</span>
|
<span class='ocrx_word' id='word_1_297' title='bbox 580 1892 663 1924; x_wconf 96'>bend,</span>
|
||||||
<span class='ocrx_word' id='word_1_298' title='bbox 675 1892 850 1920; x_wconf 96'>modulation,</span>
|
<span class='ocrx_word' id='word_1_298' title='bbox 675 1892 859 1924; x_wconf 96'>modulation,</span>
|
||||||
<span class='ocrx_word' id='word_1_299' title='bbox 854 1892 991 1926; x_wconf 93'>velocity,</span>
|
<span class='ocrx_word' id='word_1_299' title='bbox 872 1892 991 1926; x_wconf 93'>velocity,</span>
|
||||||
<span class='ocrx_word' id='word_1_300' title='bbox 1004 1892 1168 1924; x_wconf 92'>aftertouch,</span>
|
<span class='ocrx_word' id='word_1_300' title='bbox 1004 1892 1168 1924; x_wconf 92'>aftertouch,</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_35' title="bbox 346 1931 895 1965; baseline 0.002 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_35' title="bbox 346 1931 895 1965; baseline 0.002 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
@@ -465,13 +465,13 @@
|
|||||||
<span class='ocrx_word' id='word_1_307' title='bbox 383 2050 419 2076; x_wconf 96'>To</span>
|
<span class='ocrx_word' id='word_1_307' title='bbox 383 2050 419 2076; x_wconf 96'>To</span>
|
||||||
<span class='ocrx_word' id='word_1_308' title='bbox 430 2057 503 2076; x_wconf 95'>erase</span>
|
<span class='ocrx_word' id='word_1_308' title='bbox 430 2057 503 2076; x_wconf 95'>erase</span>
|
||||||
<span class='ocrx_word' id='word_1_309' title='bbox 514 2058 530 2077; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_309' title='bbox 514 2058 530 2077; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_310' title='bbox 540 2058 617 2077; x_wconf 96'>wrong</span>
|
<span class='ocrx_word' id='word_1_310' title='bbox 540 2058 634 2083; x_wconf 96'>wrong</span>
|
||||||
<span class='ocrx_word' id='word_1_311' title='bbox 618 2054 717 2083; x_wconf 96'>note,</span>
|
<span class='ocrx_word' id='word_1_311' title='bbox 644 2054 717 2082; x_wconf 96'>note,</span>
|
||||||
<span class='ocrx_word' id='word_1_312' title='bbox 729 2050 829 2083; x_wconf 96'>simply</span>
|
<span class='ocrx_word' id='word_1_312' title='bbox 729 2050 829 2083; x_wconf 96'>simply</span>
|
||||||
<span class='ocrx_word' id='word_1_313' title='bbox 839 2050 905 2077; x_wconf 96'>hold</span>
|
<span class='ocrx_word' id='word_1_313' title='bbox 839 2050 905 2077; x_wconf 96'>hold</span>
|
||||||
<span class='ocrx_word' id='word_1_314' title='bbox 916 2051 1013 2077; x_wconf 96'>ERASE</span>
|
<span class='ocrx_word' id='word_1_314' title='bbox 916 2051 1037 2077; x_wconf 96'>ERASE</span>
|
||||||
<span class='ocrx_word' id='word_1_315' title='bbox 1015 2051 1083 2077; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_315' title='bbox 1048 2050 1103 2077; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_316' title='bbox 1085 2050 1186 2084; x_wconf 96'>press</span>
|
<span class='ocrx_word' id='word_1_316' title='bbox 1113 2059 1186 2084; x_wconf 96'>press</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_38' title="bbox 346 2089 1212 2124; baseline 0.002 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_38' title="bbox 346 2089 1212 2124; baseline 0.002 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_317' title='bbox 346 2089 391 2116; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_317' title='bbox 346 2089 391 2116; x_wconf 96'>the</span>
|
||||||
@@ -480,16 +480,16 @@
|
|||||||
<span class='ocrx_word' id='word_1_320' title='bbox 515 2090 549 2117; x_wconf 97'>be</span>
|
<span class='ocrx_word' id='word_1_320' title='bbox 515 2090 549 2117; x_wconf 97'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_321' title='bbox 559 2090 652 2117; x_wconf 96'>erased</span>
|
<span class='ocrx_word' id='word_1_321' title='bbox 559 2090 652 2117; x_wconf 96'>erased</span>
|
||||||
<span class='ocrx_word' id='word_1_322' title='bbox 661 2090 718 2123; x_wconf 96'>just</span>
|
<span class='ocrx_word' id='word_1_322' title='bbox 661 2090 718 2123; x_wconf 96'>just</span>
|
||||||
<span class='ocrx_word' id='word_1_323' title='bbox 729 2090 808 2117; x_wconf 96'>before</span>
|
<span class='ocrx_word' id='word_1_323' title='bbox 729 2090 822 2117; x_wconf 96'>before</span>
|
||||||
<span class='ocrx_word' id='word_1_324' title='bbox 808 2090 852 2117; x_wconf 96'>it</span>
|
<span class='ocrx_word' id='word_1_324' title='bbox 833 2090 852 2117; x_wconf 96'>it</span>
|
||||||
<span class='ocrx_word' id='word_1_325' title='bbox 862 2090 937 2124; x_wconf 96'>plays</span>
|
<span class='ocrx_word' id='word_1_325' title='bbox 862 2090 937 2124; x_wconf 96'>plays</span>
|
||||||
<span class='ocrx_word' id='word_1_326' title='bbox 947 2090 975 2117; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_326' title='bbox 947 2090 975 2117; x_wconf 96'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_327' title='bbox 986 2090 1032 2118; x_wconf 93'>the</span>
|
<span class='ocrx_word' id='word_1_327' title='bbox 986 2090 1032 2118; x_wconf 93'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_328' title='bbox 1043 2098 1212 2124; x_wconf 88'>sequence—</span>
|
<span class='ocrx_word' id='word_1_328' title='bbox 1043 2098 1212 2124; x_wconf 88'>sequence—</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_39' title="bbox 346 2129 1134 2163; baseline 0.003 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_39' title="bbox 346 2129 1134 2163; baseline 0.003 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_329' title='bbox 346 2129 406 2156; x_wconf 96'>when</span>
|
<span class='ocrx_word' id='word_1_329' title='bbox 346 2129 425 2156; x_wconf 96'>when</span>
|
||||||
<span class='ocrx_word' id='word_1_330' title='bbox 407 2129 531 2162; x_wconf 96'>played</span>
|
<span class='ocrx_word' id='word_1_330' title='bbox 435 2129 531 2162; x_wconf 96'>played</span>
|
||||||
<span class='ocrx_word' id='word_1_331' title='bbox 542 2129 621 2161; x_wconf 96'>back,</span>
|
<span class='ocrx_word' id='word_1_331' title='bbox 542 2129 621 2161; x_wconf 96'>back,</span>
|
||||||
<span class='ocrx_word' id='word_1_332' title='bbox 633 2129 652 2156; x_wconf 96'>it</span>
|
<span class='ocrx_word' id='word_1_332' title='bbox 633 2129 652 2156; x_wconf 96'>it</span>
|
||||||
<span class='ocrx_word' id='word_1_333' title='bbox 663 2129 716 2156; x_wconf 96'>will</span>
|
<span class='ocrx_word' id='word_1_333' title='bbox 663 2129 716 2156; x_wconf 96'>will</span>
|
||||||
@@ -512,19 +512,19 @@
|
|||||||
<span class='ocrx_word' id='word_1_344' title='bbox 749 2169 829 2203; x_wconf 96'>using</span>
|
<span class='ocrx_word' id='word_1_344' title='bbox 749 2169 829 2203; x_wconf 96'>using</span>
|
||||||
<span class='ocrx_word' id='word_1_345' title='bbox 839 2169 885 2196; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_345' title='bbox 839 2169 885 2196; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_346' title='bbox 896 2170 1031 2196; x_wconf 96'>SINGLE</span>
|
<span class='ocrx_word' id='word_1_346' title='bbox 896 2170 1031 2196; x_wconf 96'>SINGLE</span>
|
||||||
<span class='ocrx_word' id='word_1_347' title='bbox 1042 2170 1107 2196; x_wconf 91'>STEP</span>
|
<span class='ocrx_word' id='word_1_347' title='bbox 1042 2170 1131 2196; x_wconf 91'>STEP</span>
|
||||||
<span class='ocrx_word' id='word_1_348' title='bbox 1109 2169 1220 2196; x_wconf 91'>func-</span>
|
<span class='ocrx_word' id='word_1_348' title='bbox 1143 2169 1220 2196; x_wconf 91'>func-</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_41' title="bbox 345 2207 1228 2242; baseline 0.002 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
<span class='ocr_line' id='line_1_41' title="bbox 345 2207 1228 2242; baseline 0.002 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_349' title='bbox 345 2207 412 2234; x_wconf 96'>tion.</span>
|
<span class='ocrx_word' id='word_1_349' title='bbox 345 2207 412 2234; x_wconf 96'>tion.</span>
|
||||||
<span class='ocrx_word' id='word_1_350' title='bbox 424 2208 461 2235; x_wconf 93'>To</span>
|
<span class='ocrx_word' id='word_1_350' title='bbox 424 2208 461 2235; x_wconf 93'>To</span>
|
||||||
<span class='ocrx_word' id='word_1_351' title='bbox 472 2208 592 2235; x_wconf 92'>overdub</span>
|
<span class='ocrx_word' id='word_1_351' title='bbox 472 2208 592 2235; x_wconf 91'>overdub</span>
|
||||||
<span class='ocrx_word' id='word_1_352' title='bbox 603 2212 667 2235; x_wconf 96'>notes</span>
|
<span class='ocrx_word' id='word_1_352' title='bbox 603 2212 680 2235; x_wconf 96'>notes</span>
|
||||||
<span class='ocrx_word' id='word_1_353' title='bbox 668 2212 718 2235; x_wconf 96'>at</span>
|
<span class='ocrx_word' id='word_1_353' title='bbox 691 2212 718 2235; x_wconf 96'>at</span>
|
||||||
<span class='ocrx_word' id='word_1_354' title='bbox 729 2208 841 2242; x_wconf 96'>specific</span>
|
<span class='ocrx_word' id='word_1_354' title='bbox 729 2208 841 2242; x_wconf 96'>specific</span>
|
||||||
<span class='ocrx_word' id='word_1_355' title='bbox 851 2209 943 2242; x_wconf 97'>points</span>
|
<span class='ocrx_word' id='word_1_355' title='bbox 851 2209 943 2242; x_wconf 97'>points</span>
|
||||||
<span class='ocrx_word' id='word_1_356' title='bbox 955 2208 1049 2236; x_wconf 97'>within</span>
|
<span class='ocrx_word' id='word_1_356' title='bbox 955 2208 1049 2236; x_wconf 96'>within</span>
|
||||||
<span class='ocrx_word' id='word_1_357' title='bbox 1060 2217 1076 2236; x_wconf 97'>a</span>
|
<span class='ocrx_word' id='word_1_357' title='bbox 1060 2217 1076 2236; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_358' title='bbox 1086 2216 1228 2242; x_wconf 96'>sequence,</span>
|
<span class='ocrx_word' id='word_1_358' title='bbox 1086 2216 1228 2242; x_wconf 96'>sequence,</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
@@ -544,10 +544,10 @@
|
|||||||
<span class='ocrx_word' id='word_1_362' title='bbox 1404 1297 1452 1316; x_wconf 96'>use</span>
|
<span class='ocrx_word' id='word_1_362' title='bbox 1404 1297 1452 1316; x_wconf 96'>use</span>
|
||||||
<span class='ocrx_word' id='word_1_363' title='bbox 1463 1290 1615 1321; x_wconf 96'>LOCATE,</span>
|
<span class='ocrx_word' id='word_1_363' title='bbox 1463 1290 1615 1321; x_wconf 96'>LOCATE,</span>
|
||||||
<span class='ocrx_word' id='word_1_364' title='bbox 1628 1290 1716 1316; x_wconf 96'>FAST</span>
|
<span class='ocrx_word' id='word_1_364' title='bbox 1628 1290 1716 1316; x_wconf 96'>FAST</span>
|
||||||
<span class='ocrx_word' id='word_1_365' title='bbox 1726 1289 1910 1316; x_wconf 95'>FORWARD,</span>
|
<span class='ocrx_word' id='word_1_365' title='bbox 1726 1289 1917 1321; x_wconf 95'>FORWARD,</span>
|
||||||
<span class='ocrx_word' id='word_1_366' title='bbox 1911 1297 1960 1321; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_366' title='bbox 1929 1297 1960 1317; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_367' title='bbox 1972 1290 2098 1316; x_wconf 95'>REWIND</span>
|
<span class='ocrx_word' id='word_1_367' title='bbox 1972 1290 2126 1316; x_wconf 95'>REWIND</span>
|
||||||
<span class='ocrx_word' id='word_1_368' title='bbox 2100 1290 2165 1317; x_wconf 95'>to</span>
|
<span class='ocrx_word' id='word_1_368' title='bbox 2136 1294 2165 1317; x_wconf 95'>to</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_44' title="bbox 1297 1328 2033 1362; baseline 0.001 -7; x_size 32; x_descenders 5; x_ascenders 8">
|
<span class='ocr_line' id='line_1_44' title="bbox 1297 1328 2033 1362; baseline 0.001 -7; x_size 32; x_descenders 5; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_369' title='bbox 1297 1328 1356 1355; x_wconf 96'>find</span>
|
<span class='ocrx_word' id='word_1_369' title='bbox 1297 1328 1356 1355; x_wconf 96'>find</span>
|
||||||
@@ -564,8 +564,8 @@
|
|||||||
<p class='ocr_par' id='par_1_24' lang='eng' title="bbox 1295 1368 2160 1593">
|
<p class='ocr_par' id='par_1_24' lang='eng' title="bbox 1295 1368 2160 1593">
|
||||||
<span class='ocr_line' id='line_1_45' title="bbox 1332 1368 2160 1402; baseline 0.002 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_45' title="bbox 1332 1368 2160 1402; baseline 0.002 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_377' title='bbox 1332 1368 1391 1395; x_wconf 93'>The</span>
|
<span class='ocrx_word' id='word_1_377' title='bbox 1332 1368 1391 1395; x_wconf 93'>The</span>
|
||||||
<span class='ocrx_word' id='word_1_378' title='bbox 1402 1368 1626 1395; x_wconf 91'>INSERT/COPY</span>
|
<span class='ocrx_word' id='word_1_378' title='bbox 1402 1368 1650 1395; x_wconf 91'>INSERT/COPY</span>
|
||||||
<span class='ocrx_word' id='word_1_379' title='bbox 1626 1368 1789 1395; x_wconf 96'>function</span>
|
<span class='ocrx_word' id='word_1_379' title='bbox 1662 1368 1789 1395; x_wconf 96'>function</span>
|
||||||
<span class='ocrx_word' id='word_1_380' title='bbox 1800 1368 1893 1395; x_wconf 96'>allows</span>
|
<span class='ocrx_word' id='word_1_380' title='bbox 1800 1368 1893 1395; x_wconf 96'>allows</span>
|
||||||
<span class='ocrx_word' id='word_1_381' title='bbox 1904 1376 1958 1402; x_wconf 96'>you</span>
|
<span class='ocrx_word' id='word_1_381' title='bbox 1904 1376 1958 1402; x_wconf 96'>you</span>
|
||||||
<span class='ocrx_word' id='word_1_382' title='bbox 1968 1373 1997 1395; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_382' title='bbox 1968 1373 1997 1395; x_wconf 96'>to</span>
|
||||||
@@ -580,8 +580,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_389' title='bbox 1616 1407 1796 1435; x_wconf 91'>another—in</span>
|
<span class='ocrx_word' id='word_1_389' title='bbox 1616 1407 1796 1435; x_wconf 91'>another—in</span>
|
||||||
<span class='ocrx_word' id='word_1_390' title='bbox 1806 1408 1852 1435; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_390' title='bbox 1806 1408 1852 1435; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_391' title='bbox 1863 1416 1937 1435; x_wconf 96'>same</span>
|
<span class='ocrx_word' id='word_1_391' title='bbox 1863 1416 1937 1435; x_wconf 96'>same</span>
|
||||||
<span class='ocrx_word' id='word_1_392' title='bbox 1948 1416 2067 1441; x_wconf 96'>sequence</span>
|
<span class='ocrx_word' id='word_1_392' title='bbox 1948 1416 2083 1441; x_wconf 96'>sequence</span>
|
||||||
<span class='ocrx_word' id='word_1_393' title='bbox 2068 1416 2125 1435; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_393' title='bbox 2093 1416 2125 1435; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_394' title='bbox 2135 1416 2151 1435; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_394' title='bbox 2135 1416 2151 1435; x_wconf 96'>a</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_47' title="bbox 1296 1447 2160 1481; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_47' title="bbox 1296 1447 2160 1481; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
@@ -603,15 +603,15 @@
|
|||||||
<span class='ocrx_word' id='word_1_408' title='bbox 1449 1487 1571 1514; x_wconf 96'>between</span>
|
<span class='ocrx_word' id='word_1_408' title='bbox 1449 1487 1571 1514; x_wconf 96'>between</span>
|
||||||
<span class='ocrx_word' id='word_1_409' title='bbox 1582 1487 1628 1514; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_409' title='bbox 1582 1487 1628 1514; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_410' title='bbox 1638 1487 1740 1514; x_wconf 96'>second</span>
|
<span class='ocrx_word' id='word_1_410' title='bbox 1638 1487 1740 1514; x_wconf 96'>second</span>
|
||||||
<span class='ocrx_word' id='word_1_411' title='bbox 1751 1487 1838 1514; x_wconf 96'>chorus</span>
|
<span class='ocrx_word' id='word_1_411' title='bbox 1751 1487 1852 1514; x_wconf 96'>chorus</span>
|
||||||
<span class='ocrx_word' id='word_1_412' title='bbox 1841 1495 1899 1514; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_412' title='bbox 1863 1487 1919 1514; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_413' title='bbox 1901 1487 1975 1514; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_413' title='bbox 1929 1487 1975 1514; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_414' title='bbox 1985 1487 2087 1521; x_wconf 96'>bridge.</span>
|
<span class='ocrx_word' id='word_1_414' title='bbox 1985 1487 2087 1521; x_wconf 96'>bridge.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_49' title="bbox 1296 1527 2047 1560; baseline 0.003 -8; x_size 32; x_descenders 6; x_ascenders 7">
|
<span class='ocr_line' id='line_1_49' title="bbox 1296 1527 2047 1560; baseline 0.003 -8; x_size 32; x_descenders 6; x_ascenders 7">
|
||||||
<span class='ocrx_word' id='word_1_415' title='bbox 1296 1527 1441 1553; x_wconf 95'>DELETE</span>
|
<span class='ocrx_word' id='word_1_415' title='bbox 1296 1527 1441 1553; x_wconf 95'>DELETE</span>
|
||||||
<span class='ocrx_word' id='word_1_416' title='bbox 1453 1527 1527 1553; x_wconf 96'>BARS</span>
|
<span class='ocrx_word' id='word_1_416' title='bbox 1453 1527 1546 1553; x_wconf 96'>BARS</span>
|
||||||
<span class='ocrx_word' id='word_1_417' title='bbox 1529 1527 1681 1559; x_wconf 96'>operates</span>
|
<span class='ocrx_word' id='word_1_417' title='bbox 1557 1531 1681 1559; x_wconf 96'>operates</span>
|
||||||
<span class='ocrx_word' id='word_1_418' title='bbox 1691 1527 1737 1553; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_418' title='bbox 1691 1527 1737 1553; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_419' title='bbox 1748 1535 1823 1554; x_wconf 96'>same</span>
|
<span class='ocrx_word' id='word_1_419' title='bbox 1748 1535 1823 1554; x_wconf 96'>same</span>
|
||||||
<span class='ocrx_word' id='word_1_420' title='bbox 1833 1535 1891 1560; x_wconf 96'>way</span>
|
<span class='ocrx_word' id='word_1_420' title='bbox 1833 1535 1891 1560; x_wconf 96'>way</span>
|
||||||
@@ -619,8 +619,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_422' title='bbox 1940 1535 2047 1554; x_wconf 96'>remove</span>
|
<span class='ocrx_word' id='word_1_422' title='bbox 1940 1535 2047 1554; x_wconf 96'>remove</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_50' title="bbox 1295 1565 1577 1593; baseline 0.004 -1; x_size 34.748871; x_descenders 6.7488689; x_ascenders 9">
|
<span class='ocr_line' id='line_1_50' title="bbox 1295 1565 1577 1593; baseline 0.004 -1; x_size 34.748871; x_descenders 6.7488689; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_423' title='bbox 1295 1569 1422 1592; x_wconf 96'>unwanted</span>
|
<span class='ocrx_word' id='word_1_423' title='bbox 1295 1565 1441 1592; x_wconf 96'>unwanted</span>
|
||||||
<span class='ocrx_word' id='word_1_424' title='bbox 1423 1565 1577 1593; x_wconf 95'>sections,</span>
|
<span class='ocrx_word' id='word_1_424' title='bbox 1452 1565 1577 1593; x_wconf 95'>sections,</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
@@ -636,12 +636,12 @@
|
|||||||
<p class='ocr_par' id='par_1_26' lang='eng' title="bbox 1295 1690 2215 1960">
|
<p class='ocr_par' id='par_1_26' lang='eng' title="bbox 1295 1690 2215 1960">
|
||||||
<span class='ocr_line' id='line_1_52' title="bbox 1333 1690 2146 1723; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_52' title="bbox 1333 1690 2146 1723; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_428' title='bbox 1333 1690 1394 1717; x_wconf 96'>One</span>
|
<span class='ocrx_word' id='word_1_428' title='bbox 1333 1690 1394 1717; x_wconf 96'>One</span>
|
||||||
<span class='ocrx_word' id='word_1_429' title='bbox 1404 1698 1446 1717; x_wconf 96'>way</span>
|
<span class='ocrx_word' id='word_1_429' title='bbox 1404 1698 1462 1723; x_wconf 96'>way</span>
|
||||||
<span class='ocrx_word' id='word_1_430' title='bbox 1445 1694 1500 1723; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_430' title='bbox 1472 1694 1500 1717; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_431' title='bbox 1511 1694 1598 1717; x_wconf 96'>create</span>
|
<span class='ocrx_word' id='word_1_431' title='bbox 1511 1694 1598 1717; x_wconf 96'>create</span>
|
||||||
<span class='ocrx_word' id='word_1_432' title='bbox 1608 1698 1625 1717; x_wconf 96'>a</span>
|
<span class='ocrx_word' id='word_1_432' title='bbox 1608 1698 1625 1717; x_wconf 96'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_433' title='bbox 1635 1698 1687 1717; x_wconf 95'>song</span>
|
<span class='ocrx_word' id='word_1_433' title='bbox 1635 1698 1704 1723; x_wconf 95'>song</span>
|
||||||
<span class='ocrx_word' id='word_1_434' title='bbox 1688 1690 1736 1723; x_wconf 95'>is</span>
|
<span class='ocrx_word' id='word_1_434' title='bbox 1715 1690 1736 1717; x_wconf 95'>is</span>
|
||||||
<span class='ocrx_word' id='word_1_435' title='bbox 1747 1694 1776 1717; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_435' title='bbox 1747 1694 1776 1717; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_436' title='bbox 1787 1690 1880 1717; x_wconf 96'>record</span>
|
<span class='ocrx_word' id='word_1_436' title='bbox 1787 1690 1880 1717; x_wconf 96'>record</span>
|
||||||
<span class='ocrx_word' id='word_1_437' title='bbox 1891 1690 1958 1717; x_wconf 96'>each</span>
|
<span class='ocrx_word' id='word_1_437' title='bbox 1891 1690 1958 1717; x_wconf 96'>each</span>
|
||||||
@@ -657,8 +657,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_445' title='bbox 1592 1730 1644 1756; x_wconf 96'>999</span>
|
<span class='ocrx_word' id='word_1_445' title='bbox 1592 1730 1644 1756; x_wconf 96'>999</span>
|
||||||
<span class='ocrx_word' id='word_1_446' title='bbox 1654 1729 1738 1762; x_wconf 96'>bars).</span>
|
<span class='ocrx_word' id='word_1_446' title='bbox 1654 1729 1738 1762; x_wconf 96'>bars).</span>
|
||||||
<span class='ocrx_word' id='word_1_447' title='bbox 1751 1729 1878 1757; x_wconf 96'>Another</span>
|
<span class='ocrx_word' id='word_1_447' title='bbox 1751 1729 1878 1757; x_wconf 96'>Another</span>
|
||||||
<span class='ocrx_word' id='word_1_448' title='bbox 1888 1737 1930 1757; x_wconf 96'>way</span>
|
<span class='ocrx_word' id='word_1_448' title='bbox 1888 1737 1945 1762; x_wconf 96'>way</span>
|
||||||
<span class='ocrx_word' id='word_1_449' title='bbox 1929 1729 1977 1762; x_wconf 96'>is</span>
|
<span class='ocrx_word' id='word_1_449' title='bbox 1956 1729 1977 1757; x_wconf 96'>is</span>
|
||||||
<span class='ocrx_word' id='word_1_450' title='bbox 1987 1733 2016 1757; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_450' title='bbox 1987 1733 2016 1757; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_451' title='bbox 2027 1729 2121 1757; x_wconf 96'>record</span>
|
<span class='ocrx_word' id='word_1_451' title='bbox 2027 1729 2121 1757; x_wconf 96'>record</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -667,8 +667,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_453' title='bbox 1373 1768 1448 1796; x_wconf 96'>basic</span>
|
<span class='ocrx_word' id='word_1_453' title='bbox 1373 1768 1448 1796; x_wconf 96'>basic</span>
|
||||||
<span class='ocrx_word' id='word_1_454' title='bbox 1458 1768 1562 1796; x_wconf 96'>section</span>
|
<span class='ocrx_word' id='word_1_454' title='bbox 1458 1768 1562 1796; x_wconf 96'>section</span>
|
||||||
<span class='ocrx_word' id='word_1_455' title='bbox 1574 1769 1666 1802; x_wconf 96'>(verse,</span>
|
<span class='ocrx_word' id='word_1_455' title='bbox 1574 1769 1666 1802; x_wconf 96'>(verse,</span>
|
||||||
<span class='ocrx_word' id='word_1_456' title='bbox 1679 1769 1779 1796; x_wconf 96'>chorus,</span>
|
<span class='ocrx_word' id='word_1_456' title='bbox 1679 1769 1788 1801; x_wconf 96'>chorus,</span>
|
||||||
<span class='ocrx_word' id='word_1_457' title='bbox 1782 1769 1865 1802; x_wconf 96'>etc.)</span>
|
<span class='ocrx_word' id='word_1_457' title='bbox 1800 1769 1865 1802; x_wconf 96'>etc.)</span>
|
||||||
<span class='ocrx_word' id='word_1_458' title='bbox 1876 1768 1904 1795; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_458' title='bbox 1876 1768 1904 1795; x_wconf 96'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_459' title='bbox 1914 1768 2066 1796; x_wconf 96'>individual</span>
|
<span class='ocrx_word' id='word_1_459' title='bbox 1914 1768 2066 1796; x_wconf 96'>individual</span>
|
||||||
</span>
|
</span>
|
||||||
@@ -678,23 +678,23 @@
|
|||||||
<span class='ocrx_word' id='word_1_462' title='bbox 1538 1816 1587 1835; x_wconf 96'>use</span>
|
<span class='ocrx_word' id='word_1_462' title='bbox 1538 1816 1587 1835; x_wconf 96'>use</span>
|
||||||
<span class='ocrx_word' id='word_1_463' title='bbox 1597 1808 1643 1835; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_463' title='bbox 1597 1808 1643 1835; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_464' title='bbox 1653 1809 1799 1835; x_wconf 96'>CREATE</span>
|
<span class='ocrx_word' id='word_1_464' title='bbox 1653 1809 1799 1835; x_wconf 96'>CREATE</span>
|
||||||
<span class='ocrx_word' id='word_1_465' title='bbox 1810 1808 1883 1835; x_wconf 96'>SONG</span>
|
<span class='ocrx_word' id='word_1_465' title='bbox 1810 1808 1911 1835; x_wconf 96'>SONG</span>
|
||||||
<span class='ocrx_word' id='word_1_466' title='bbox 1885 1808 2050 1836; x_wconf 96'>function</span>
|
<span class='ocrx_word' id='word_1_466' title='bbox 1923 1808 2050 1836; x_wconf 96'>function</span>
|
||||||
<span class='ocrx_word' id='word_1_467' title='bbox 2060 1812 2089 1835; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_467' title='bbox 2060 1812 2089 1835; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_468' title='bbox 2103 1808 2215 1836; x_wconf 91'>“chain”</span>
|
<span class='ocrx_word' id='word_1_468' title='bbox 2103 1808 2215 1836; x_wconf 93'>“chain”</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_56' title="bbox 1295 1847 2135 1881; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_56' title="bbox 1295 1847 2135 1881; baseline 0.001 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_469' title='bbox 1295 1848 1370 1874; x_wconf 96'>them</span>
|
<span class='ocrx_word' id='word_1_469' title='bbox 1295 1848 1370 1874; x_wconf 96'>them</span>
|
||||||
<span class='ocrx_word' id='word_1_470' title='bbox 1381 1848 1508 1881; x_wconf 95'>together.</span>
|
<span class='ocrx_word' id='word_1_470' title='bbox 1381 1848 1508 1881; x_wconf 95'>together.</span>
|
||||||
<span class='ocrx_word' id='word_1_471' title='bbox 1521 1848 1667 1875; x_wconf 96'>CREATE</span>
|
<span class='ocrx_word' id='word_1_471' title='bbox 1521 1848 1667 1875; x_wconf 96'>CREATE</span>
|
||||||
<span class='ocrx_word' id='word_1_472' title='bbox 1678 1848 1751 1875; x_wconf 96'>SONG</span>
|
<span class='ocrx_word' id='word_1_472' title='bbox 1678 1848 1779 1875; x_wconf 96'>SONG</span>
|
||||||
<span class='ocrx_word' id='word_1_473' title='bbox 1753 1847 1842 1874; x_wconf 96'>will</span>
|
<span class='ocrx_word' id='word_1_473' title='bbox 1789 1847 1842 1874; x_wconf 96'>will</span>
|
||||||
<span class='ocrx_word' id='word_1_474' title='bbox 1853 1848 1918 1875; x_wconf 96'>then</span>
|
<span class='ocrx_word' id='word_1_474' title='bbox 1853 1848 1918 1875; x_wconf 96'>then</span>
|
||||||
<span class='ocrx_word' id='word_1_475' title='bbox 1929 1848 2135 1881; x_wconf 96'>automatically</span>
|
<span class='ocrx_word' id='word_1_475' title='bbox 1929 1848 2135 1881; x_wconf 96'>automatically</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_57' title="bbox 1296 1887 2162 1920; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_57' title="bbox 1296 1887 2162 1920; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_476' title='bbox 1296 1895 1349 1920; x_wconf 96'>copy</span>
|
<span class='ocrx_word' id='word_1_476' title='bbox 1296 1895 1366 1920; x_wconf 96'>copy</span>
|
||||||
<span class='ocrx_word' id='word_1_477' title='bbox 1349 1887 1412 1920; x_wconf 96'>all</span>
|
<span class='ocrx_word' id='word_1_477' title='bbox 1377 1887 1412 1914; x_wconf 96'>all</span>
|
||||||
<span class='ocrx_word' id='word_1_478' title='bbox 1422 1887 1468 1914; x_wconf 96'>the</span>
|
<span class='ocrx_word' id='word_1_478' title='bbox 1422 1887 1468 1914; x_wconf 96'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_479' title='bbox 1478 1891 1552 1920; x_wconf 96'>parts</span>
|
<span class='ocrx_word' id='word_1_479' title='bbox 1478 1891 1552 1920; x_wconf 96'>parts</span>
|
||||||
<span class='ocrx_word' id='word_1_480' title='bbox 1563 1887 1621 1914; x_wconf 95'>into</span>
|
<span class='ocrx_word' id='word_1_480' title='bbox 1563 1887 1621 1914; x_wconf 95'>into</span>
|
||||||
@@ -714,8 +714,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_492' title='bbox 1540 1926 1590 1953; x_wconf 96'>few</span>
|
<span class='ocrx_word' id='word_1_492' title='bbox 1540 1926 1590 1953; x_wconf 96'>few</span>
|
||||||
<span class='ocrx_word' id='word_1_493' title='bbox 1601 1927 1664 1953; x_wconf 96'>bars</span>
|
<span class='ocrx_word' id='word_1_493' title='bbox 1601 1927 1664 1953; x_wconf 96'>bars</span>
|
||||||
<span class='ocrx_word' id='word_1_494' title='bbox 1675 1931 1704 1954; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_494' title='bbox 1675 1931 1704 1954; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_495' title='bbox 1715 1935 1795 1960; x_wconf 96'>repeat</span>
|
<span class='ocrx_word' id='word_1_495' title='bbox 1715 1931 1806 1960; x_wconf 96'>repeat</span>
|
||||||
<span class='ocrx_word' id='word_1_496' title='bbox 1795 1926 1955 1960; x_wconf 96'>infinitely,</span>
|
<span class='ocrx_word' id='word_1_496' title='bbox 1816 1926 1955 1960; x_wconf 96'>infinitely,</span>
|
||||||
<span class='ocrx_word' id='word_1_497' title='bbox 1968 1926 2011 1954; x_wconf 95'>for</span>
|
<span class='ocrx_word' id='word_1_497' title='bbox 1968 1926 2011 1954; x_wconf 95'>for</span>
|
||||||
<span class='ocrx_word' id='word_1_498' title='bbox 2022 1935 2038 1954; x_wconf 93'>a</span>
|
<span class='ocrx_word' id='word_1_498' title='bbox 2022 1935 2038 1954; x_wconf 93'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_499' title='bbox 2049 1927 2169 1954; x_wconf 92'>fadeout.</span>
|
<span class='ocrx_word' id='word_1_499' title='bbox 2049 1927 2169 1954; x_wconf 92'>fadeout.</span>
|
||||||
@@ -757,8 +757,8 @@
|
|||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_62' title="bbox 1293 2130 2156 2164; baseline 0 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_62' title="bbox 1293 2130 2156 2164; baseline 0 -7; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_522' title='bbox 1293 2130 1339 2157; x_wconf 93'>the</span>
|
<span class='ocrx_word' id='word_1_522' title='bbox 1293 2130 1339 2157; x_wconf 93'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_523' title='bbox 1350 2130 1562 2164; x_wconf 90'>LinnSequencer</span>
|
<span class='ocrx_word' id='word_1_523' title='bbox 1350 2130 1576 2164; x_wconf 90'>LinnSequencer</span>
|
||||||
<span class='ocrx_word' id='word_1_524' title='bbox 1564 2130 1607 2157; x_wconf 97'>is</span>
|
<span class='ocrx_word' id='word_1_524' title='bbox 1586 2130 1607 2157; x_wconf 97'>is</span>
|
||||||
<span class='ocrx_word' id='word_1_525' title='bbox 1619 2130 1747 2164; x_wconf 96'>designed</span>
|
<span class='ocrx_word' id='word_1_525' title='bbox 1619 2130 1747 2164; x_wconf 96'>designed</span>
|
||||||
<span class='ocrx_word' id='word_1_526' title='bbox 1758 2134 1787 2157; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_526' title='bbox 1758 2134 1787 2157; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_527' title='bbox 1798 2130 1834 2157; x_wconf 96'>let</span>
|
<span class='ocrx_word' id='word_1_527' title='bbox 1798 2130 1834 2157; x_wconf 96'>let</span>
|
||||||
@@ -767,8 +767,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_530' title='bbox 2063 2130 2156 2157; x_wconf 96'>record</span>
|
<span class='ocrx_word' id='word_1_530' title='bbox 2063 2130 2156 2157; x_wconf 96'>record</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_63' title="bbox 1294 2169 2145 2203; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_line' id='line_1_63' title="bbox 1294 2169 2145 2203; baseline 0 -6; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_531' title='bbox 1294 2178 1330 2197; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_531' title='bbox 1294 2170 1349 2197; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_532' title='bbox 1331 2169 1414 2197; x_wconf 96'>edit</span>
|
<span class='ocrx_word' id='word_1_532' title='bbox 1360 2169 1414 2197; x_wconf 96'>edit</span>
|
||||||
<span class='ocrx_word' id='word_1_533' title='bbox 1425 2170 1503 2197; x_wconf 96'>while</span>
|
<span class='ocrx_word' id='word_1_533' title='bbox 1425 2170 1503 2197; x_wconf 96'>while</span>
|
||||||
<span class='ocrx_word' id='word_1_534' title='bbox 1514 2170 1643 2203; x_wconf 96'>devoting</span>
|
<span class='ocrx_word' id='word_1_534' title='bbox 1514 2170 1643 2203; x_wconf 96'>devoting</span>
|
||||||
<span class='ocrx_word' id='word_1_535' title='bbox 1653 2178 1721 2203; x_wconf 96'>your</span>
|
<span class='ocrx_word' id='word_1_535' title='bbox 1653 2178 1721 2203; x_wconf 96'>your</span>
|
||||||
@@ -792,7 +792,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_21' title="bbox 347 2343 2171 2378">
|
<div class='ocr_carea' id='block_1_21' title="bbox 347 2343 2171 2378">
|
||||||
<p class='ocr_par' id='par_1_29' lang='eng' title="bbox 347 2343 2171 2378">
|
<p class='ocr_par' id='par_1_29' lang='eng' title="bbox 347 2343 2171 2378">
|
||||||
<span class='ocr_line' id='line_1_65' title="bbox 347 2343 2171 2378; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_header' id='line_1_65' title="bbox 347 2343 2171 2378; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_549' title='bbox 347 2350 361 2363; x_wconf 58'>*</span>
|
<span class='ocrx_word' id='word_1_549' title='bbox 347 2350 361 2363; x_wconf 58'>*</span>
|
||||||
<span class='ocrx_word' id='word_1_550' title='bbox 373 2343 483 2377; x_wconf 96'>Simple,</span>
|
<span class='ocrx_word' id='word_1_550' title='bbox 373 2343 483 2377; x_wconf 96'>Simple,</span>
|
||||||
<span class='ocrx_word' id='word_1_551' title='bbox 495 2352 559 2377; x_wconf 96'>easy</span>
|
<span class='ocrx_word' id='word_1_551' title='bbox 495 2352 559 2377; x_wconf 96'>easy</span>
|
||||||
@@ -806,8 +806,8 @@
|
|||||||
<span class='ocrx_word' id='word_1_559' title='bbox 1326 2345 1424 2378; x_wconf 97'>clearly</span>
|
<span class='ocrx_word' id='word_1_559' title='bbox 1326 2345 1424 2378; x_wconf 97'>clearly</span>
|
||||||
<span class='ocrx_word' id='word_1_560' title='bbox 1434 2345 1528 2378; x_wconf 96'>guides</span>
|
<span class='ocrx_word' id='word_1_560' title='bbox 1434 2345 1528 2378; x_wconf 96'>guides</span>
|
||||||
<span class='ocrx_word' id='word_1_561' title='bbox 1539 2353 1594 2378; x_wconf 97'>you</span>
|
<span class='ocrx_word' id='word_1_561' title='bbox 1539 2353 1594 2378; x_wconf 97'>you</span>
|
||||||
<span class='ocrx_word' id='word_1_562' title='bbox 1604 2345 1705 2378; x_wconf 96'>through</span>
|
<span class='ocrx_word' id='word_1_562' title='bbox 1604 2345 1724 2378; x_wconf 96'>through</span>
|
||||||
<span class='ocrx_word' id='word_1_563' title='bbox 1706 2344 1770 2371; x_wconf 96'>all</span>
|
<span class='ocrx_word' id='word_1_563' title='bbox 1735 2344 1770 2371; x_wconf 96'>all</span>
|
||||||
<span class='ocrx_word' id='word_1_564' title='bbox 1781 2344 1947 2377; x_wconf 96'>operations.</span>
|
<span class='ocrx_word' id='word_1_564' title='bbox 1781 2344 1947 2377; x_wconf 96'>operations.</span>
|
||||||
<span class='ocrx_word' id='word_1_565' title='bbox 1961 2344 1989 2371; x_wconf 96'>If</span>
|
<span class='ocrx_word' id='word_1_565' title='bbox 1961 2344 1989 2371; x_wconf 96'>If</span>
|
||||||
<span class='ocrx_word' id='word_1_566' title='bbox 1997 2344 2112 2376; x_wconf 96'>needed,</span>
|
<span class='ocrx_word' id='word_1_566' title='bbox 1997 2344 2112 2376; x_wconf 96'>needed,</span>
|
||||||
@@ -818,8 +818,8 @@
|
|||||||
<div class='ocr_carea' id='block_1_22' title="bbox 373 2381 1083 2415">
|
<div class='ocr_carea' id='block_1_22' title="bbox 373 2381 1083 2415">
|
||||||
<p class='ocr_par' id='par_1_30' lang='eng' title="bbox 373 2381 1083 2415">
|
<p class='ocr_par' id='par_1_30' lang='eng' title="bbox 373 2381 1083 2415">
|
||||||
<span class='ocr_line' id='line_1_66' title="bbox 373 2381 1083 2415; baseline 0.003 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_line' id='line_1_66' title="bbox 373 2381 1083 2415; baseline 0.003 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_568' title='bbox 373 2381 448 2407; x_wconf 96'>HELP</span>
|
<span class='ocrx_word' id='word_1_568' title='bbox 373 2381 472 2407; x_wconf 96'>HELP</span>
|
||||||
<span class='ocrx_word' id='word_1_569' title='bbox 450 2381 583 2408; x_wconf 96'>button</span>
|
<span class='ocrx_word' id='word_1_569' title='bbox 483 2381 583 2408; x_wconf 96'>button</span>
|
||||||
<span class='ocrx_word' id='word_1_570' title='bbox 594 2381 711 2415; x_wconf 96'>displays</span>
|
<span class='ocrx_word' id='word_1_570' title='bbox 594 2381 711 2415; x_wconf 96'>displays</span>
|
||||||
<span class='ocrx_word' id='word_1_571' title='bbox 722 2382 875 2409; x_wconf 96'>additional</span>
|
<span class='ocrx_word' id='word_1_571' title='bbox 722 2382 875 2409; x_wconf 96'>additional</span>
|
||||||
<span class='ocrx_word' id='word_1_572' title='bbox 886 2382 1083 2415; x_wconf 96'>explanations.</span>
|
<span class='ocrx_word' id='word_1_572' title='bbox 886 2382 1083 2415; x_wconf 96'>explanations.</span>
|
||||||
@@ -828,7 +828,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_23' title="bbox 347 2427 2145 2507">
|
<div class='ocr_carea' id='block_1_23' title="bbox 347 2427 2145 2507">
|
||||||
<p class='ocr_par' id='par_1_31' lang='eng' title="bbox 347 2427 2145 2507">
|
<p class='ocr_par' id='par_1_31' lang='eng' title="bbox 347 2427 2145 2507">
|
||||||
<span class='ocr_line' id='line_1_67' title="bbox 347 2427 1468 2461; baseline 0.002 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_header' id='line_1_67' title="bbox 347 2427 1468 2461; baseline 0.002 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_573' title='bbox 347 2432 361 2446; x_wconf 70'>*</span>
|
<span class='ocrx_word' id='word_1_573' title='bbox 347 2432 361 2446; x_wconf 70'>*</span>
|
||||||
<span class='ocrx_word' id='word_1_574' title='bbox 373 2427 612 2454; x_wconf 91'>Non-destructive</span>
|
<span class='ocrx_word' id='word_1_574' title='bbox 373 2427 612 2454; x_wconf 91'>Non-destructive</span>
|
||||||
<span class='ocrx_word' id='word_1_575' title='bbox 622 2427 914 2461; x_wconf 89'>recording—existing</span>
|
<span class='ocrx_word' id='word_1_575' title='bbox 622 2427 914 2461; x_wconf 89'>recording—existing</span>
|
||||||
@@ -839,14 +839,14 @@
|
|||||||
<span class='ocrx_word' id='word_1_580' title='bbox 1231 2428 1309 2455; x_wconf 96'>while</span>
|
<span class='ocrx_word' id='word_1_580' title='bbox 1231 2428 1309 2455; x_wconf 96'>while</span>
|
||||||
<span class='ocrx_word' id='word_1_581' title='bbox 1319 2428 1468 2461; x_wconf 92'>recording.</span>
|
<span class='ocrx_word' id='word_1_581' title='bbox 1319 2428 1468 2461; x_wconf 92'>recording.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_68' title="bbox 347 2473 2145 2507; baseline 0.001 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
<span class='ocr_header' id='line_1_68' title="bbox 347 2473 2145 2507; baseline 0.001 -8; x_size 35; x_descenders 7; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_582' title='bbox 347 2478 361 2492; x_wconf 70'>¢</span>
|
<span class='ocrx_word' id='word_1_582' title='bbox 347 2478 361 2492; x_wconf 70'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_583' title='bbox 372 2473 433 2500; x_wconf 93'>Two</span>
|
<span class='ocrx_word' id='word_1_583' title='bbox 372 2473 433 2500; x_wconf 93'>Two</span>
|
||||||
<span class='ocrx_word' id='word_1_584' title='bbox 444 2473 689 2500; x_wconf 90'>FOOTSWITCH</span>
|
<span class='ocrx_word' id='word_1_584' title='bbox 444 2473 689 2500; x_wconf 90'>FOOTSWITCH</span>
|
||||||
<span class='ocrx_word' id='word_1_585' title='bbox 701 2474 818 2500; x_wconf 95'>INPUTS</span>
|
<span class='ocrx_word' id='word_1_585' title='bbox 701 2474 837 2501; x_wconf 95'>INPUTS</span>
|
||||||
<span class='ocrx_word' id='word_1_586' title='bbox 819 2474 894 2501; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_586' title='bbox 848 2481 910 2507; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_587' title='bbox 893 2473 939 2507; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_587' title='bbox 921 2473 955 2500; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_588' title='bbox 941 2473 1091 2507; x_wconf 96'>assigned</span>
|
<span class='ocrx_word' id='word_1_588' title='bbox 966 2473 1091 2507; x_wconf 96'>assigned</span>
|
||||||
<span class='ocrx_word' id='word_1_589' title='bbox 1101 2478 1130 2501; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_589' title='bbox 1101 2478 1130 2501; x_wconf 96'>to</span>
|
||||||
<span class='ocrx_word' id='word_1_590' title='bbox 1141 2474 1271 2507; x_wconf 96'>remotely</span>
|
<span class='ocrx_word' id='word_1_590' title='bbox 1141 2474 1271 2507; x_wconf 96'>remotely</span>
|
||||||
<span class='ocrx_word' id='word_1_591' title='bbox 1281 2473 1387 2501; x_wconf 96'>control</span>
|
<span class='ocrx_word' id='word_1_591' title='bbox 1281 2473 1387 2501; x_wconf 96'>control</span>
|
||||||
@@ -873,12 +873,12 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_25' title="bbox 347 2556 1768 2590">
|
<div class='ocr_carea' id='block_1_25' title="bbox 347 2556 1768 2590">
|
||||||
<p class='ocr_par' id='par_1_33' lang='eng' title="bbox 347 2556 1768 2590">
|
<p class='ocr_par' id='par_1_33' lang='eng' title="bbox 347 2556 1768 2590">
|
||||||
<span class='ocr_line' id='line_1_70' title="bbox 347 2556 1768 2590; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
<span class='ocr_header' id='line_1_70' title="bbox 347 2556 1768 2590; baseline 0.001 -8; x_size 34; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_604' title='bbox 347 2561 361 2575; x_wconf 85'>¢</span>
|
<span class='ocrx_word' id='word_1_604' title='bbox 347 2561 361 2575; x_wconf 86'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_605' title='bbox 372 2556 433 2583; x_wconf 84'>Iwo</span>
|
<span class='ocrx_word' id='word_1_605' title='bbox 372 2556 433 2583; x_wconf 85'>Iwo</span>
|
||||||
<span class='ocrx_word' id='word_1_606' title='bbox 443 2556 612 2583; x_wconf 96'>TRIGGER</span>
|
<span class='ocrx_word' id='word_1_606' title='bbox 443 2556 612 2583; x_wconf 96'>TRIGGER</span>
|
||||||
<span class='ocrx_word' id='word_1_607' title='bbox 623 2556 778 2584; x_wconf 95'>OUTPUTS</span>
|
<span class='ocrx_word' id='word_1_607' title='bbox 623 2556 797 2584; x_wconf 96'>OUTPUTS</span>
|
||||||
<span class='ocrx_word' id='word_1_608' title='bbox 780 2557 871 2590; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_608' title='bbox 808 2565 871 2590; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_609' title='bbox 881 2557 915 2584; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_609' title='bbox 881 2557 915 2584; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_610' title='bbox 925 2557 1119 2590; x_wconf 96'>programmed</span>
|
<span class='ocrx_word' id='word_1_610' title='bbox 925 2557 1119 2590; x_wconf 96'>programmed</span>
|
||||||
<span class='ocrx_word' id='word_1_611' title='bbox 1129 2561 1158 2584; x_wconf 96'>to</span>
|
<span class='ocrx_word' id='word_1_611' title='bbox 1129 2561 1158 2584; x_wconf 96'>to</span>
|
||||||
@@ -887,7 +887,7 @@
|
|||||||
<span class='ocrx_word' id='word_1_614' title='bbox 1381 2561 1409 2584; x_wconf 96'>at</span>
|
<span class='ocrx_word' id='word_1_614' title='bbox 1381 2561 1409 2584; x_wconf 96'>at</span>
|
||||||
<span class='ocrx_word' id='word_1_615' title='bbox 1419 2565 1472 2590; x_wconf 96'>any</span>
|
<span class='ocrx_word' id='word_1_615' title='bbox 1419 2565 1472 2590; x_wconf 96'>any</span>
|
||||||
<span class='ocrx_word' id='word_1_616' title='bbox 1483 2557 1598 2584; x_wconf 96'>selected</span>
|
<span class='ocrx_word' id='word_1_616' title='bbox 1483 2557 1598 2584; x_wconf 96'>selected</span>
|
||||||
<span class='ocrx_word' id='word_1_617' title='bbox 1608 2561 1673 2584; x_wconf 96'>note</span>
|
<span class='ocrx_word' id='word_1_617' title='bbox 1608 2561 1673 2584; x_wconf 97'>note</span>
|
||||||
<span class='ocrx_word' id='word_1_618' title='bbox 1683 2556 1768 2583; x_wconf 96'>value.</span>
|
<span class='ocrx_word' id='word_1_618' title='bbox 1683 2556 1768 2583; x_wconf 96'>value.</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
@@ -904,22 +904,22 @@
|
|||||||
<span class='ocrx_word' id='word_1_625' title='bbox 875 2610 907 2629; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_625' title='bbox 875 2610 907 2629; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_626' title='bbox 918 2602 989 2629; x_wconf 96'>Linn</span>
|
<span class='ocrx_word' id='word_1_626' title='bbox 918 2602 989 2629; x_wconf 96'>Linn</span>
|
||||||
<span class='ocrx_word' id='word_1_627' title='bbox 1000 2603 1069 2629; x_wconf 95'>9000</span>
|
<span class='ocrx_word' id='word_1_627' title='bbox 1000 2603 1069 2629; x_wconf 95'>9000</span>
|
||||||
<span class='ocrx_word' id='word_1_628' title='bbox 1080 2610 1129 2635; x_wconf 96'>sync</span>
|
<span class='ocrx_word' id='word_1_628' title='bbox 1080 2610 1145 2635; x_wconf 96'>sync</span>
|
||||||
<span class='ocrx_word' id='word_1_629' title='bbox 1131 2607 1226 2630; x_wconf 96'>tone.</span>
|
<span class='ocrx_word' id='word_1_629' title='bbox 1155 2607 1226 2630; x_wconf 96'>tone.</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_27' title="bbox 347 2648 2100 2727">
|
<div class='ocr_carea' id='block_1_27' title="bbox 347 2648 2100 2727">
|
||||||
<p class='ocr_par' id='par_1_35' lang='eng' title="bbox 347 2648 2100 2727">
|
<p class='ocr_par' id='par_1_35' lang='eng' title="bbox 347 2648 2100 2727">
|
||||||
<span class='ocr_line' id='line_1_72' title="bbox 347 2648 1664 2682; baseline 0 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_header' id='line_1_72' title="bbox 347 2648 1664 2682; baseline 0.001 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_630' title='bbox 347 2654 360 2667; x_wconf 51'>©</span>
|
<span class='ocrx_word' id='word_1_630' title='bbox 347 2654 360 2667; x_wconf 45'>©</span>
|
||||||
<span class='ocrx_word' id='word_1_631' title='bbox 372 2648 483 2675; x_wconf 96'>Utilizes</span>
|
<span class='ocrx_word' id='word_1_631' title='bbox 372 2648 483 2675; x_wconf 95'>Utilizes</span>
|
||||||
<span class='ocrx_word' id='word_1_632' title='bbox 493 2648 564 2677; x_wconf 96'>ultra</span>
|
<span class='ocrx_word' id='word_1_632' title='bbox 493 2648 564 2680; x_wconf 96'>ultra</span>
|
||||||
<span class='ocrx_word' id='word_1_633' title='bbox 573 2648 744 2682; x_wconf 96'>high-speed,</span>
|
<span class='ocrx_word' id='word_1_633' title='bbox 573 2648 744 2682; x_wconf 96'>high-speed,</span>
|
||||||
<span class='ocrx_word' id='word_1_634' title='bbox 757 2649 772 2676; x_wconf 95'>8</span>
|
<span class='ocrx_word' id='word_1_634' title='bbox 757 2649 772 2676; x_wconf 95'>8</span>
|
||||||
<span class='ocrx_word' id='word_1_635' title='bbox 783 2649 862 2675; x_wconf 94'>MHz</span>
|
<span class='ocrx_word' id='word_1_635' title='bbox 783 2649 862 2675; x_wconf 94'>MHz</span>
|
||||||
<span class='ocrx_word' id='word_1_636' title='bbox 873 2649 954 2676; x_wconf 95'>80186</span>
|
<span class='ocrx_word' id='word_1_636' title='bbox 873 2649 954 2676; x_wconf 96'>80186</span>
|
||||||
<span class='ocrx_word' id='word_1_637' title='bbox 965 2649 994 2676; x_wconf 95'>16</span>
|
<span class='ocrx_word' id='word_1_637' title='bbox 965 2649 994 2676; x_wconf 96'>16</span>
|
||||||
<span class='ocrx_word' id='word_1_638' title='bbox 1004 2648 1043 2676; x_wconf 96'>bit</span>
|
<span class='ocrx_word' id='word_1_638' title='bbox 1004 2648 1043 2676; x_wconf 96'>bit</span>
|
||||||
<span class='ocrx_word' id='word_1_639' title='bbox 1054 2653 1197 2682; x_wconf 96'>computer</span>
|
<span class='ocrx_word' id='word_1_639' title='bbox 1054 2653 1197 2682; x_wconf 96'>computer</span>
|
||||||
<span class='ocrx_word' id='word_1_640' title='bbox 1208 2649 1350 2682; x_wconf 96'>internally</span>
|
<span class='ocrx_word' id='word_1_640' title='bbox 1208 2649 1350 2682; x_wconf 96'>internally</span>
|
||||||
@@ -927,17 +927,17 @@
|
|||||||
<span class='ocrx_word' id='word_1_642' title='bbox 1414 2649 1502 2676; x_wconf 96'>FAST</span>
|
<span class='ocrx_word' id='word_1_642' title='bbox 1414 2649 1502 2676; x_wconf 96'>FAST</span>
|
||||||
<span class='ocrx_word' id='word_1_643' title='bbox 1512 2648 1664 2682; x_wconf 96'>operation.</span>
|
<span class='ocrx_word' id='word_1_643' title='bbox 1512 2648 1664 2682; x_wconf 96'>operation.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_73' title="bbox 347 2694 2100 2727; baseline 0.001 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_header' id='line_1_73' title="bbox 347 2694 2100 2727; baseline 0.001 -7; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_644' title='bbox 347 2699 361 2713; x_wconf 31'>*</span>
|
<span class='ocrx_word' id='word_1_644' title='bbox 347 2699 361 2713; x_wconf 31'>*</span>
|
||||||
<span class='ocrx_word' id='word_1_645' title='bbox 372 2694 476 2720; x_wconf 96'>TEMPO</span>
|
<span class='ocrx_word' id='word_1_645' title='bbox 372 2694 504 2721; x_wconf 96'>TEMPO</span>
|
||||||
<span class='ocrx_word' id='word_1_646' title='bbox 478 2694 562 2721; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_646' title='bbox 515 2702 578 2727; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_647' title='bbox 561 2694 606 2727; x_wconf 95'>be</span>
|
<span class='ocrx_word' id='word_1_647' title='bbox 589 2694 623 2721; x_wconf 95'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_648' title='bbox 608 2694 764 2727; x_wconf 95'>specified</span>
|
<span class='ocrx_word' id='word_1_648' title='bbox 633 2694 764 2727; x_wconf 95'>specified</span>
|
||||||
<span class='ocrx_word' id='word_1_649' title='bbox 774 2694 802 2721; x_wconf 93'>in</span>
|
<span class='ocrx_word' id='word_1_649' title='bbox 774 2694 802 2721; x_wconf 93'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_650' title='bbox 814 2695 1172 2722; x_wconf 91'>BEATS-PER-MINUTE</span>
|
<span class='ocrx_word' id='word_1_650' title='bbox 814 2695 1172 2722; x_wconf 91'>BEATS-PER-MINUTE</span>
|
||||||
<span class='ocrx_word' id='word_1_651' title='bbox 1183 2703 1215 2722; x_wconf 93'>or</span>
|
<span class='ocrx_word' id='word_1_651' title='bbox 1183 2703 1215 2722; x_wconf 93'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_652' title='bbox 1225 2695 1547 2722; x_wconf 91'>FRAMES-PER-BEAT</span>
|
<span class='ocrx_word' id='word_1_652' title='bbox 1225 2695 1567 2722; x_wconf 91'>FRAMES-PER-BEAT</span>
|
||||||
<span class='ocrx_word' id='word_1_653' title='bbox 1543 2695 1605 2721; x_wconf 96'>at</span>
|
<span class='ocrx_word' id='word_1_653' title='bbox 1577 2698 1605 2721; x_wconf 96'>at</span>
|
||||||
<span class='ocrx_word' id='word_1_654' title='bbox 1616 2695 1659 2726; x_wconf 96'>24,</span>
|
<span class='ocrx_word' id='word_1_654' title='bbox 1616 2695 1659 2726; x_wconf 96'>24,</span>
|
||||||
<span class='ocrx_word' id='word_1_655' title='bbox 1672 2695 1716 2726; x_wconf 96'>25,</span>
|
<span class='ocrx_word' id='word_1_655' title='bbox 1672 2695 1716 2726; x_wconf 96'>25,</span>
|
||||||
<span class='ocrx_word' id='word_1_656' title='bbox 1728 2702 1760 2721; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_656' title='bbox 1728 2702 1760 2721; x_wconf 96'>or</span>
|
||||||
@@ -959,19 +959,19 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_29' title="bbox 347 2777 2174 2811">
|
<div class='ocr_carea' id='block_1_29' title="bbox 347 2777 2174 2811">
|
||||||
<p class='ocr_par' id='par_1_37' lang='eng' title="bbox 347 2777 2174 2811">
|
<p class='ocr_par' id='par_1_37' lang='eng' title="bbox 347 2777 2174 2811">
|
||||||
<span class='ocr_line' id='line_1_75' title="bbox 347 2777 2174 2811; baseline 0.001 -8; x_size 33; x_descenders 5; x_ascenders 9">
|
<span class='ocr_header' id='line_1_75' title="bbox 347 2777 2174 2811; baseline 0.001 -8; x_size 33; x_descenders 5; x_ascenders 9">
|
||||||
<span class='ocrx_word' id='word_1_664' title='bbox 347 2782 360 2796; x_wconf 79'>¢</span>
|
<span class='ocrx_word' id='word_1_664' title='bbox 347 2782 360 2796; x_wconf 81'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_665' title='bbox 372 2777 476 2803; x_wconf 94'>TEMPO</span>
|
<span class='ocrx_word' id='word_1_665' title='bbox 372 2777 504 2804; x_wconf 94'>TEMPO</span>
|
||||||
<span class='ocrx_word' id='word_1_666' title='bbox 478 2777 562 2804; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_666' title='bbox 515 2785 578 2810; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_667' title='bbox 561 2777 622 2810; x_wconf 95'>be</span>
|
<span class='ocrx_word' id='word_1_667' title='bbox 588 2777 622 2804; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_668' title='bbox 633 2778 741 2804; x_wconf 95'>entered</span>
|
<span class='ocrx_word' id='word_1_668' title='bbox 633 2778 741 2804; x_wconf 95'>entered</span>
|
||||||
<span class='ocrx_word' id='word_1_669' title='bbox 751 2777 934 2811; x_wconf 96'>numerically,</span>
|
<span class='ocrx_word' id='word_1_669' title='bbox 751 2777 934 2811; x_wconf 96'>numerically,</span>
|
||||||
<span class='ocrx_word' id='word_1_670' title='bbox 946 2777 1101 2811; x_wconf 96'>adjustable</span>
|
<span class='ocrx_word' id='word_1_670' title='bbox 946 2777 1101 2811; x_wconf 95'>adjustable</span>
|
||||||
<span class='ocrx_word' id='word_1_671' title='bbox 1111 2777 1139 2804; x_wconf 96'>in</span>
|
<span class='ocrx_word' id='word_1_671' title='bbox 1111 2777 1139 2804; x_wconf 95'>in</span>
|
||||||
<span class='ocrx_word' id='word_1_672' title='bbox 1149 2778 1239 2805; x_wconf 96'>tenths</span>
|
<span class='ocrx_word' id='word_1_672' title='bbox 1149 2778 1239 2805; x_wconf 96'>tenths</span>
|
||||||
<span class='ocrx_word' id='word_1_673' title='bbox 1250 2778 1282 2805; x_wconf 96'>of</span>
|
<span class='ocrx_word' id='word_1_673' title='bbox 1250 2778 1282 2805; x_wconf 96'>of</span>
|
||||||
<span class='ocrx_word' id='word_1_674' title='bbox 1290 2786 1307 2805; x_wconf 93'>a</span>
|
<span class='ocrx_word' id='word_1_674' title='bbox 1290 2786 1307 2805; x_wconf 91'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_675' title='bbox 1317 2777 1567 2805; x_wconf 92'>Beat-Per-Minute</span>
|
<span class='ocrx_word' id='word_1_675' title='bbox 1317 2777 1567 2805; x_wconf 91'>Beat-Per-Minute</span>
|
||||||
<span class='ocrx_word' id='word_1_676' title='bbox 1577 2777 1748 2809; x_wconf 96'>increments,</span>
|
<span class='ocrx_word' id='word_1_676' title='bbox 1577 2777 1748 2809; x_wconf 96'>increments,</span>
|
||||||
<span class='ocrx_word' id='word_1_677' title='bbox 1760 2785 1792 2804; x_wconf 96'>or</span>
|
<span class='ocrx_word' id='word_1_677' title='bbox 1760 2785 1792 2804; x_wconf 96'>or</span>
|
||||||
<span class='ocrx_word' id='word_1_678' title='bbox 1803 2777 1839 2810; x_wconf 96'>by</span>
|
<span class='ocrx_word' id='word_1_678' title='bbox 1803 2777 1839 2810; x_wconf 96'>by</span>
|
||||||
@@ -987,36 +987,36 @@
|
|||||||
<span class='ocrx_word' id='word_1_682' title='bbox 372 2822 410 2841; x_wconf 96'>on</span>
|
<span class='ocrx_word' id='word_1_682' title='bbox 372 2822 410 2841; x_wconf 96'>on</span>
|
||||||
<span class='ocrx_word' id='word_1_683' title='bbox 420 2815 466 2842; x_wconf 95'>the</span>
|
<span class='ocrx_word' id='word_1_683' title='bbox 420 2815 466 2842; x_wconf 95'>the</span>
|
||||||
<span class='ocrx_word' id='word_1_684' title='bbox 476 2815 545 2841; x_wconf 95'>TAP</span>
|
<span class='ocrx_word' id='word_1_684' title='bbox 476 2815 545 2841; x_wconf 95'>TAP</span>
|
||||||
<span class='ocrx_word' id='word_1_685' title='bbox 556 2816 660 2842; x_wconf 95'>TEMPO</span>
|
<span class='ocrx_word' id='word_1_685' title='bbox 556 2815 689 2842; x_wconf 95'>TEMPO</span>
|
||||||
<span class='ocrx_word' id='word_1_686' title='bbox 662 2815 808 2842; x_wconf 96'>button.</span>
|
<span class='ocrx_word' id='word_1_686' title='bbox 699 2815 808 2842; x_wconf 96'>button.</span>
|
||||||
</span>
|
</span>
|
||||||
</p>
|
</p>
|
||||||
</div>
|
</div>
|
||||||
<div class='ocr_carea' id='block_1_31' title="bbox 347 2861 1792 2940">
|
<div class='ocr_carea' id='block_1_31' title="bbox 347 2861 1792 2940">
|
||||||
<p class='ocr_par' id='par_1_39' lang='eng' title="bbox 347 2861 1792 2940">
|
<p class='ocr_par' id='par_1_39' lang='eng' title="bbox 347 2861 1792 2940">
|
||||||
<span class='ocr_line' id='line_1_77' title="bbox 347 2861 1792 2895; baseline 0.001 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
<span class='ocr_header' id='line_1_77' title="bbox 347 2861 1792 2895; baseline 0.001 -8; x_size 33; x_descenders 6; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_687' title='bbox 347 2866 360 2880; x_wconf 43'>¢</span>
|
<span class='ocrx_word' id='word_1_687' title='bbox 347 2866 360 2880; x_wconf 43'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_688' title='bbox 372 2861 504 2887; x_wconf 96'>TEMPO</span>
|
<span class='ocrx_word' id='word_1_688' title='bbox 372 2861 504 2887; x_wconf 96'>TEMPO</span>
|
||||||
<span class='ocrx_word' id='word_1_689' title='bbox 515 2861 677 2888; x_wconf 96'>CHANGES</span>
|
<span class='ocrx_word' id='word_1_689' title='bbox 515 2861 696 2888; x_wconf 96'>CHANGES</span>
|
||||||
<span class='ocrx_word' id='word_1_690' title='bbox 679 2861 771 2894; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_690' title='bbox 707 2869 771 2894; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_691' title='bbox 781 2861 815 2888; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_691' title='bbox 781 2861 815 2888; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_692' title='bbox 825 2869 1000 2894; x_wconf 96'>programmed</span>
|
<span class='ocrx_word' id='word_1_692' title='bbox 825 2861 1019 2894; x_wconf 96'>programmed</span>
|
||||||
<span class='ocrx_word' id='word_1_693' title='bbox 1001 2861 1087 2888; x_wconf 96'>into</span>
|
<span class='ocrx_word' id='word_1_693' title='bbox 1030 2861 1087 2888; x_wconf 96'>into</span>
|
||||||
<span class='ocrx_word' id='word_1_694' title='bbox 1099 2869 1115 2888; x_wconf 95'>a</span>
|
<span class='ocrx_word' id='word_1_694' title='bbox 1099 2869 1115 2888; x_wconf 95'>a</span>
|
||||||
<span class='ocrx_word' id='word_1_695' title='bbox 1126 2870 1268 2895; x_wconf 96'>sequence,</span>
|
<span class='ocrx_word' id='word_1_695' title='bbox 1126 2870 1268 2895; x_wconf 96'>sequence,</span>
|
||||||
<span class='ocrx_word' id='word_1_696' title='bbox 1280 2861 1344 2888; x_wconf 96'>with</span>
|
<span class='ocrx_word' id='word_1_696' title='bbox 1280 2861 1344 2888; x_wconf 96'>with</span>
|
||||||
<span class='ocrx_word' id='word_1_697' title='bbox 1356 2866 1448 2888; x_wconf 95'>smooth</span>
|
<span class='ocrx_word' id='word_1_697' title='bbox 1356 2862 1467 2888; x_wconf 95'>smooth</span>
|
||||||
<span class='ocrx_word' id='word_1_698' title='bbox 1450 2861 1635 2888; x_wconf 96'>transitions</span>
|
<span class='ocrx_word' id='word_1_698' title='bbox 1478 2861 1635 2888; x_wconf 96'>transitions</span>
|
||||||
<span class='ocrx_word' id='word_1_699' title='bbox 1646 2861 1670 2887; x_wconf 96'>if</span>
|
<span class='ocrx_word' id='word_1_699' title='bbox 1646 2861 1670 2887; x_wconf 96'>if</span>
|
||||||
<span class='ocrx_word' id='word_1_700' title='bbox 1679 2861 1792 2888; x_wconf 87'>desired.</span>
|
<span class='ocrx_word' id='word_1_700' title='bbox 1679 2861 1792 2888; x_wconf 87'>desired.</span>
|
||||||
</span>
|
</span>
|
||||||
<span class='ocr_line' id='line_1_78' title="bbox 347 2906 1507 2940; baseline 0.002 -8; x_size 33; x_descenders 7; x_ascenders 8">
|
<span class='ocr_header' id='line_1_78' title="bbox 347 2906 1507 2940; baseline 0.002 -8; x_size 33; x_descenders 7; x_ascenders 8">
|
||||||
<span class='ocrx_word' id='word_1_701' title='bbox 347 2911 360 2925; x_wconf 69'>¢</span>
|
<span class='ocrx_word' id='word_1_701' title='bbox 347 2911 360 2925; x_wconf 69'>¢</span>
|
||||||
<span class='ocrx_word' id='word_1_702' title='bbox 371 2906 434 2938; x_wconf 96'>Any</span>
|
<span class='ocrx_word' id='word_1_702' title='bbox 371 2906 434 2938; x_wconf 96'>Any</span>
|
||||||
<span class='ocrx_word' id='word_1_703' title='bbox 444 2906 539 2932; x_wconf 96'>TIME</span>
|
<span class='ocrx_word' id='word_1_703' title='bbox 444 2906 539 2932; x_wconf 96'>TIME</span>
|
||||||
<span class='ocrx_word' id='word_1_704' title='bbox 550 2906 739 2933; x_wconf 96'>SIGNATURE</span>
|
<span class='ocrx_word' id='word_1_704' title='bbox 550 2906 763 2933; x_wconf 96'>SIGNATURE</span>
|
||||||
<span class='ocrx_word' id='word_1_705' title='bbox 740 2907 820 2933; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_705' title='bbox 773 2915 836 2939; x_wconf 96'>may</span>
|
||||||
<span class='ocrx_word' id='word_1_706' title='bbox 819 2907 880 2939; x_wconf 96'>be</span>
|
<span class='ocrx_word' id='word_1_706' title='bbox 846 2907 880 2933; x_wconf 96'>be</span>
|
||||||
<span class='ocrx_word' id='word_1_707' title='bbox 891 2907 968 2938; x_wconf 96'>used,</span>
|
<span class='ocrx_word' id='word_1_707' title='bbox 891 2907 968 2938; x_wconf 96'>used,</span>
|
||||||
<span class='ocrx_word' id='word_1_708' title='bbox 980 2907 1036 2934; x_wconf 96'>and</span>
|
<span class='ocrx_word' id='word_1_708' title='bbox 980 2907 1036 2934; x_wconf 96'>and</span>
|
||||||
<span class='ocrx_word' id='word_1_709' title='bbox 1046 2915 1109 2940; x_wconf 96'>may</span>
|
<span class='ocrx_word' id='word_1_709' title='bbox 1046 2915 1109 2940; x_wconf 96'>may</span>
|
||||||
|
|||||||
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
+1
-1
@@ -1 +1 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
Tesseract Open Source OCR Engine v4.1.1 with Leptonica
|
||||||
|
|||||||
BIN
Binary file not shown.
-1
@@ -1 +0,0 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
|
||||||
-13
@@ -1,13 +0,0 @@
|
|||||||
Portez ce vieux whisky au juge
|
|
||||||
blond qui fume sur son Ile
|
|
||||||
interieure, a cöte de l'alcöve
|
|
||||||
ovoide, oU les büches se
|
|
||||||
consument dans l'ätre, ce qui
|
|
||||||
lui permet de penser & la
|
|
||||||
caenogenese de |'etre dont il
|
|
||||||
est question dans la cause
|
|
||||||
ambigu& entendue a MoY, dans
|
|
||||||
un capharnaüm qui, pense-t-il,
|
|
||||||
diminue ca et la la qualite de son
|
|
||||||
ceuvre.
|
|
||||||
|
|
||||||
-54
@@ -1,54 +0,0 @@
|
|||||||
<?xml version="1.0" encoding="UTF-8"?>
|
|
||||||
<!DOCTYPE html PUBLIC "-//W3C//DTD XHTML 1.0 Transitional//EN"
|
|
||||||
"http://www.w3.org/TR/xhtml1/DTD/xhtml1-transitional.dtd">
|
|
||||||
<html xmlns="http://www.w3.org/1999/xhtml" xml:lang="en" lang="en">
|
|
||||||
<head>
|
|
||||||
<title></title>
|
|
||||||
<meta http-equiv="Content-Type" content="text/html;charset=utf-8" />
|
|
||||||
<meta name='ocr-system' content='tesseract 4.0.0' />
|
|
||||||
<meta name='ocr-capabilities' content='ocr_page ocr_carea ocr_par ocr_line ocrx_word ocrp_wconf'/>
|
|
||||||
</head>
|
|
||||||
<body>
|
|
||||||
<div class='ocr_page' id='page_1' title='image "/var/folders/2s/7t022mgj0h5cprbq0dtb1ksm0000gn/T/com.github.ocrmypdf.iw9uvooj/000001_ocr.png"; bbox 0 0 900 900; ppageno 0'>
|
|
||||||
<div class='ocr_carea' id='block_1_1' title="bbox 97 87 502 204">
|
|
||||||
<p class='ocr_par' id='par_1_1' lang='eng' title="bbox 97 87 502 204">
|
|
||||||
<span class='ocr_line' id='line_1_1' title="bbox 97 87 399 122; baseline 0 0; x_size 40.714287; x_descenders 5.7142859; x_ascenders 15">
|
|
||||||
<span class='ocrx_word' id='word_1_1' title='bbox 97 87 285 122; x_wconf 95'>Multicolor</span>
|
|
||||||
<span class='ocrx_word' id='word_1_2' title='bbox 302 87 399 122; x_wconf 96'>Black</span>
|
|
||||||
</span>
|
|
||||||
<span class='ocr_line' id='line_1_2' title="bbox 114 160 502 204; baseline 0 -9; x_size 44; x_descenders 9; x_ascenders 15">
|
|
||||||
<span class='ocrx_word' id='word_1_3' title='bbox 114 163 191 195; x_wconf 96'>Pure</span>
|
|
||||||
<span class='ocrx_word' id='word_1_4' title='bbox 210 160 308 195; x_wconf 93'>Black</span>
|
|
||||||
<span class='ocrx_word' id='word_1_5' title='bbox 325 160 363 204; x_wconf 93'>(K</span>
|
|
||||||
<span class='ocrx_word' id='word_1_6' title='bbox 378 174 400 183; x_wconf 96'>=</span>
|
|
||||||
<span class='ocrx_word' id='word_1_7' title='bbox 421 160 502 204; x_wconf 70'>100}</span>
|
|
||||||
</span>
|
|
||||||
</p>
|
|
||||||
</div>
|
|
||||||
<div class='ocr_carea' id='block_1_2' title="bbox 121 282 383 326">
|
|
||||||
<p class='ocr_par' id='par_1_2' lang='eng' title="bbox 121 282 383 326">
|
|
||||||
<span class='ocr_line' id='line_1_3' title="bbox 121 282 383 326; baseline 0 -11; x_size 44; x_descenders 11; x_ascenders 13">
|
|
||||||
<span class='ocrx_word' id='word_1_8' title='bbox 121 283 199 315; x_wconf 96'>Pure</span>
|
|
||||||
<span class='ocrx_word' id='word_1_9' title='bbox 216 282 383 326; x_wconf 96'>Magenta</span>
|
|
||||||
</span>
|
|
||||||
</p>
|
|
||||||
</div>
|
|
||||||
<div class='ocr_carea' id='block_1_3' title="bbox 140 375 329 418">
|
|
||||||
<p class='ocr_par' id='par_1_3' lang='eng' title="bbox 140 375 329 418">
|
|
||||||
<span class='ocr_line' id='line_1_4' title="bbox 140 375 329 418; baseline 0 -11; x_size 43; x_descenders 11; x_ascenders 12">
|
|
||||||
<span class='ocrx_word' id='word_1_10' title='bbox 140 375 218 407; x_wconf 96'>Pure</span>
|
|
||||||
<span class='ocrx_word' id='word_1_11' title='bbox 234 375 329 418; x_wconf 96'>Cyan</span>
|
|
||||||
</span>
|
|
||||||
</p>
|
|
||||||
</div>
|
|
||||||
<div class='ocr_carea' id='block_1_4' title="bbox 207 570 422 605">
|
|
||||||
<p class='ocr_par' id='par_1_4' lang='eng' title="bbox 207 570 422 605">
|
|
||||||
<span class='ocr_line' id='line_1_5' title="bbox 207 570 422 605; baseline 0 0; x_size 40.365852; x_descenders 5.3658538; x_ascenders 15">
|
|
||||||
<span class='ocrx_word' id='word_1_12' title='bbox 207 573 285 605; x_wconf 96'>Pure</span>
|
|
||||||
<span class='ocrx_word' id='word_1_13' title='bbox 299 570 422 605; x_wconf 95'>Yellow</span>
|
|
||||||
</span>
|
|
||||||
</p>
|
|
||||||
</div>
|
|
||||||
</div>
|
|
||||||
</body>
|
|
||||||
</html>
|
|
||||||
-1
@@ -1 +0,0 @@
|
|||||||
Tesseract Open Source OCR Engine v4.0.0 with Leptonica
|
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user