Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
14365d10b8 | ||
|
|
5e5320020f | ||
|
|
103c3e0cd6 | ||
|
|
7a1c89edd9 | ||
|
|
a5ff3d2f42 |
@@ -404,7 +404,7 @@ the following when running in an Administrator command prompt):
|
|||||||
* ``choco install --pre tesseract``
|
* ``choco install --pre tesseract``
|
||||||
* ``choco install pngquant`` (optional)
|
* ``choco install pngquant`` (optional)
|
||||||
|
|
||||||
Either set of commands will install the required software. At the mmoment there is no
|
Either set of commands will install the required software. At the moment there is no
|
||||||
single command to install Windows.
|
single command to install Windows.
|
||||||
|
|
||||||
You may then use ``pip`` to install ocrmypdf. (This can performed by a user or
|
You may then use ``pip`` to install ocrmypdf. (This can performed by a user or
|
||||||
@@ -681,4 +681,4 @@ can still install and use OCRmyPDF. A warning message will appear.
|
|||||||
In practice, OCRmyPDF may need more than 32-bit memory space to run when
|
In practice, OCRmyPDF may need more than 32-bit memory space to run when
|
||||||
large documents are processed, so there are practical limitations to what
|
large documents are processed, so there are practical limitations to what
|
||||||
users can accomplish with it. Still, for the common use case of an 32-bit
|
users can accomplish with it. Still, for the common use case of an 32-bit
|
||||||
ARM NAS or Raspberry Pi processing small documents, it should work.
|
ARM NAS or Raspberry Pi processing small documents, it should work.
|
||||||
|
|||||||
+31
-13
@@ -18,16 +18,26 @@ Tesseract's documentation also lists the three-letter code for your language.
|
|||||||
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
Some are anglicized, e.g. Spanish is ``spa`` rather than ``esp``, while others
|
||||||
are not, e.g. German is ``deu`` and French is ``fra``.
|
are not, e.g. German is ``deu`` and French is ``fra``.
|
||||||
|
|
||||||
|
Language packs (strictly speaking, Tesseract "traineddata" files) generally correspond
|
||||||
|
to the language in question, but different language packs are used in certain
|
||||||
|
situations. For German, the "Fraktur" language pack can assist with reading older
|
||||||
|
materials in the Fraktur typeface family (``deu_frak``). Some communities have changed
|
||||||
|
their script from Cyrillic to Latin; the Cyrillic version of Uzbek is available
|
||||||
|
as ``uzb_cyrl`` and the Latin version is ``uzb``.
|
||||||
|
|
||||||
After you have installed a language pack, you can use it with ``ocrmypdf -l <language>``,
|
After you have installed a language pack, you can use it with ``ocrmypdf -l <language>``,
|
||||||
for example ``ocrmypdf -l spa``. For multilingual documents, you can specify
|
for example ``ocrmypdf -l spa``. For multilingual documents, you can specify
|
||||||
all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French.
|
all languages to be expected, e.g. ``ocrmypdf -l eng+fra`` for English and French.
|
||||||
English is assumed by default unless other language(s) are specified.
|
English is assumed by default unless other language(s) are specified.
|
||||||
|
|
||||||
For Linux users, you can often find packages that provide language
|
For Linux users, you can often find packages that provide language
|
||||||
packs:
|
packs.
|
||||||
|
|
||||||
Debian and Ubuntu users
|
Platform install steps
|
||||||
=======================
|
======================
|
||||||
|
|
||||||
|
Debian and Ubuntu (apt)
|
||||||
|
-----------------------
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -42,8 +52,8 @@ to what languages it should search for. Multiple languages can be
|
|||||||
requested using either ``-l eng+fra`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fra``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
Fedora users
|
Fedora
|
||||||
============
|
------
|
||||||
|
|
||||||
.. code-block:: bash
|
.. code-block:: bash
|
||||||
|
|
||||||
@@ -58,8 +68,8 @@ to what languages it should search for. Multiple languages can be
|
|||||||
requested using either ``-l eng+fra`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fra``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
Gentoo users
|
Gentoo
|
||||||
============
|
------
|
||||||
|
|
||||||
On Gentoo the package ``app-text/tessdata_fast``, which ``app-text/tesseract`` depends on, handles Tesseract languages.
|
On Gentoo the package ``app-text/tessdata_fast``, which ``app-text/tesseract`` depends on, handles Tesseract languages.
|
||||||
It accepts USE flags to select what languages should be installed, these can be set in ``/etc/portage/package.use``.
|
It accepts USE flags to select what languages should be installed, these can be set in ``/etc/portage/package.use``.
|
||||||
@@ -85,23 +95,31 @@ to what languages it should search for. Multiple languages can be
|
|||||||
requested using either ``-l eng+fra`` (English and French) or
|
requested using either ``-l eng+fra`` (English and French) or
|
||||||
``-l eng -l fra``.
|
``-l eng -l fra``.
|
||||||
|
|
||||||
macOS users
|
macOS
|
||||||
===========
|
-----
|
||||||
|
|
||||||
You can install additional language packs by
|
You can install additional language packs by
|
||||||
:ref:`installing Tesseract using Homebrew with all language packs <macos-all-languages>`.
|
:ref:`installing Tesseract using Homebrew with all language packs <macos-all-languages>`.
|
||||||
|
|
||||||
Docker users
|
Docker
|
||||||
============
|
------
|
||||||
|
|
||||||
Users of the OCRmyPDF Docker image should install language packs into a
|
Users of the OCRmyPDF Docker image should install language packs into a
|
||||||
derived Docker image as
|
derived Docker image as
|
||||||
:ref:`described in that section <docker-lang-packs>`.
|
:ref:`described in that section <docker-lang-packs>`.
|
||||||
|
|
||||||
Windows users
|
Windows
|
||||||
=============
|
-------
|
||||||
|
|
||||||
The Tesseract installer provided by Chocolatey currently includes only English language.
|
The Tesseract installer provided by Chocolatey currently includes only English language.
|
||||||
To install other languages, download the respective language pack (``.traineddata`` file)
|
To install other languages, download the respective language pack (``.traineddata`` file)
|
||||||
from https://github.com/tesseract-ocr/tessdata/ and place it in
|
from https://github.com/tesseract-ocr/tessdata/ and place it in
|
||||||
``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed).
|
``C:\\Program Files\\Tesseract-OCR\\tessdata`` (or wherever Tesseract OCR is installed).
|
||||||
|
|
||||||
|
Custom language packs
|
||||||
|
=====================
|
||||||
|
|
||||||
|
If you have fine-tuned or trained Tesseract and generated custom trained data, you can
|
||||||
|
copy your ``customlang.traineddata`` file into your Tesseract "tessdata" folder, and
|
||||||
|
then use the ``-l customlang`` argument to tell OCRmyPDF to pass that language on to
|
||||||
|
Tesseract.
|
||||||
@@ -31,6 +31,14 @@ OCRmyPDF typically supports the three most recent Python versions.
|
|||||||
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
.. |OCRmyPDF PyPI| image:: https://img.shields.io/pypi/v/ocrmypdf.svg
|
||||||
|
|
||||||
|
|
||||||
|
v16.0.3
|
||||||
|
=======
|
||||||
|
|
||||||
|
- Changed minimum required Ghostscript to 9.54, to support users of RHEL 9 and its
|
||||||
|
derivatives, since that is the latest version available there.
|
||||||
|
- Removed warning message about CVE-2023-43115, on the assumption that most
|
||||||
|
distributions have backported the patch by now.
|
||||||
|
|
||||||
v16.0.2
|
v16.0.2
|
||||||
=======
|
=======
|
||||||
|
|
||||||
|
|||||||
@@ -54,7 +54,7 @@ def check_options(options):
|
|||||||
program='gs',
|
program='gs',
|
||||||
package='ghostscript',
|
package='ghostscript',
|
||||||
version_checker=ghostscript.version,
|
version_checker=ghostscript.version,
|
||||||
need_version='9.55', # Ubuntu 22.04's version
|
need_version='9.54', # RHEL 9's version; Ubuntu 22.04 has 9.55
|
||||||
)
|
)
|
||||||
gs_version = ghostscript.version()
|
gs_version = ghostscript.version()
|
||||||
if gs_version in BLACKLISTED_GS_VERSIONS:
|
if gs_version in BLACKLISTED_GS_VERSIONS:
|
||||||
@@ -62,14 +62,6 @@ def check_options(options):
|
|||||||
f"Ghostscript {gs_version} contains serious regressions and is not "
|
f"Ghostscript {gs_version} contains serious regressions and is not "
|
||||||
"supported. Please upgrade to a newer version."
|
"supported. Please upgrade to a newer version."
|
||||||
)
|
)
|
||||||
if gs_version < Version('10.02.0'):
|
|
||||||
log.warning(
|
|
||||||
f"The installed version of Ghostscript {gs_version}, contains a remote "
|
|
||||||
"code execution security vulnerability. Please upgrade to a newer "
|
|
||||||
"version. For details see CVE-2023-43115. The issue is not known to "
|
|
||||||
"affect OCRmyPDF or processing PDFs with Ghostscript, but upgrading "
|
|
||||||
"Ghostscript is recommended."
|
|
||||||
)
|
|
||||||
|
|
||||||
if options.output_type == 'pdfa':
|
if options.output_type == 'pdfa':
|
||||||
options.output_type = 'pdfa-2'
|
options.output_type = 'pdfa-2'
|
||||||
|
|||||||
@@ -4,6 +4,7 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import os
|
import os
|
||||||
|
import platform
|
||||||
|
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
@@ -13,6 +14,9 @@ from .conftest import run_ocrmypdf_api
|
|||||||
|
|
||||||
|
|
||||||
@pytest.mark.skipif(os.name == 'nt', reason="Windows doesn't have SIGKILL")
|
@pytest.mark.skipif(os.name == 'nt', reason="Windows doesn't have SIGKILL")
|
||||||
|
@pytest.mark.skipif(
|
||||||
|
platform.python_version_tuple() >= ('3', '12'), reason="can deadlock due to fork"
|
||||||
|
)
|
||||||
def test_simulate_oom_killer(multipage, no_outpdf):
|
def test_simulate_oom_killer(multipage, no_outpdf):
|
||||||
exitcode = run_ocrmypdf_api(
|
exitcode = run_ocrmypdf_api(
|
||||||
multipage,
|
multipage,
|
||||||
|
|||||||
Reference in New Issue
Block a user