Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6eb393590b | ||
|
|
07c6654057 | ||
|
|
4e15eb8d14 | ||
|
|
8b01ab8ad2 | ||
|
|
e0a522ad50 |
@@ -12,10 +12,18 @@ may be unreliable. Use the API to depend on precise behavior.
|
||||
The public API may be useful in scripts that launch OCRmyPDF processes or that
|
||||
wish to use some of its features for working with PDFs.
|
||||
|
||||
v11.2.0
|
||||
=======
|
||||
|
||||
- Fixed an issue with optimizing PNG-type images that had soft masks or image masks.
|
||||
This is a regression introduced in (or about) v11.1.0.
|
||||
- Improved type checking of the ``plugins`` parameter for the ``ocrmypdf.ocr``
|
||||
API call.
|
||||
|
||||
v11.1.2
|
||||
=======
|
||||
|
||||
- Fix hOCR renderer writing the text in roughly reverse order. This should not
|
||||
- Fixed hOCR renderer writing the text in roughly reverse order. This should not
|
||||
affect reasonably smart PDF readers that properly locate the position of all
|
||||
text, but may confuse those that rely on the order of objects in the content
|
||||
stream. (#642)
|
||||
|
||||
@@ -18,6 +18,25 @@
|
||||
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
# SOFTWARE.
|
||||
|
||||
"""
|
||||
An example of an OCRmyPDF plugin.
|
||||
|
||||
This plugin adds two new command line arguments
|
||||
--grayscale-ocr: converts the image to grayscale before performing OCR on it
|
||||
(This is occasionally useful for images whose color confounds OCR. It only
|
||||
affects the image shown to OCR. The image is not saved.)
|
||||
--mono-page: converts pages all pages in the output file to black and white
|
||||
|
||||
To use this from the command line:
|
||||
ocrmypdf --plugin path/to/example_plugin.py --mono-page input.pdf output.pdf
|
||||
|
||||
To use this as an API:
|
||||
import ocrmypdf
|
||||
ocrmypdf.ocr('input.pdf', 'output.pdf',
|
||||
plugins=['path/to/example_plugin.py'], mono_page=True
|
||||
)
|
||||
"""
|
||||
|
||||
import logging
|
||||
|
||||
from PIL import Image
|
||||
|
||||
+3
-1
@@ -226,7 +226,7 @@ def ocr( # pylint: disable=unused-argument
|
||||
user_words: os.PathLike = None,
|
||||
user_patterns: os.PathLike = None,
|
||||
fast_web_view: float = None,
|
||||
plugins: Iterable[str] = None,
|
||||
plugins: Iterable[Union[str, Path]] = None,
|
||||
keep_temporary_files: bool = None,
|
||||
progress_bar: bool = None,
|
||||
**kwargs,
|
||||
@@ -280,6 +280,8 @@ def ocr( # pylint: disable=unused-argument
|
||||
"""
|
||||
if not plugins:
|
||||
plugins = []
|
||||
elif isinstance(plugins, (str, Path)):
|
||||
plugins = [plugins]
|
||||
else:
|
||||
plugins = list(plugins)
|
||||
|
||||
|
||||
@@ -77,23 +77,26 @@ def extract_image_filter(
|
||||
if image.Subtype != Name.Image:
|
||||
return None
|
||||
if image.Length < 100:
|
||||
log.debug("Skipping small image, xref %s", xref)
|
||||
log.debug(f"Skipping small image, xref {xref}")
|
||||
return None
|
||||
|
||||
pim = PdfImage(image)
|
||||
|
||||
if len(pim.filter_decodeparms) > 1:
|
||||
log.debug("Skipping multiply filtered, xref %s", xref)
|
||||
log.debug(f"Skipping multiply filtered image, xref {xref}")
|
||||
return None
|
||||
filtdp = pim.filter_decodeparms[0]
|
||||
|
||||
if pim.bits_per_component > 8:
|
||||
log.debug(f"Skipping wide gamut image, xref {xref}")
|
||||
return None # Don't mess with wide gamut images
|
||||
|
||||
if filtdp[0] == Name.JPXDecode:
|
||||
log.debug(f"Skipping JPEG2000 iamge, xref {xref}")
|
||||
return None # Don't do JPEG2000
|
||||
|
||||
if Name.Decode in image:
|
||||
log.debug(f"Skipping image with Decode table, xref {xref}")
|
||||
return None # Don't mess with custom Decode tables
|
||||
|
||||
return pim, filtdp
|
||||
@@ -229,7 +232,9 @@ def extract_images(
|
||||
# Ignore soft masks
|
||||
smask_xref = Xref(image.SMask.objgen[0])
|
||||
exclude_xrefs.add(smask_xref)
|
||||
log.debug(f"Skipping image {smask_xref} because it is an SMask")
|
||||
include_xrefs.add(xref)
|
||||
log.debug(f"Treating {xref} as an optimization candidate")
|
||||
if xref not in pageno_for_xref:
|
||||
pageno_for_xref[xref] = pageno
|
||||
|
||||
@@ -411,9 +416,25 @@ def _transcode_png(pike: Pdf, filename: Path, xref: Xref) -> bool:
|
||||
decode_parms=local_image.DecodeParms,
|
||||
)
|
||||
|
||||
# Don't copy keys from the new image...
|
||||
del_keys = set(im_obj.keys()) - set(local_image.keys())
|
||||
# ...except for the keep_fields, which are essential to displaying
|
||||
# the image correctly and preserving its metadata. (/Decode arrays
|
||||
# and /SMaskInData are implicitly discarded prior to this point.)
|
||||
keep_fields = {
|
||||
'/ID',
|
||||
'/Intent',
|
||||
'/Interpolate',
|
||||
'/Mask',
|
||||
'/Metadata',
|
||||
'/OC',
|
||||
'/OPI',
|
||||
'/SMask',
|
||||
'/StructParent',
|
||||
}
|
||||
del_keys -= keep_fields
|
||||
for key in local_image.keys():
|
||||
if key != Name.Length:
|
||||
if key != Name.Length and str(key) not in keep_fields:
|
||||
im_obj[key] = local_image[key]
|
||||
for key in del_keys:
|
||||
del im_obj[key]
|
||||
|
||||
Reference in New Issue
Block a user