251 lines
8.8 KiB
Python
251 lines
8.8 KiB
Python
# SPDX-FileCopyrightText: 2026 James R. Barlow
|
|
# SPDX-License-Identifier: AGPL-3.0-or-later
|
|
|
|
"""FastAPI application exposing OCRmyPDF as a batch web service."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
from contextlib import asynccontextmanager
|
|
from pathlib import Path
|
|
|
|
from fastapi import FastAPI, File, Form, HTTPException, Request, UploadFile
|
|
from fastapi.responses import (
|
|
FileResponse,
|
|
JSONResponse,
|
|
PlainTextResponse,
|
|
Response,
|
|
)
|
|
from fastapi.staticfiles import StaticFiles
|
|
from pydantic import ValidationError
|
|
from starlette.concurrency import run_in_threadpool
|
|
|
|
from webui import __version__
|
|
from webui.config import ALLOWED_SUFFIXES, get_settings, installed_languages
|
|
from webui.jobs import Batch, JobManager, JobStatus, safe_display_name
|
|
from webui.options import Mode, OcrOptions, OutputType
|
|
|
|
logging.basicConfig(
|
|
level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s"
|
|
)
|
|
log = logging.getLogger("webui")
|
|
|
|
STATIC_DIR = Path(__file__).parent / "static"
|
|
|
|
#: Bytes pulled from the request body per iteration while saving an upload.
|
|
CHUNK_SIZE = 1024 * 1024
|
|
|
|
|
|
@asynccontextmanager
|
|
async def lifespan(app: FastAPI):
|
|
"""Start the job manager with the app and tear it down cleanly."""
|
|
settings = get_settings()
|
|
log.info(
|
|
"Starting OCRmyPDF web UI %s (workers=%d, ocr_jobs=%d, work_dir=%s)",
|
|
__version__,
|
|
settings.workers,
|
|
settings.ocr_jobs,
|
|
settings.work_dir,
|
|
)
|
|
manager = JobManager(settings)
|
|
manager.sweep() # clear anything a previous container left behind
|
|
app.state.manager = manager
|
|
try:
|
|
yield
|
|
finally:
|
|
manager.shutdown()
|
|
|
|
|
|
app = FastAPI(
|
|
title="OCRmyPDF Web UI",
|
|
version=__version__,
|
|
lifespan=lifespan,
|
|
docs_url="/api/docs",
|
|
openapi_url="/api/openapi.json",
|
|
)
|
|
|
|
|
|
def get_manager(request: Request) -> JobManager:
|
|
"""Return the job manager created for this app instance."""
|
|
return request.app.state.manager
|
|
|
|
|
|
def require_batch(request: Request, batch_id: str) -> Batch:
|
|
"""Look up a batch, raising 404 if it has expired or never existed."""
|
|
batch = get_manager(request).get_batch(batch_id)
|
|
if batch is None:
|
|
raise HTTPException(
|
|
status_code=404, detail="This batch has expired or does not exist."
|
|
)
|
|
return batch
|
|
|
|
|
|
@app.get("/healthz", include_in_schema=False)
|
|
async def healthz() -> PlainTextResponse:
|
|
"""Liveness probe for container orchestration."""
|
|
return PlainTextResponse("ok")
|
|
|
|
|
|
@app.get("/api/config")
|
|
async def api_config() -> JSONResponse:
|
|
"""Describe what this deployment supports, so the UI can adapt to it."""
|
|
settings = get_settings()
|
|
return JSONResponse(
|
|
{
|
|
"version": __version__,
|
|
"languages": list(installed_languages()),
|
|
"modes": [mode.value for mode in Mode],
|
|
"output_types": [output.value for output in OutputType],
|
|
"max_files": settings.max_files_per_batch,
|
|
"max_upload_bytes": settings.max_upload_bytes,
|
|
"batch_ttl_seconds": settings.batch_ttl_seconds,
|
|
"accepted_extensions": sorted(ALLOWED_SUFFIXES),
|
|
"defaults": OcrOptions().model_dump(mode="json"),
|
|
}
|
|
)
|
|
|
|
|
|
@app.post("/api/batches", status_code=201)
|
|
async def create_batch(
|
|
request: Request,
|
|
# FastAPI's dependency markers are meant to be evaluated in the signature.
|
|
files: list[UploadFile] = File(...), # noqa: B008
|
|
options: str = Form("{}"), # noqa: B008
|
|
) -> JSONResponse:
|
|
"""Accept a set of files plus shared options and begin processing."""
|
|
settings = get_settings()
|
|
manager = get_manager(request)
|
|
|
|
try:
|
|
raw_options = json.loads(options or "{}")
|
|
if not isinstance(raw_options, dict):
|
|
raise ValueError("options must be a JSON object")
|
|
parsed = OcrOptions.model_validate(raw_options)
|
|
except (ValueError, ValidationError) as exc:
|
|
raise HTTPException(status_code=422, detail=f"Invalid options: {exc}") from exc
|
|
|
|
if not files:
|
|
raise HTTPException(status_code=400, detail="No files were uploaded.")
|
|
if len(files) > settings.max_files_per_batch:
|
|
raise HTTPException(
|
|
status_code=413,
|
|
detail=(
|
|
f"Too many files: {len(files)} sent, "
|
|
f"limit is {settings.max_files_per_batch} per batch."
|
|
),
|
|
)
|
|
|
|
batch = manager.create_batch(parsed)
|
|
try:
|
|
for index, upload in enumerate(files):
|
|
display_name = safe_display_name(upload.filename, f"file-{index + 1}.pdf")
|
|
suffix = Path(display_name).suffix.lower()
|
|
if suffix not in ALLOWED_SUFFIXES:
|
|
raise HTTPException(
|
|
status_code=415,
|
|
detail=(
|
|
f"“{display_name}” has an unsupported type. Accepted: "
|
|
+ ", ".join(sorted(ALLOWED_SUFFIXES))
|
|
),
|
|
)
|
|
job = manager.add_file(batch, display_name, suffix)
|
|
written = await _save_upload(
|
|
upload, job.input_path, settings.max_upload_bytes
|
|
)
|
|
if written == 0:
|
|
raise HTTPException(
|
|
status_code=400, detail=f"“{display_name}” is empty."
|
|
)
|
|
job.size_bytes = written
|
|
except HTTPException:
|
|
manager.delete_batch(batch.id)
|
|
raise
|
|
except Exception:
|
|
manager.delete_batch(batch.id)
|
|
log.exception("Failed to accept upload batch")
|
|
raise HTTPException(
|
|
status_code=500, detail="Could not store the uploaded files."
|
|
) from None
|
|
|
|
manager.start(batch)
|
|
log.info("Batch %s accepted with %d file(s)", batch.id, len(batch.jobs))
|
|
return JSONResponse(batch.to_dict(), status_code=201)
|
|
|
|
|
|
async def _save_upload(upload: UploadFile, destination: Path, limit: int) -> int:
|
|
"""Stream one upload to disk, aborting if it exceeds ``limit`` bytes."""
|
|
written = 0
|
|
with destination.open("wb") as out:
|
|
while chunk := await upload.read(CHUNK_SIZE):
|
|
written += len(chunk)
|
|
if written > limit:
|
|
out.close()
|
|
destination.unlink(missing_ok=True)
|
|
raise HTTPException(
|
|
status_code=413,
|
|
detail=(
|
|
f"“{safe_display_name(upload.filename, 'A file')}” exceeds "
|
|
f"the {limit // (1024 * 1024)} MB per-file limit."
|
|
),
|
|
)
|
|
out.write(chunk)
|
|
await upload.close()
|
|
return written
|
|
|
|
|
|
@app.get("/api/batches/{batch_id}")
|
|
async def get_batch(request: Request, batch_id: str) -> JSONResponse:
|
|
"""Poll the status of a batch and of every file in it."""
|
|
return JSONResponse(require_batch(request, batch_id).to_dict())
|
|
|
|
|
|
@app.delete("/api/batches/{batch_id}", status_code=204)
|
|
async def delete_batch(request: Request, batch_id: str) -> Response:
|
|
"""Cancel a batch and delete its files immediately."""
|
|
if not get_manager(request).delete_batch(batch_id):
|
|
raise HTTPException(status_code=404, detail="No such batch.")
|
|
return Response(status_code=204)
|
|
|
|
|
|
@app.get("/api/batches/{batch_id}/files/{job_id}")
|
|
async def download_file(request: Request, batch_id: str, job_id: str) -> FileResponse:
|
|
"""Download one finished PDF."""
|
|
found = get_manager(request).get_job(batch_id, job_id)
|
|
if found is None:
|
|
raise HTTPException(status_code=404, detail="No such file in this batch.")
|
|
_, job = found
|
|
if job.status is not JobStatus.succeeded or not job.output_path.exists():
|
|
raise HTTPException(status_code=409, detail="This file is not ready.")
|
|
return FileResponse(
|
|
job.output_path, media_type="application/pdf", filename=job.output_name
|
|
)
|
|
|
|
|
|
@app.get("/api/batches/{batch_id}/files/{job_id}/log")
|
|
async def file_log(request: Request, batch_id: str, job_id: str) -> PlainTextResponse:
|
|
"""Return the ocrmypdf output captured for one file."""
|
|
found = get_manager(request).get_job(batch_id, job_id)
|
|
if found is None:
|
|
raise HTTPException(status_code=404, detail="No such file in this batch.")
|
|
_, job = found
|
|
return PlainTextResponse("\n".join(job.log_lines) or "(no output)")
|
|
|
|
|
|
@app.get("/api/batches/{batch_id}/download")
|
|
async def download_zip(request: Request, batch_id: str) -> FileResponse:
|
|
"""Download every finished PDF in the batch as one zip archive."""
|
|
batch = require_batch(request, batch_id)
|
|
manager = get_manager(request)
|
|
zip_path = await run_in_threadpool(manager.build_zip, batch)
|
|
if zip_path is None:
|
|
raise HTTPException(
|
|
status_code=409, detail="No files in this batch finished successfully."
|
|
)
|
|
return FileResponse(
|
|
zip_path, media_type="application/zip", filename="ocrmypdf-results.zip"
|
|
)
|
|
|
|
|
|
app.mount("/", StaticFiles(directory=STATIC_DIR, html=True), name="static")
|