Remove synthetic training path and add fine-text OCR option
This commit is contained in:
+11
-4
@@ -132,7 +132,7 @@ def _prepare(data: bytes, header: str | None) -> tuple[bytes, tuple[int, int]]:
|
||||
raise InputError("invalid_image") from None
|
||||
|
||||
|
||||
def _classic(data: bytes, directory: Path) -> dict:
|
||||
def _classic(data: bytes, directory: Path, *, fine_text: bool = False) -> dict:
|
||||
started = time.monotonic()
|
||||
models = MODEL_HOME / "official_models"
|
||||
if any(not (models / name / "inference.pdiparams").is_file() for name in
|
||||
@@ -147,8 +147,11 @@ def _classic(data: bytes, directory: Path) -> dict:
|
||||
TRANSFORMERS_OFFLINE="1", PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK="True",
|
||||
OMP_NUM_THREADS="1", OPENBLAS_NUM_THREADS="1", OMP_THREAD_LIMIT="1")
|
||||
try:
|
||||
command = [sys.executable, "-m", "ocr_compare.classic_worker", str(image_path), str(output_path)]
|
||||
if fine_text:
|
||||
command.append("--fine-text")
|
||||
process = subprocess.run(
|
||||
[sys.executable, "-m", "ocr_compare.classic_worker", str(image_path), str(output_path)],
|
||||
command,
|
||||
cwd=PROJECT, env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
timeout=120, check=False,
|
||||
)
|
||||
@@ -165,7 +168,8 @@ def _classic(data: bytes, directory: Path) -> dict:
|
||||
for row in raw["lines"] if row.get("bbox")]
|
||||
except (OSError, ValueError, KeyError, TypeError):
|
||||
return {"status": "error", "error": "classic_invalid", "elapsed_ms": elapsed}
|
||||
return {"status": "ok", "model": "PP-OCRv5 mobile (Latin), lokal",
|
||||
model = "PP-OCRv5 mobile (Latin), lokal" + (" · Feintext" if fine_text else "")
|
||||
return {"status": "ok", "model": model,
|
||||
"elapsed_ms": elapsed, "lines": lines, "fields": assign_fields(lines),
|
||||
"regions": []}
|
||||
|
||||
@@ -269,6 +273,9 @@ async def compare(request: Request):
|
||||
return JSONResponse({"error": "busy"}, status_code=429)
|
||||
if request.headers.get("content-type", "").split(";", 1)[0] not in ("image/jpeg", "image/png"):
|
||||
return JSONResponse({"error": "unsupported_media_type"}, status_code=415)
|
||||
detail = request.headers.get("x-ocr-detail", "standard")
|
||||
if detail not in ("standard", "fine"):
|
||||
return JSONResponse({"error": "invalid_detail"}, status_code=422)
|
||||
async with _slot:
|
||||
data = bytearray()
|
||||
try:
|
||||
@@ -287,7 +294,7 @@ async def compare(request: Request):
|
||||
with tempfile.TemporaryDirectory(prefix="ocr-compare-") as path:
|
||||
directory = Path(path)
|
||||
classic, vision = await asyncio.gather(
|
||||
asyncio.to_thread(_classic, image, directory),
|
||||
asyncio.to_thread(_classic, image, directory, fine_text=detail == "fine"),
|
||||
asyncio.to_thread(_vision, image),
|
||||
)
|
||||
# Exact JPEG bytes sent to both pipelines are returned for evidence
|
||||
|
||||
Reference in New Issue
Block a user