Add private local OCR comparison package
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Local comparison UI for real-document OCR evaluation."""
|
||||
@@ -0,0 +1,318 @@
|
||||
"use strict";
|
||||
|
||||
const $ = id => document.getElementById(id);
|
||||
const input = $("image-file"), original = $("original"), cornersCanvas = $("corners");
|
||||
const processed = $("processed"), evidenceCanvas = $("evidence");
|
||||
const run = $("run"), status = $("status"), rotation = $("rotation");
|
||||
const initialCorners = () => [[0, 0], [1, 0], [1, 1], [0, 1]];
|
||||
let corners = initialCorners(), edited = false, busy = false, ready = false;
|
||||
let objectUrl = null, drag = null, result = null, selectedEvidence = null;
|
||||
const manual = {classic: new Map(), vision: new Map()};
|
||||
const suppressed = {classic: new Set(), vision: new Set()};
|
||||
const priorityCodes = ["C.1.1", "C.1.2", "C.1.3", "B", "2.1", "2.2"];
|
||||
const reviewMethods = new Set(["merged_code", "vl_table_context_code", "caption_region_next_line", "surname_above_forename"]);
|
||||
const fieldNames = {"A":"Kennzeichen","C.1.1":"Name / Firmenname","C.1.2":"Vorname",
|
||||
"C.1.3":"Anschrift","B":"Erstzulassung","2.1":"Herstellerschlüssel (HSN)",
|
||||
"2.2":"Typschlüssel (TSN)","2":"Hersteller","D.1":"Marke","D.2":"Typ / Variante",
|
||||
"D.3":"Handelsbezeichnung","E":"Fahrzeug-Identifizierungsnummer","P.3":"Kraftstoffart",
|
||||
"R":"Farbe","17":"Merkmal zur Betriebserlaubnis"};
|
||||
|
||||
function setStatus(message, kind = "") {
|
||||
status.textContent = message;
|
||||
status.className = `status ${kind}`;
|
||||
}
|
||||
function validCorners() {
|
||||
if (!ready) return false;
|
||||
if (!edited) return true;
|
||||
const w = original.naturalWidth, h = original.naturalHeight;
|
||||
const points = corners.map(([x, y]) => [x * w, y * h]);
|
||||
const edges = points.map((p, i) => [points[(i + 1) % 4][0] - p[0], points[(i + 1) % 4][1] - p[1]]);
|
||||
const crosses = edges.map((a, i) => a[0] * edges[(i + 1) % 4][1] - a[1] * edges[(i + 1) % 4][0]);
|
||||
const area = Math.abs(points.reduce((sum, p, i) => sum + p[0] * points[(i + 1) % 4][1] - p[1] * points[(i + 1) % 4][0], 0)) / 2;
|
||||
return crosses.every(value => value > 0) && area > .02 * w * h &&
|
||||
edges.every(([x, y]) => Math.hypot(x, y) >= 30);
|
||||
}
|
||||
function updateControls() {
|
||||
input.disabled = busy;
|
||||
$("reset-corners").disabled = !ready || busy;
|
||||
rotation.disabled = busy;
|
||||
run.disabled = busy || !validCorners();
|
||||
}
|
||||
function drawCorners() {
|
||||
const w = original.naturalWidth, h = original.naturalHeight;
|
||||
if (!w || !h) return;
|
||||
cornersCanvas.width = w;
|
||||
cornersCanvas.height = h;
|
||||
const ctx = cornersCanvas.getContext("2d");
|
||||
const pts = corners.map(([x, y]) => [x * (w - 1), y * (h - 1)]);
|
||||
ctx.strokeStyle = "#1459dd";
|
||||
ctx.lineWidth = Math.max(5, w / 550);
|
||||
ctx.beginPath();
|
||||
pts.forEach(([x, y], i) => i ? ctx.lineTo(x, y) : ctx.moveTo(x, y));
|
||||
ctx.closePath();ctx.stroke();
|
||||
const radius = Math.max(25, w / 110);
|
||||
pts.forEach(([x, y], i) => {
|
||||
ctx.beginPath();ctx.arc(x, y, radius, 0, 2 * Math.PI);
|
||||
ctx.fillStyle = "#fff";ctx.fill();ctx.stroke();
|
||||
ctx.fillStyle = "#1547a5";ctx.font = `bold ${Math.max(22, w / 125)}px system-ui`;
|
||||
ctx.fillText(String(i + 1), Math.min(w - radius, x + radius * .9), Math.max(radius, y - radius * .65));
|
||||
});
|
||||
}
|
||||
function pointInCanvas(event) {
|
||||
const b = cornersCanvas.getBoundingClientRect();
|
||||
return [(event.clientX - b.left) / b.width, (event.clientY - b.top) / b.height];
|
||||
}
|
||||
cornersCanvas.addEventListener("pointerdown", event => {
|
||||
if (!ready || busy) return;
|
||||
const b = cornersCanvas.getBoundingClientRect();
|
||||
const [x, y] = pointInCanvas(event);
|
||||
const distances = corners.map(([cx, cy]) => Math.hypot((cx - x) * b.width, (cy - y) * b.height));
|
||||
const index = distances.indexOf(Math.min(...distances));
|
||||
if (distances[index] > 32) return;
|
||||
drag = index;
|
||||
cornersCanvas.setPointerCapture(event.pointerId);
|
||||
event.preventDefault();
|
||||
});
|
||||
cornersCanvas.addEventListener("pointermove", event => {
|
||||
if (drag === null || busy) return;
|
||||
const [x, y] = pointInCanvas(event);
|
||||
corners[drag] = [Math.max(0, Math.min(1, x)), Math.max(0, Math.min(1, y))];
|
||||
edited = true; result = null; $("comparison").hidden = true;
|
||||
drawCorners();updateControls();
|
||||
setStatus(validCorners() ? "Ecken geändert. Bitte beide Verfahren neu starten." :
|
||||
"Ecken müssen ein nicht gekreuztes Viereck um das Dokument bilden.", validCorners() ? "" : "error");
|
||||
});
|
||||
function finishDrag() {drag = null;}
|
||||
cornersCanvas.addEventListener("pointerup", finishDrag);
|
||||
cornersCanvas.addEventListener("pointercancel", finishDrag);
|
||||
$("reset-corners").addEventListener("click", () => {
|
||||
corners = initialCorners(); edited = false; result = null;
|
||||
$("comparison").hidden = true;drawCorners();updateControls();setStatus("Ganzes Bild ausgewählt.");
|
||||
});
|
||||
rotation.addEventListener("change", () => {
|
||||
result = null;$("comparison").hidden = true;setStatus("Leserichtung geändert. Bitte erneut vergleichen.");
|
||||
});
|
||||
input.addEventListener("change", () => {
|
||||
if (objectUrl) URL.revokeObjectURL(objectUrl);
|
||||
objectUrl = null; ready = false; edited = false; corners = initialCorners();
|
||||
drag = null; result = null; selectedEvidence = null;
|
||||
manual.classic.clear();manual.vision.clear();
|
||||
suppressed.classic.clear();suppressed.vision.clear();
|
||||
$("comparison").hidden = true;rotation.value = "0";
|
||||
const file = input.files?.[0];
|
||||
$("original-wrap").hidden = !file;
|
||||
$("file-name").textContent = file ? `${file.name} · ${(file.size / 1048576).toFixed(2)} MiB` : "JPEG oder PNG, maximal 16 MiB.";
|
||||
if (file) {objectUrl = URL.createObjectURL(file);original.src = objectUrl;}
|
||||
else original.removeAttribute("src");
|
||||
setStatus("");updateControls();
|
||||
});
|
||||
original.addEventListener("load", () => {
|
||||
ready = true;drawCorners();updateControls();
|
||||
$("file-name").textContent += ` · ${original.naturalWidth} × ${original.naturalHeight} Pixel`;
|
||||
});
|
||||
original.addEventListener("error", () => {
|
||||
ready = false;updateControls();setStatus("Das Bild kann nicht geöffnet werden. Bitte JPEG oder PNG verwenden.", "error");
|
||||
});
|
||||
window.addEventListener("pagehide", () => {if (objectUrl) URL.revokeObjectURL(objectUrl);});
|
||||
|
||||
const errorText = {
|
||||
busy: "Ein Vergleich läuft bereits.", unsupported_media_type: "Nur JPEG und PNG sind erlaubt.",
|
||||
upload_too_large: "Das Bild ist größer als 16 MiB.", upload_timeout: "Der Upload hat zu lange gedauert.",
|
||||
invalid_image: "Die Bilddatei lässt sich nicht vollständig lesen.",
|
||||
unsupported_encoding: "Der Bildinhalt ist kein unterstütztes JPEG oder PNG.",
|
||||
image_too_large: "Die Auflösung überschreitet 24 Megapixel oder 10.000 Pixel pro Seite.",
|
||||
invalid_corners: "Die vier Ecken bilden kein gültiges Dokumentviereck.",
|
||||
invalid_rotation: "Die gewählte Drehung ist ungültig.",
|
||||
classic_timeout: "Die klassische OCR hat das Zeitlimit überschritten.",
|
||||
classic_models_missing: "Die lokalen OCR-Modelle fehlen. Zuerst die Modell-Einrichtung aus der README ausführen.",
|
||||
classic_failed: "Die klassische OCR ist lokal fehlgeschlagen.",
|
||||
classic_invalid: "Die klassische OCR hat ein ungültiges Ergebnis geliefert.",
|
||||
vision_timeout: "PaddleOCR-VL hat das Zeitlimit überschritten.",
|
||||
vision_failed: "PaddleOCR-VL konnte auf der konfigurierten eigenen Hardware nicht ausgeführt werden.",
|
||||
vision_disabled: "Vision ist in dieser Installation deaktiviert.",
|
||||
vision_config_error: "Die Vision-Konfiguration ist unvollständig oder ungültig.",
|
||||
vision_models_missing: "Die lokalen Vision-Modelle fehlen. Zuerst die Modell-Einrichtung aus der README ausführen.",
|
||||
vision_invalid: "PaddleOCR-VL hat ein ungültiges Ergebnis geliefert.",
|
||||
};
|
||||
|
||||
function element(tag, value, className = "") {
|
||||
const node = document.createElement(tag);
|
||||
node.textContent = value;
|
||||
if (className) node.className = className;
|
||||
return node;
|
||||
}
|
||||
function drawEvidence() {
|
||||
if (!result || !processed.naturalWidth) return;
|
||||
const w = processed.naturalWidth, h = processed.naturalHeight;
|
||||
evidenceCanvas.width = w;evidenceCanvas.height = h;
|
||||
const ctx = evidenceCanvas.getContext("2d");
|
||||
if ($( "all-boxes" ).checked) {
|
||||
for (const [engine, branch] of Object.entries(result.results)) {
|
||||
if (branch.status !== "ok") continue;
|
||||
ctx.strokeStyle = engine === "classic" ? "#008778" : "#425edf";
|
||||
ctx.lineWidth = Math.max(2, w / 900);
|
||||
for (const line of branch.lines) if (line.box) {
|
||||
const [x1, y1, x2, y2] = line.box;
|
||||
ctx.strokeRect(x1, y1, x2 - x1, y2 - y1);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (selectedEvidence?.anchor && selectedEvidence.anchor.join() !== selectedEvidence.box.join()) {
|
||||
const [x1, y1, x2, y2] = selectedEvidence.anchor;
|
||||
ctx.strokeStyle = "#d88614";ctx.fillStyle = "rgba(216,134,20,.13)";
|
||||
ctx.lineWidth = Math.max(4, w / 400);
|
||||
ctx.fillRect(x1, y1, x2 - x1, y2 - y1);
|
||||
ctx.strokeRect(x1, y1, x2 - x1, y2 - y1);
|
||||
}
|
||||
if (selectedEvidence?.box) {
|
||||
const [x1, y1, x2, y2] = selectedEvidence.box;
|
||||
ctx.strokeStyle = "#d72b4a";ctx.fillStyle = "rgba(215,43,74,.16)";
|
||||
ctx.lineWidth = Math.max(4, w / 400);
|
||||
ctx.fillRect(x1, y1, x2 - x1, y2 - y1);
|
||||
ctx.strokeRect(x1, y1, x2 - x1, y2 - y1);
|
||||
}
|
||||
}
|
||||
function focusEvidence(box, anchor = null) {
|
||||
if (!box) return;
|
||||
selectedEvidence = {box, anchor};drawEvidence();
|
||||
$("comparison").scrollIntoView({behavior:"smooth",block:"start"});
|
||||
}
|
||||
processed.addEventListener("load", drawEvidence);
|
||||
$("all-boxes").addEventListener("change", drawEvidence);
|
||||
|
||||
function renderCard(engine) {
|
||||
const card = $(`${engine}-card`), branch = result.results[engine];
|
||||
card.replaceChildren(element("h3", engine === "classic" ? "A · Klassische OCR" : "B · Vision-Dokumentmodell"));
|
||||
if (branch.status !== "ok") {
|
||||
card.append(element("p", errorText[branch.error] || "Dieses Verfahren ist fehlgeschlagen.", "error"),
|
||||
element("p", `${(branch.elapsed_ms / 1000).toFixed(1)} s`, "meta"));
|
||||
return;
|
||||
}
|
||||
const automatic = branch.fields.filter(field => !suppressed[engine].has(field.code));
|
||||
card.append(element("p", `${branch.model} · ${(branch.elapsed_ms / 1000).toFixed(1)} s · ${branch.lines.length} Textbereiche · ${automatic.length} automatische Feldvorschläge`, "meta"));
|
||||
if (branch.layout_note) card.append(element("p", branch.layout_note, "note"));
|
||||
card.append(element("p", "Die Zuordnung ist ein Vorschlag. Orange markiert den gelesenen Feldcode, Rot den Wert. Falsche Vorschläge lassen sich verwerfen; der Rohtext bleibt erhalten.", "assignment-note"));
|
||||
card.append(element("h4", "Strukturierte Feldvorschläge"));
|
||||
const merged = new Map(automatic.map(field => [field.code, field]));
|
||||
for (const [code, field] of manual[engine]) merged.set(code, field);
|
||||
const priority = element("div", "", "priority-fields");
|
||||
priority.append(element("h5", "Wichtige Angaben"));
|
||||
for (const code of priorityCodes) {
|
||||
const field = merged.get(code);
|
||||
priority.append(element("span", `${code} ${fieldNames[code]}: ${field ? (field.manual ? "manuell" : reviewMethods.has(field.method) ? "prüfen" : "Vorschlag") : "offen"}`,
|
||||
`priority-item${field ? "" : " missing"}`));
|
||||
}
|
||||
card.append(priority);
|
||||
if (!merged.size) card.append(element("p", "Keine sicher belegbaren Felder. Rohtext prüfen und bei Bedarf manuell zuordnen.", "meta"));
|
||||
const fieldList = element("div", "", "field-list");
|
||||
for (const field of merged.values()) {
|
||||
const row = element("div", "", `field-item${reviewMethods.has(field.method) ? " needs-review" : ""}`);
|
||||
const show = element("button", "", "field-show");show.type = "button";
|
||||
const heading = element("span", "", "field-heading");
|
||||
heading.append(element("strong", field.code), element("small", fieldNames[field.code] || "Feldcode"));
|
||||
const method = field.manual ? "Manuell aus OCR-Text" :
|
||||
field.method === "adjacent_code" ? "Code und Wert in Nachbarboxen" :
|
||||
field.method === "below_caption" ? "Wert unter gedruckter Beschriftung" :
|
||||
field.method === "caption_region_next_line" ? "Nächste Textzeile derselben Layoutregion · bitte prüfen" :
|
||||
field.method === "surname_above_forename" ? "C.1.1-Code nicht gelesen · Wert oberhalb von C.1.2 · bitte prüfen" :
|
||||
field.method === "vl_table_adjacent_cell" ? "Benachbarte Modellzellen · grober Tabellenbeleg" :
|
||||
field.method === "vl_table_context_code" ? "Punkt im Feldcode fehlt · Tabellenfolge B/2.1/2.2 · bitte prüfen" :
|
||||
field.method === "merged_code" ? "Code und Wert verschmolzen · bitte prüfen" :
|
||||
"Code und Wert in einer Textzeile";
|
||||
show.append(heading, element("span", field.value, "field-value"), element("small", method, "field-method"));
|
||||
show.disabled = !field.box;
|
||||
show.addEventListener("click", () => focusEvidence(field.box, field.anchor_box));
|
||||
const dismiss = element("button", field.manual ? "Entfernen" : "Verwerfen", "field-dismiss");
|
||||
dismiss.type = "button";
|
||||
dismiss.setAttribute("aria-label", `${field.code} ${field.manual ? "entfernen" : "verwerfen"}`);
|
||||
dismiss.addEventListener("click", () => {
|
||||
if (field.manual) manual[engine].delete(field.code);
|
||||
else suppressed[engine].add(field.code);
|
||||
renderCard(engine);
|
||||
});
|
||||
row.append(show, dismiss);
|
||||
fieldList.append(row);
|
||||
}
|
||||
card.append(fieldList, element("h4", "Rohtext mit Bildbelegen"));
|
||||
if (!branch.lines.length) card.append(element("p", "Kein Text erkannt.", "meta"));
|
||||
const raw = element("div", "", "raw-lines");
|
||||
let code, lineSelect, value, form;
|
||||
branch.lines.forEach((line, index) => {
|
||||
const entry = element("div", "", "raw-entry");
|
||||
const row = element("button", "", "raw-line");row.type = "button";
|
||||
row.append(element("span", `${index + 1}.`), element("span", line.text));
|
||||
row.disabled = !line.box;
|
||||
row.title = line.source || "Bildbereich markieren";
|
||||
row.addEventListener("click", () => {
|
||||
lineSelect.value = String(index);value.value = line.text;
|
||||
focusEvidence(line.box);
|
||||
});
|
||||
const pick = element("button", "Zuordnen", "raw-pick");pick.type = "button";
|
||||
pick.disabled = !line.box;
|
||||
pick.addEventListener("click", () => {
|
||||
lineSelect.value = String(index);value.value = line.text;
|
||||
selectedEvidence = {box:line.box,anchor:null};drawEvidence();
|
||||
form.scrollIntoView({behavior:"smooth",block:"center"});code.focus();
|
||||
});
|
||||
entry.append(row,pick);raw.append(entry);
|
||||
});
|
||||
card.append(raw);
|
||||
if (!branch.lines.length) return;
|
||||
card.append(element("h4", "Zeile manuell zuordnen"));
|
||||
form = element("div", "", "manual");
|
||||
code = document.createElement("select");lineSelect = document.createElement("select");
|
||||
code.setAttribute("aria-label", "Feldcode");lineSelect.setAttribute("aria-label", "OCR-Zeile");
|
||||
code.append(new Option("Feldcode", ""));
|
||||
result.field_codes.forEach(fieldCode => code.append(new Option(`${fieldCode} · ${fieldNames[fieldCode] || "Feldcode"}`, fieldCode)));
|
||||
lineSelect.append(new Option("OCR-Zeile wählen", ""));
|
||||
branch.lines.forEach((row, index) => lineSelect.append(new Option(`${index + 1}: ${row.text.slice(0, 55)}`, String(index))));
|
||||
value = document.createElement("input");value.type = "text";value.maxLength = 500;
|
||||
value.setAttribute("aria-label", "Wörtlicher Ausschnitt aus der OCR-Zeile");
|
||||
value.placeholder = "OCR-Zeile wählen, dann bei Bedarf auf den Wert kürzen";
|
||||
lineSelect.addEventListener("change", () => {value.value = lineSelect.value === "" ? "" : branch.lines[Number(lineSelect.value)].text;});
|
||||
const add = element("button", "Zuordnen");add.type = "button";
|
||||
const hint = element("p", "Nur wörtlich erkannter Text mit Bildbereich ist zulässig.", "manual-hint");
|
||||
add.addEventListener("click", () => {
|
||||
const row = lineSelect.value === "" ? null : branch.lines[Number(lineSelect.value)];
|
||||
const chosen = value.value.trim();
|
||||
if (!code.value || !chosen || !row?.box || !row.text.includes(chosen)) {
|
||||
hint.textContent = "Feldcode und wörtlichen Ausschnitt aus der gewählten Zeile angeben.";
|
||||
hint.className = "manual-hint error";return;
|
||||
}
|
||||
manual[engine].set(code.value, {code:code.value,value:chosen,box:row.box,manual:true});
|
||||
suppressed[engine].delete(code.value);
|
||||
renderCard(engine);
|
||||
});
|
||||
const clear = element("button", "Manuelle Zuordnungen löschen");clear.type = "button";
|
||||
clear.addEventListener("click", () => {manual[engine].clear();renderCard(engine);});
|
||||
form.append(code,lineSelect,value,add,clear,hint);card.append(form);
|
||||
}
|
||||
|
||||
run.addEventListener("click", async () => {
|
||||
const file = input.files?.[0];
|
||||
if (!file || !validCorners() || busy) return;
|
||||
if (file.size > 16 * 1048576) {setStatus(errorText.upload_too_large, "error");return;}
|
||||
const mime = file.type || (/\.jpe?g$/i.test(file.name) ? "image/jpeg" : /\.png$/i.test(file.name) ? "image/png" : "");
|
||||
if (!["image/jpeg", "image/png"].includes(mime)) {setStatus(errorText.unsupported_media_type, "error");return;}
|
||||
busy = true;updateControls();result = null;selectedEvidence = null;
|
||||
manual.classic.clear();manual.vision.clear();$("comparison").hidden = true;
|
||||
suppressed.classic.clear();suppressed.vision.clear();
|
||||
setStatus("Die lokalen Verfahren verarbeiten dasselbe Bild …");
|
||||
try {
|
||||
const geometry = {rotation:Number(rotation.value)};
|
||||
if (edited) geometry.corners = corners;
|
||||
const response = await fetch("/api/compare", {method:"POST",body:file,
|
||||
headers:{"Content-Type":mime,"X-OCR-Geometry":JSON.stringify(geometry)},cache:"no-store"});
|
||||
const data = await response.json();
|
||||
if (!response.ok) {setStatus(errorText[data.error] || `Vergleich fehlgeschlagen (${response.status}).`,"error");return;}
|
||||
result = data;processed.src = data.image;
|
||||
$("image-size").textContent = `Gemeinsames OCR-Bild: ${data.image_size.width} × ${data.image_size.height} Pixel. Klick auf Wert oder Rohtext markiert seinen Bildbeleg.`;
|
||||
renderCard("classic");renderCard("vision");$("comparison").hidden = false;
|
||||
const both = Object.values(data.results).every(branch => branch.status === "ok");
|
||||
setStatus(both ? "Beide Ergebnisse bereit. Feldvorschläge anhand der Bildbelege prüfen." :
|
||||
"Ein Verfahren ist fehlgeschlagen; das andere Ergebnis und der Fehler sind sichtbar.",both ? "ok" : "error");
|
||||
} catch {
|
||||
setStatus(result ? "Die Antwort des lokalen Vergleichs konnte nicht angezeigt werden. Seite neu laden." :
|
||||
"Der lokale Dienst antwortet nicht. Prozess und /api/health prüfen.","error");
|
||||
} finally {busy = false;updateControls();}
|
||||
});
|
||||
@@ -0,0 +1,48 @@
|
||||
"""PP-OCRv5 Latin inference using preloaded local model directories only."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
|
||||
ROOT = Path(os.environ.get("OCR_MODEL_HOME", Path(__file__).resolve().parents[1] / ".cache")).expanduser().resolve()
|
||||
MODELS = ROOT / "official_models"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
if len(sys.argv) != 3:
|
||||
raise SystemExit(2)
|
||||
image, output = map(Path, sys.argv[1:])
|
||||
detector = MODELS / "PP-OCRv5_mobile_det"
|
||||
recognizer = MODELS / "latin_PP-OCRv5_mobile_rec"
|
||||
if not (detector / "inference.pdiparams").is_file() or not (recognizer / "inference.pdiparams").is_file():
|
||||
raise SystemExit(3)
|
||||
from paddleocr import PaddleOCR
|
||||
|
||||
engine = PaddleOCR(
|
||||
text_detection_model_name="PP-OCRv5_mobile_det",
|
||||
text_detection_model_dir=str(detector),
|
||||
text_recognition_model_name="latin_PP-OCRv5_mobile_rec",
|
||||
text_recognition_model_dir=str(recognizer),
|
||||
use_doc_orientation_classify=False,
|
||||
use_doc_unwarping=False,
|
||||
use_textline_orientation=False,
|
||||
device="cpu",
|
||||
)
|
||||
results = list(engine.predict(str(image)))
|
||||
if len(results) != 1:
|
||||
raise RuntimeError("expected_one_page")
|
||||
raw = results[0].json["res"]
|
||||
lines = []
|
||||
for text, polygon in zip(raw["rec_texts"], raw["rec_polys"]):
|
||||
points = [[int(x), int(y)] for x, y in polygon]
|
||||
xs, ys = zip(*points)
|
||||
lines.append({"text": str(text), "bbox": [min(xs), min(ys), max(xs), max(ys)]})
|
||||
output.write_text(json.dumps({"lines": lines}, ensure_ascii=False))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,356 @@
|
||||
"""Assign observed OCR text to ZB Teil I fields with image evidence.
|
||||
|
||||
No value is generated. Weak code inferences from neighboring observed fields
|
||||
are explicit and remain reviewable in the UI.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from difflib import SequenceMatcher
|
||||
import re
|
||||
|
||||
|
||||
FIELD_CODES = tuple("""A C.1.1 C.1.2 C.1.3 X I B 2.1 2.2 J 4 E 3 D.1 D.2 D.3 2 5
|
||||
V.9 14 P.3 10 14.1 P.1 L 9 P.2/P.4 T 18 19 20 G 12 13 Q V.7 F.1 F.2
|
||||
7.1 7.2 7.3 8.1 8.2 8.3 U.1 U.2 U.3 O.1 O.2 S.1 S.2 15.1 15.2 15.3
|
||||
R 11 K 6 17 16 21 22""".split())
|
||||
|
||||
|
||||
def _norm(text: str) -> str:
|
||||
return re.sub(r"[^A-Z0-9]", "", text.upper())
|
||||
|
||||
|
||||
def _code(text: str) -> str | None:
|
||||
# Exact dotted labels win. Dot loss is accepted only when it is
|
||||
# unambiguous; arbitrary words and strings with values are not labels.
|
||||
literal = text.strip().upper().strip("() ")
|
||||
if literal in ("21", "22"):
|
||||
# Without the dot, 2.1/2.2 and 21/22 cannot be distinguished.
|
||||
return None
|
||||
if literal in FIELD_CODES:
|
||||
return literal
|
||||
token = _norm(text)
|
||||
matches = [code for code in FIELD_CODES if _norm(code) == token]
|
||||
return matches[0] if len(matches) == 1 else None
|
||||
|
||||
|
||||
_OWNER_CAPTIONS = {
|
||||
"C.1.1": re.compile(r"\b(?:name|firmenname)\b", re.I),
|
||||
"C.1.2": re.compile(r"\bvorname", re.I),
|
||||
"C.1.3": re.compile(r"\b(?:anschrift|adresse)\b", re.I),
|
||||
}
|
||||
_OWNER_PRINTED = {"C.1.1": "NAMEODERFIRMENNAME", "C.1.2": "VORNAMEN",
|
||||
"C.1.3": "ANSCHRIFT"}
|
||||
|
||||
|
||||
def _owner_caption(text: str) -> str | None:
|
||||
match = re.match(r"^\s*\(?\s*C\s*\.?\s*1\s*\.?\s*([123])\s*\)?\s*(.*)$", text, re.I)
|
||||
if match:
|
||||
code = f"C.1.{match[1]}"
|
||||
tail = match[2].strip(" \t:;=-")
|
||||
if not tail or _OWNER_CAPTIONS[code].search(tail):
|
||||
return code
|
||||
letters = re.sub(r"[^A-Z]", "", tail.upper())
|
||||
if len(letters) >= 6 and SequenceMatcher(None, letters, _OWNER_PRINTED[code]).ratio() >= .62:
|
||||
return code
|
||||
# Small printed codes can be lost while the full caption survives.
|
||||
# Use only the distinctive short printed wording, never a person's name.
|
||||
if len(text) < 55:
|
||||
if re.search(r"\bName\s+oder\s+Firmenname\b", text, re.I):
|
||||
return "C.1.1"
|
||||
if re.search(r"\bVorname\b", text, re.I):
|
||||
return "C.1.2"
|
||||
if re.search(r"\bAnschrift\b", text, re.I):
|
||||
return "C.1.3"
|
||||
return None
|
||||
|
||||
|
||||
def _height(box: list[float]) -> float:
|
||||
return box[3] - box[1]
|
||||
|
||||
|
||||
def _vertical_overlap(a: list[float], b: list[float]) -> float:
|
||||
return max(0, min(a[3], b[3]) - max(a[1], b[1]))
|
||||
|
||||
|
||||
def _nearby_value(anchor: dict, lines: list[dict], labels: set[int]) -> tuple[int, float] | None:
|
||||
ax1, ay1, ax2, ay2 = anchor["box"]
|
||||
ah = _height(anchor["box"])
|
||||
candidates = []
|
||||
for index, line in enumerate(lines):
|
||||
if (index in labels or line is anchor or not line.get("box") or
|
||||
not line["text"].strip(" \t-–—.,")):
|
||||
continue
|
||||
bx1, by1, bx2, by2 = line["box"]
|
||||
bh = _height(line["box"])
|
||||
if bx2 <= (ax1 + ax2) / 2 or bx1 < ax1 - 0.2 * ah:
|
||||
continue
|
||||
if bx1 - ax2 > max(120, 5 * ah):
|
||||
continue
|
||||
overlap = _vertical_overlap(anchor["box"], line["box"])
|
||||
if overlap < 0.45 * min(ah, bh):
|
||||
continue
|
||||
center_gap = abs((by1 + by2) / 2 - (ay1 + ay2) / 2)
|
||||
# Prefer the closest value to the right; reject rows that only touch
|
||||
# because large OCR boxes extend into a neighbouring printed row.
|
||||
candidates.append((max(0, bx1 - ax2) + center_gap * 1.5, index))
|
||||
if not candidates:
|
||||
return None
|
||||
candidates.sort()
|
||||
return candidates[0][1], candidates[0][0]
|
||||
|
||||
|
||||
def _owner_below(anchor_index: int, lines: list[dict], labels: set[int]) -> dict | None:
|
||||
"""Read holder value rows below an observed caption, within its column."""
|
||||
anchor = lines[anchor_index]
|
||||
code = _owner_caption(anchor["text"])
|
||||
if "Layoutblock" in anchor.get("source", ""):
|
||||
# The VL parser sometimes puts a printed caption and its value in
|
||||
# consecutive text lines of one coarse region. Their box is shared;
|
||||
# line order is the only honest evidence for this relation.
|
||||
same_region = []
|
||||
for index in range(anchor_index + 1, min(len(lines), anchor_index + (3 if code == "C.1.3" else 2))):
|
||||
line = lines[index]
|
||||
if index in labels or line.get("box") != anchor["box"] or not line["text"].strip():
|
||||
break
|
||||
same_region.append(index)
|
||||
if same_region:
|
||||
value = "\n".join(lines[index]["text"].strip() for index in same_region)
|
||||
return {"value": value[:500], "raw_text": value, "box": anchor["box"],
|
||||
"anchor_box": anchor["box"], "line_indices": [anchor_index, *same_region],
|
||||
"method": "caption_region_next_line", "confidence": None}
|
||||
ax1, _, _, ay2 = anchor["box"]
|
||||
ah = _height(anchor["box"])
|
||||
candidates = []
|
||||
for index, line in enumerate(lines):
|
||||
if index in labels or not line.get("box") or not line["text"].strip():
|
||||
continue
|
||||
x1, y1, x2, y2 = line["box"]
|
||||
if not (ay2 - .15 * ah <= y1 <= ay2 + max(85, 2.8 * ah)):
|
||||
continue
|
||||
if not (ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)):
|
||||
continue
|
||||
if x2 <= ax1 or re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I):
|
||||
continue
|
||||
candidates.append((max(0, y1 - ay2), abs(x1 - ax1), index))
|
||||
if not candidates:
|
||||
return None
|
||||
candidates.sort()
|
||||
first = lines[candidates[0][2]]["box"]
|
||||
first_indices = [index for index, line in enumerate(lines)
|
||||
if (index not in labels or line["text"].strip().isdigit())
|
||||
and line.get("box") and line["text"].strip()
|
||||
and line["box"][2] > ax1
|
||||
and _vertical_overlap(first, line["box"]) >=
|
||||
.35 * min(_height(first), _height(line["box"]))]
|
||||
first_indices.sort(key=lambda index: lines[index]["box"][0])
|
||||
# Only join fragments that touch the same printed value row. A remote
|
||||
# table column with a large gap is not part of the holder's name.
|
||||
row = [min((index for index in first_indices
|
||||
if ax1 - max(65, 1.5 * ah) <= lines[index]["box"][0]
|
||||
<= ax1 + max(170, 3 * ah)),
|
||||
key=lambda index: lines[index]["box"][0], default=candidates[0][2])]
|
||||
for index in first_indices:
|
||||
if index == row[0] or lines[index]["box"][0] < lines[row[0]]["box"][0]:
|
||||
continue
|
||||
previous = lines[row[-1]]["box"]
|
||||
box = lines[index]["box"]
|
||||
if box[0] - previous[2] <= max(45, .8 * _height(first)):
|
||||
row.append(index)
|
||||
rows = [row]
|
||||
if code == "C.1.3" and "Layoutblock" not in anchor.get("source", ""):
|
||||
bottom = max(lines[index]["box"][3] for index in row)
|
||||
row_height = max(_height(lines[index]["box"]) for index in row)
|
||||
next_row = []
|
||||
for index, line in enumerate(lines):
|
||||
if ((index in labels and not line["text"].strip().isdigit())
|
||||
or index in row or not line.get("box") or not line["text"].strip()):
|
||||
continue
|
||||
x1, y1, x2, _ = line["box"]
|
||||
if (bottom - .1 * row_height <= y1 <= bottom + max(25, .55 * row_height)
|
||||
and x1 >= ax1 - max(65, 1.5 * ah)
|
||||
and x2 > ax1
|
||||
and not re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I)):
|
||||
next_row.append(index)
|
||||
next_row.sort(key=lambda index: lines[index]["box"][0])
|
||||
if next_row:
|
||||
starter = next((index for index in next_row
|
||||
if lines[index]["box"][0] <= ax1 + max(170, 3 * ah)), None)
|
||||
contiguous = [starter] if starter is not None else []
|
||||
for index in next_row:
|
||||
if not contiguous or index == starter or lines[index]["box"][0] < lines[starter]["box"][0]:
|
||||
continue
|
||||
if lines[index]["box"][0] - lines[contiguous[-1]]["box"][2] <= max(45, .8 * row_height):
|
||||
contiguous.append(index)
|
||||
if contiguous:
|
||||
rows.append(contiguous)
|
||||
indices = [index for row in rows for index in row]
|
||||
value = "\n".join("\n".join(lines[index]["text"].strip() for index in row)
|
||||
if len(row) > 1 and all(lines[index]["box"] == lines[row[0]]["box"] for index in row)
|
||||
else " ".join(lines[index]["text"].strip() for index in row)
|
||||
for row in rows)
|
||||
boxes = [lines[index]["box"] for index in indices]
|
||||
box = [min(b[0] for b in boxes), min(b[1] for b in boxes),
|
||||
max(b[2] for b in boxes), max(b[3] for b in boxes)]
|
||||
return {"value": value[:500], "raw_text": value, "box": box,
|
||||
"anchor_box": anchor["box"], "line_indices": [anchor_index, *indices],
|
||||
"method": "below_caption", "confidence": None}
|
||||
|
||||
|
||||
def _surname_above_forename(lines: list[dict], owner_labels: dict[int, str],
|
||||
labels: set[int]) -> dict | None:
|
||||
"""Weak suggestion when C.1.1 is unreadable but C.1.2 is observed."""
|
||||
if "C.1.1" in owner_labels.values():
|
||||
return None
|
||||
forename = [index for index, code in owner_labels.items() if code == "C.1.2"]
|
||||
if len(forename) != 1:
|
||||
return None
|
||||
anchor_index = forename[0]
|
||||
anchor = lines[anchor_index]
|
||||
ax1, ay1, _, _ = anchor["box"]
|
||||
ah = _height(anchor["box"])
|
||||
candidates = []
|
||||
for index, line in enumerate(lines):
|
||||
if index in labels or not line.get("box"):
|
||||
continue
|
||||
text = line["text"].strip()
|
||||
if len(re.sub(r"[^A-ZÄÖÜ]", "", text.upper())) < 3 or any(c.isdigit() for c in text):
|
||||
continue
|
||||
x1, _, x2, y2 = line["box"]
|
||||
gap = ay1 - y2
|
||||
if (0 <= gap <= max(150, 3 * ah)
|
||||
and ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)
|
||||
and x2 > ax1):
|
||||
candidates.append((gap, abs(x1 - ax1), index))
|
||||
candidates.sort()
|
||||
if not candidates or (len(candidates) > 1 and candidates[1][0] - candidates[0][0] < ah):
|
||||
return None
|
||||
value_index = candidates[0][2]
|
||||
value_line = lines[value_index]
|
||||
return {"code": "C.1.1", "value": value_line["text"].strip()[:500],
|
||||
"raw_text": value_line["text"], "box": value_line["box"],
|
||||
"anchor_box": anchor["box"], "line_indices": [anchor_index, value_index],
|
||||
"method": "surname_above_forename", "confidence": None}
|
||||
|
||||
|
||||
def assign_fields(lines: list[dict], *, allow_adjacent: bool = True) -> list[dict]:
|
||||
"""Return only observed, localized values; duplicates stay in raw lines."""
|
||||
proposed: dict[str, list[dict]] = {}
|
||||
labels = {index: code for index, line in enumerate(lines)
|
||||
if line.get("box") and (code := _code(line["text"]))}
|
||||
owner_labels = {index: code for index, line in enumerate(lines)
|
||||
if line.get("box") and (code := _owner_caption(line["text"]))}
|
||||
label_indices = set(labels) | set(owner_labels)
|
||||
for index, line in enumerate(lines):
|
||||
box = line.get("box")
|
||||
if not box:
|
||||
continue
|
||||
if isinstance(line.get("cells"), list):
|
||||
cells = line["cells"]
|
||||
contextual = {}
|
||||
if (len(cells) >= 6 and _code(cells[0]) == "B"
|
||||
and _norm(cells[2]) == "21" and _norm(cells[4]) == "22"
|
||||
and _code(cells[2]) is None and _code(cells[4]) is None):
|
||||
# In one observed table row, the printed B / 2.1 / 2.2
|
||||
# sequence can lose both tiny dots. Keep this weaker reading
|
||||
# explicit and reviewable; never reinterpret a lone 21/22.
|
||||
contextual = {2: "2.1", 4: "2.2"}
|
||||
consumed_values: set[int] = set()
|
||||
for cell_index, cell in enumerate(cells[:-1]):
|
||||
if cell_index in consumed_values:
|
||||
continue
|
||||
code = (contextual.get(cell_index) or _code(cell)) if isinstance(cell, str) else None
|
||||
value = cells[cell_index + 1] if isinstance(cells[cell_index + 1], str) else ""
|
||||
value = value.strip()
|
||||
if (code and value and value not in ("-", "–", "—")
|
||||
and _code(value) is None
|
||||
and not (code in _OWNER_CAPTIONS and _OWNER_CAPTIONS[code].search(value))):
|
||||
consumed_values.add(cell_index + 1)
|
||||
proposed.setdefault(code, []).append({
|
||||
"code": code, "value": value[:500], "raw_text": line["text"],
|
||||
"box": box, "anchor_box": box, "line_indices": [index],
|
||||
"method": "vl_table_context_code" if cell_index in contextual else "vl_table_adjacent_cell",
|
||||
"confidence": None,
|
||||
})
|
||||
continue
|
||||
text = line["text"].strip()
|
||||
# Code and value in the same OCR line.
|
||||
matched = False
|
||||
for code in sorted(FIELD_CODES, key=len, reverse=True):
|
||||
if code.startswith("C.1."):
|
||||
# These printed captions precede the holder data on another row.
|
||||
continue
|
||||
printed = re.escape(code)
|
||||
marker = (rf"(?:\({printed}\)|{printed}(?=\s*[:;=\-]))"
|
||||
if len(code) == 1 or code.isdigit() else
|
||||
rf"(?:\({printed}\)|{printed}(?=\s|[:;=\-]))")
|
||||
match = re.match(rf"^\s*{marker}\s*[:;=\-]?\s*", text, flags=re.I)
|
||||
if match and (value := text[match.end():].strip(" \t:;=-")) and value.strip("-–—.,"):
|
||||
proposed.setdefault(code, []).append({
|
||||
"code": code, "value": value[:500], "raw_text": text,
|
||||
"box": box, "anchor_box": box, "line_indices": [index],
|
||||
"method": "same_line_code", "confidence": None,
|
||||
})
|
||||
matched = True
|
||||
break
|
||||
if not matched:
|
||||
# Small printed codes can merge with a value in one OCR box.
|
||||
# Keep the original box and mark this weaker reading for review.
|
||||
for code in sorted((c for c in FIELD_CODES if re.fullmatch(r"[A-Z]\.\d(?:\.\d)?", c)),
|
||||
key=len, reverse=True):
|
||||
marker = re.escape(code).replace(r"\.", r"\.?" )
|
||||
match = re.match(rf"^\s*{marker}(?=[A-Z0-9])", text, flags=re.I)
|
||||
if match and (value := text[match.end():].strip(" \t:;=-")):
|
||||
proposed.setdefault(code, []).append({
|
||||
"code": code, "value": value[:500], "raw_text": text,
|
||||
"box": box, "anchor_box": box, "line_indices": [index],
|
||||
"method": "merged_code", "confidence": None,
|
||||
})
|
||||
break
|
||||
for index, code in (labels.items() if allow_adjacent else []):
|
||||
if code in _OWNER_CAPTIONS:
|
||||
continue
|
||||
nearby = _nearby_value(lines[index], lines, label_indices)
|
||||
if nearby is None:
|
||||
continue
|
||||
value_index, distance = nearby
|
||||
value_line = lines[value_index]
|
||||
proposed.setdefault(code, []).append({
|
||||
"code": code, "value": value_line["text"].strip()[:500],
|
||||
"raw_text": value_line["text"], "box": value_line["box"],
|
||||
"anchor_box": lines[index]["box"], "line_indices": [index, value_index],
|
||||
"method": "adjacent_code", "confidence": None, "_distance": distance,
|
||||
})
|
||||
for index, code in owner_labels.items():
|
||||
below = _owner_below(index, lines, label_indices)
|
||||
if below:
|
||||
proposed.setdefault(code, []).append({"code": code, **below})
|
||||
if allow_adjacent and (weak_surname := _surname_above_forename(lines, owner_labels, label_indices)):
|
||||
proposed.setdefault("C.1.1", []).append(weak_surname)
|
||||
selected = {}
|
||||
for code in FIELD_CODES:
|
||||
unique = {}
|
||||
for entry in proposed.get(code, []):
|
||||
# Overlapping local image tiles may observe the same printed value
|
||||
# more than once. Identical readings are one proposal, even when
|
||||
# their honest layout regions have different sizes.
|
||||
unique.setdefault(_norm(entry["value"]), entry)
|
||||
if len(unique) == 1:
|
||||
selected[code] = next(iter(unique.values()))
|
||||
# One OCR value box cannot prove two different adjacent fields. Prefer a
|
||||
# clearly nearer printed code; otherwise leave both assignments open.
|
||||
by_value: dict[int, list[dict]] = {}
|
||||
for entry in selected.values():
|
||||
if entry["method"] == "adjacent_code":
|
||||
by_value.setdefault(entry["line_indices"][-1], []).append(entry)
|
||||
for collisions in by_value.values():
|
||||
if len(collisions) < 2:
|
||||
continue
|
||||
collisions.sort(key=lambda item: item["_distance"])
|
||||
keep = collisions[0] if collisions[0]["_distance"] < .7 * collisions[1]["_distance"] else None
|
||||
for entry in collisions:
|
||||
if entry is not keep:
|
||||
selected.pop(entry["code"], None)
|
||||
for entry in selected.values():
|
||||
entry.pop("_distance", None)
|
||||
return [selected[code] for code in FIELD_CODES if code in selected]
|
||||
@@ -0,0 +1,51 @@
|
||||
<!doctype html>
|
||||
<html lang="de">
|
||||
<head>
|
||||
<meta charset="utf-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<title>Fahrzeugschein · lokaler OCR-Vergleich</title>
|
||||
<link rel="stylesheet" href="/style.css">
|
||||
<script src="/app.js" defer></script>
|
||||
</head>
|
||||
<body>
|
||||
<header class="top"><div class="wrap">
|
||||
<p class="eyebrow">Lokaler Dokumenttest</p>
|
||||
<h1>Fahrzeugschein-OCR im Vergleich</h1>
|
||||
<p>Ein Bild, zwei unabhängige Verfahren. Werte erst anhand des Bildbelegs bestätigen.</p>
|
||||
</div></header>
|
||||
<main class="wrap">
|
||||
<section class="panel input-panel" aria-labelledby="input-title">
|
||||
<div class="section-head"><div><p class="eyebrow">01 / Eingabe</p><h2 id="input-title">Bild vorbereiten</h2></div>
|
||||
<span class="privacy">Lokale Verarbeitung; Vision optional auf eigener Hardware</span></div>
|
||||
<label class="upload-label" for="image-file">Fahrzeugschein auswählen</label>
|
||||
<input id="image-file" type="file" accept="image/jpeg,image/png,.jpg,.jpeg,.png">
|
||||
<p id="file-name" class="muted">JPEG oder PNG, maximal 16 MiB. Die Originaldatei wird nicht verändert.</p>
|
||||
<div id="original-wrap" class="image-wrap" hidden>
|
||||
<img id="original" alt="Ausgewähltes Originalbild mit veränderbaren Ecken">
|
||||
<canvas id="corners" aria-label="Vier Dokumentecken"></canvas>
|
||||
</div>
|
||||
<div class="toolbar">
|
||||
<button id="reset-corners" type="button" disabled>Ganzes Bild</button>
|
||||
<label for="rotation">Leserichtung</label>
|
||||
<select id="rotation"><option value="0">0°</option><option value="90">90° rechts</option>
|
||||
<option value="180">180°</option><option value="270">90° links</option></select>
|
||||
<button id="run" class="primary" type="button" disabled>Beide Verfahren vergleichen</button>
|
||||
</div>
|
||||
<p class="muted">Die vier blauen Punkte lassen sich mit Maus oder Finger an die äußeren Dokumentecken ziehen. Ohne Änderung läuft das ganze Bild. Beide Verfahren erhalten exakt dieselben aufbereiteten Pixel.</p>
|
||||
<p id="status" class="status" role="status" aria-live="polite"></p>
|
||||
</section>
|
||||
|
||||
<section id="comparison" class="panel" hidden aria-labelledby="compare-title">
|
||||
<div class="section-head"><div><p class="eyebrow">02 / Belege</p><h2 id="compare-title">Ergebnis am Bild prüfen</h2></div>
|
||||
<label class="toggle"><input id="all-boxes" type="checkbox"> Alle Textbereiche zeigen</label></div>
|
||||
<div class="processed-wrap"><img id="processed" alt="Aufbereitetes Bild für beide OCR-Verfahren">
|
||||
<canvas id="evidence" aria-label="Markierte Bildbelege"></canvas></div>
|
||||
<p id="image-size" class="muted"></p>
|
||||
<div class="results">
|
||||
<article id="classic-card" class="method" aria-labelledby="classic-title"><h3 id="classic-title">A · Klassische OCR</h3></article>
|
||||
<article id="vision-card" class="method" aria-labelledby="vision-title"><h3 id="vision-title">B · Vision-Dokumentmodell</h3></article>
|
||||
</div>
|
||||
</section>
|
||||
</main>
|
||||
</body>
|
||||
</html>
|
||||
@@ -0,0 +1,304 @@
|
||||
"""Local-only, privacy-preserving comparison of classic OCR and PaddleOCR-VL."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from io import BytesIO
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
import warnings
|
||||
|
||||
import cv2
|
||||
import numpy as np
|
||||
from PIL import Image, ImageOps, UnidentifiedImageError
|
||||
from starlette.applications import Starlette
|
||||
from starlette.requests import Request
|
||||
from starlette.responses import FileResponse, JSONResponse
|
||||
from starlette.routing import Route
|
||||
|
||||
from .fields import FIELD_CODES, assign_fields
|
||||
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
PROJECT = HERE.parent
|
||||
MODEL_HOME = Path(os.environ.get("OCR_MODEL_HOME", PROJECT / ".cache")).expanduser().resolve()
|
||||
|
||||
|
||||
def _remote_command(seconds: int = 180, *, tiles_only: bool = False) -> str:
|
||||
remote_dir = os.environ["OCR_VISION_REMOTE_DIR"]
|
||||
remote_python = os.environ.get("OCR_VISION_REMOTE_PYTHON", "./.venv-vision/bin/python")
|
||||
runner = f'{shlex.quote(remote_python)} -m ocr_compare.vision_worker'
|
||||
if tiles_only:
|
||||
runner += ' --tiles-only'
|
||||
isolation = 'unshare -Urn ' if os.environ.get("OCR_VISION_REMOTE_UNSHARE") == "1" else ""
|
||||
device = shlex.quote(os.environ.get("OCR_VISION_REMOTE_DEVICE", "gpu:0"))
|
||||
model_home = os.environ.get("OCR_VISION_REMOTE_MODEL_HOME")
|
||||
model_env = f'OCR_MODEL_HOME={shlex.quote(model_home)} ' if model_home else ""
|
||||
return (
|
||||
f'set -eu; cd {shlex.quote(remote_dir)}; directory=$(mktemp -d); '
|
||||
'trap \'rm -rf -- "$directory"\' EXIT; '
|
||||
f'OCR_PRIVATE_DIR="$directory" OCR_VISION_DEVICE={device} {model_env}'
|
||||
f'timeout -k 5s {seconds}s {isolation}{runner}'
|
||||
)
|
||||
|
||||
|
||||
MAX_UPLOAD = 16 * 1024 * 1024
|
||||
MAX_PIXELS = 24_000_000
|
||||
MAX_SIDE = 10_000
|
||||
OCR_SIDE = 4000
|
||||
Image.MAX_IMAGE_PIXELS = MAX_PIXELS
|
||||
_slot = asyncio.Semaphore(1)
|
||||
|
||||
|
||||
class InputError(ValueError):
|
||||
pass
|
||||
|
||||
|
||||
def _geometry(image: Image.Image, header: str | None) -> Image.Image:
|
||||
try:
|
||||
spec = json.loads(header) if header else {}
|
||||
rotation = int(spec.get("rotation", 0))
|
||||
if rotation not in (0, 90, 180, 270):
|
||||
raise InputError("invalid_rotation")
|
||||
points = spec.get("corners")
|
||||
if points is not None:
|
||||
if (not isinstance(points, list) or len(points) != 4 or
|
||||
any(not isinstance(p, list) or len(p) != 2 or
|
||||
any(not isinstance(v, (int, float)) or isinstance(v, bool) or
|
||||
not 0 <= v <= 1 for v in p) for p in points)):
|
||||
raise InputError("invalid_corners")
|
||||
w, h = image.size
|
||||
source = np.array([[x * (w - 1), y * (h - 1)] for x, y in points], np.float32)
|
||||
area = cv2.contourArea(source)
|
||||
edges = [float(np.linalg.norm(source[(i + 1) % 4] - source[i])) for i in range(4)]
|
||||
turns = []
|
||||
for i in range(4):
|
||||
ax, ay = source[(i + 1) % 4] - source[i]
|
||||
bx, by = source[(i + 2) % 4] - source[(i + 1) % 4]
|
||||
turns.append(float(ax * by - ay * bx))
|
||||
if area < .02 * w * h or min(edges) < 30 or min(turns) <= 0:
|
||||
raise InputError("invalid_corners")
|
||||
out_w = round(max(edges[0], edges[2]))
|
||||
out_h = round(max(edges[1], edges[3]))
|
||||
if out_w < 30 or out_h < 30 or out_w * out_h > MAX_PIXELS:
|
||||
raise InputError("invalid_corners")
|
||||
# The four positions are always TL, TR, BR, BL. This avoids the
|
||||
# previous fixed-template aspect-ratio and rotation ambiguity.
|
||||
target = np.array([[0, 0], [out_w - 1, 0],
|
||||
[out_w - 1, out_h - 1], [0, out_h - 1]], np.float32)
|
||||
matrix = cv2.getPerspectiveTransform(source, target)
|
||||
warped = cv2.warpPerspective(np.asarray(image), matrix, (out_w, out_h),
|
||||
flags=cv2.INTER_LINEAR,
|
||||
borderMode=cv2.BORDER_CONSTANT,
|
||||
borderValue=(255, 255, 255))
|
||||
image = Image.fromarray(warped)
|
||||
if rotation:
|
||||
image = image.rotate(-rotation, expand=True)
|
||||
if max(image.size) > OCR_SIDE:
|
||||
image.thumbnail((OCR_SIDE, OCR_SIDE), Image.Resampling.LANCZOS)
|
||||
return image
|
||||
except (ValueError, TypeError, KeyError, json.JSONDecodeError) as exc:
|
||||
if isinstance(exc, InputError):
|
||||
raise
|
||||
raise InputError("invalid_corners") from None
|
||||
|
||||
|
||||
def _prepare(data: bytes, header: str | None) -> tuple[bytes, tuple[int, int]]:
|
||||
try:
|
||||
with warnings.catch_warnings():
|
||||
warnings.simplefilter("error", Image.DecompressionBombWarning)
|
||||
with Image.open(BytesIO(data)) as source:
|
||||
if source.format not in ("JPEG", "PNG", "MPO"):
|
||||
raise InputError("unsupported_encoding")
|
||||
w, h = source.size
|
||||
if min(w, h) < 1 or max(w, h) > MAX_SIDE or w * h > MAX_PIXELS:
|
||||
raise InputError("image_too_large")
|
||||
source.load()
|
||||
image = ImageOps.exif_transpose(source).convert("RGB")
|
||||
image = _geometry(image, header)
|
||||
output = BytesIO()
|
||||
image.save(output, format="JPEG", quality=92, subsampling=0)
|
||||
return output.getvalue(), image.size
|
||||
except InputError:
|
||||
raise
|
||||
except (OSError, ValueError, UnidentifiedImageError, Image.DecompressionBombError,
|
||||
Image.DecompressionBombWarning):
|
||||
raise InputError("invalid_image") from None
|
||||
|
||||
|
||||
def _classic(data: bytes, directory: Path) -> dict:
|
||||
started = time.monotonic()
|
||||
models = MODEL_HOME / "official_models"
|
||||
if any(not (models / name / "inference.pdiparams").is_file() for name in
|
||||
("PP-OCRv5_mobile_det", "latin_PP-OCRv5_mobile_rec")):
|
||||
return {"status": "error", "error": "classic_models_missing", "elapsed_ms": 0}
|
||||
image_path = directory / "same-input.jpg"
|
||||
output_path = directory / "classic.json"
|
||||
image_path.write_bytes(data)
|
||||
env = os.environ.copy()
|
||||
env.update(PADDLE_PDX_CACHE_HOME=str(MODEL_HOME),
|
||||
HF_HOME=str(MODEL_HOME / "hf"), HF_HUB_OFFLINE="1",
|
||||
TRANSFORMERS_OFFLINE="1", PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK="True",
|
||||
OMP_NUM_THREADS="1", OPENBLAS_NUM_THREADS="1", OMP_THREAD_LIMIT="1")
|
||||
try:
|
||||
process = subprocess.run(
|
||||
[sys.executable, "-m", "ocr_compare.classic_worker", str(image_path), str(output_path)],
|
||||
cwd=PROJECT, env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
timeout=120, check=False,
|
||||
)
|
||||
except subprocess.TimeoutExpired:
|
||||
return {"status": "error", "error": "classic_timeout", "elapsed_ms": round((time.monotonic()-started)*1000)}
|
||||
except OSError:
|
||||
return {"status": "error", "error": "classic_failed", "elapsed_ms": round((time.monotonic()-started)*1000)}
|
||||
elapsed = round((time.monotonic() - started) * 1000)
|
||||
if process.returncode or not output_path.is_file():
|
||||
return {"status": "error", "error": "classic_failed", "elapsed_ms": elapsed}
|
||||
try:
|
||||
raw = json.loads(output_path.read_text())
|
||||
lines = [{"text": row["text"], "box": row["bbox"], "source": "OCR-Zeile"}
|
||||
for row in raw["lines"] if row.get("bbox")]
|
||||
except (OSError, ValueError, KeyError, TypeError):
|
||||
return {"status": "error", "error": "classic_invalid", "elapsed_ms": elapsed}
|
||||
return {"status": "ok", "model": "PP-OCRv5 mobile (Latin), lokal",
|
||||
"elapsed_ms": elapsed, "lines": lines, "fields": assign_fields(lines),
|
||||
"regions": []}
|
||||
|
||||
|
||||
def _vision(data: bytes) -> dict:
|
||||
started = time.monotonic()
|
||||
mode = os.environ.get("OCR_VISION_MODE", "local").lower()
|
||||
if mode == "off":
|
||||
return {"status": "error", "error": "vision_disabled", "elapsed_ms": 0}
|
||||
if mode not in ("local", "ssh"):
|
||||
return {"status": "error", "error": "vision_config_error", "elapsed_ms": 0}
|
||||
if mode == "local" and any(not path.is_file() for path in (
|
||||
MODEL_HOME / "official_models" / "PP-DocLayoutV3" / "inference.pdiparams",
|
||||
MODEL_HOME / "official_models" / "PaddleOCR-VL-1.6" / "model.safetensors")):
|
||||
return {"status": "error", "error": "vision_models_missing", "elapsed_ms": 0}
|
||||
if mode == "ssh" and (not os.environ.get("OCR_VISION_SSH_TARGET") or
|
||||
not os.environ.get("OCR_VISION_REMOTE_DIR")):
|
||||
return {"status": "error", "error": "vision_config_error", "elapsed_ms": 0}
|
||||
with Image.open(BytesIO(data)) as page:
|
||||
wide = page.width > 2500 and page.width / page.height > 1.65
|
||||
limit = 90 if wide else 180
|
||||
|
||||
def run_worker(seconds: int, *, tiles_only: bool = False):
|
||||
if mode == "ssh":
|
||||
command = ["ssh", "-T", "-o", "BatchMode=yes", "-o", "StrictHostKeyChecking=yes",
|
||||
"-o", "ConnectTimeout=10"]
|
||||
identity = os.environ.get("OCR_VISION_SSH_IDENTITY")
|
||||
if identity:
|
||||
command += ["-i", str(Path(identity).expanduser()), "-o", "IdentitiesOnly=yes"]
|
||||
command += [os.environ["OCR_VISION_SSH_TARGET"],
|
||||
_remote_command(seconds, tiles_only=tiles_only)]
|
||||
env = None
|
||||
else:
|
||||
command = [os.environ.get("OCR_VISION_PYTHON", sys.executable),
|
||||
"-m", "ocr_compare.vision_worker"]
|
||||
if tiles_only:
|
||||
command.append("--tiles-only")
|
||||
env = os.environ.copy()
|
||||
env.update(OCR_MODEL_HOME=str(MODEL_HOME),
|
||||
OCR_VISION_DEVICE=os.environ.get("OCR_VISION_DEVICE", "cpu"))
|
||||
try:
|
||||
with tempfile.TemporaryDirectory(prefix="ocr-vision-") as private:
|
||||
if env is not None:
|
||||
env["OCR_PRIVATE_DIR"] = private
|
||||
return subprocess.run(command, input=data, stdout=subprocess.PIPE,
|
||||
stderr=subprocess.DEVNULL, timeout=seconds + 15,
|
||||
check=False, cwd=PROJECT if mode == "local" else None, env=env)
|
||||
except subprocess.TimeoutExpired:
|
||||
return None
|
||||
except OSError:
|
||||
return subprocess.CompletedProcess(command, 127, b"")
|
||||
|
||||
process = run_worker(limit)
|
||||
if wide and (process is None or process.returncode == 124):
|
||||
# Wide documents can form one prohibitively expensive layout block.
|
||||
# The three generic views cover the same uploaded page without a
|
||||
# document-specific field template.
|
||||
process = run_worker(90, tiles_only=True)
|
||||
elapsed = round((time.monotonic() - started) * 1000)
|
||||
if process is None or process.returncode == 124:
|
||||
return {"status": "error", "error": "vision_timeout", "elapsed_ms": elapsed}
|
||||
if process.returncode or len(process.stdout) > 2_000_000:
|
||||
return {"status": "error", "error": "vision_failed", "elapsed_ms": elapsed}
|
||||
try:
|
||||
raw = json.loads(process.stdout)
|
||||
if raw.get("error"):
|
||||
code = "vision_models_missing" if raw["error"] == "models_missing" else "vision_failed"
|
||||
return {"status": "error", "error": code, "elapsed_ms": elapsed}
|
||||
lines = raw["lines"]
|
||||
regions = raw["regions"]
|
||||
if not isinstance(lines, list) or not isinstance(regions, list):
|
||||
raise ValueError("invalid collections")
|
||||
except (ValueError, TypeError, KeyError):
|
||||
return {"status": "error", "error": "vision_invalid", "elapsed_ms": elapsed}
|
||||
return {"status": "ok", "model": raw.get("model", "PaddleOCR-VL"),
|
||||
"elapsed_ms": elapsed, "lines": lines,
|
||||
"fields": assign_fields(lines, allow_adjacent=False),
|
||||
"regions": regions, "layout_note": raw.get("layout_note", ""),
|
||||
"retry_used": bool(raw.get("retry_used", False))}
|
||||
|
||||
|
||||
async def index(_: Request):
|
||||
return FileResponse(HERE / "index.html", media_type="text/html; charset=utf-8")
|
||||
|
||||
|
||||
async def script(_: Request):
|
||||
return FileResponse(HERE / "app.js", media_type="text/javascript; charset=utf-8")
|
||||
|
||||
|
||||
async def style(_: Request):
|
||||
return FileResponse(HERE / "style.css", media_type="text/css; charset=utf-8")
|
||||
|
||||
|
||||
async def health(_: Request):
|
||||
return JSONResponse({"ready": True, "scope": "localhost only",
|
||||
"vision_mode": os.environ.get("OCR_VISION_MODE", "local")})
|
||||
|
||||
|
||||
async def compare(request: Request):
|
||||
if _slot.locked():
|
||||
return JSONResponse({"error": "busy"}, status_code=429)
|
||||
if request.headers.get("content-type", "").split(";", 1)[0] not in ("image/jpeg", "image/png"):
|
||||
return JSONResponse({"error": "unsupported_media_type"}, status_code=415)
|
||||
async with _slot:
|
||||
data = bytearray()
|
||||
try:
|
||||
async with asyncio.timeout(30):
|
||||
async for chunk in request.stream():
|
||||
data.extend(chunk)
|
||||
if len(data) > MAX_UPLOAD:
|
||||
return JSONResponse({"error": "upload_too_large"}, status_code=413)
|
||||
except TimeoutError:
|
||||
return JSONResponse({"error": "upload_timeout"}, status_code=408)
|
||||
try:
|
||||
image, (width, height) = await asyncio.to_thread(
|
||||
_prepare, bytes(data), request.headers.get("x-ocr-geometry"))
|
||||
except InputError as exc:
|
||||
return JSONResponse({"error": str(exc)}, status_code=422)
|
||||
with tempfile.TemporaryDirectory(prefix="ocr-compare-") as path:
|
||||
directory = Path(path)
|
||||
classic, vision = await asyncio.gather(
|
||||
asyncio.to_thread(_classic, image, directory),
|
||||
asyncio.to_thread(_vision, image),
|
||||
)
|
||||
# Exact JPEG bytes sent to both pipelines are returned for evidence
|
||||
# highlighting. They never enter a report, log, or external OCR API.
|
||||
import base64
|
||||
return JSONResponse({"image": "data:image/jpeg;base64," + base64.b64encode(image).decode(),
|
||||
"image_size": {"width": width, "height": height},
|
||||
"field_codes": FIELD_CODES,
|
||||
"results": {"classic": classic, "vision": vision}})
|
||||
|
||||
|
||||
app = Starlette(routes=[Route("/", index), Route("/app.js", script),
|
||||
Route("/style.css", style), Route("/api/health", health),
|
||||
Route("/api/compare", compare, methods=["POST"])])
|
||||
@@ -0,0 +1,71 @@
|
||||
"""Download official model weights and verify them on a synthetic page."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
from pathlib import Path
|
||||
import tempfile
|
||||
|
||||
from PIL import Image, ImageDraw
|
||||
|
||||
|
||||
ROOT = Path(os.environ.get("OCR_MODEL_HOME", Path(__file__).resolve().parents[1] / ".cache")).expanduser().resolve()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--classic", action="store_true")
|
||||
parser.add_argument("--vision", action="store_true")
|
||||
parser.add_argument("--device", default="cpu", help="Vision device, for example cpu or gpu:0")
|
||||
args = parser.parse_args()
|
||||
if not args.classic and not args.vision:
|
||||
parser.error("choose --classic and/or --vision")
|
||||
|
||||
ROOT.mkdir(parents=True, exist_ok=True)
|
||||
os.environ["PADDLE_PDX_CACHE_HOME"] = str(ROOT)
|
||||
os.environ["HF_HOME"] = str(ROOT / "hf")
|
||||
os.environ.pop("HF_HUB_OFFLINE", None)
|
||||
os.environ.pop("TRANSFORMERS_OFFLINE", None)
|
||||
with tempfile.TemporaryDirectory(prefix="ocr-synthetic-setup-") as temporary:
|
||||
image = Path(temporary) / "synthetic.png"
|
||||
page = Image.new("RGB", (1000, 600), "white")
|
||||
draw = ImageDraw.Draw(page)
|
||||
draw.text((50, 50), "SYNTHETIC TEST / NO PERSONAL DATA", fill="black")
|
||||
draw.text((50, 130), "D.3 SAMPLE 123", fill="black")
|
||||
page.save(image)
|
||||
if args.classic:
|
||||
from paddleocr import PaddleOCR
|
||||
engine = PaddleOCR(
|
||||
text_detection_model_name="PP-OCRv5_mobile_det",
|
||||
text_recognition_model_name="latin_PP-OCRv5_mobile_rec",
|
||||
use_doc_orientation_classify=False,
|
||||
use_doc_unwarping=False,
|
||||
use_textline_orientation=False,
|
||||
device="cpu",
|
||||
)
|
||||
list(engine.predict(str(image)))
|
||||
if args.vision:
|
||||
from paddleocr import PaddleOCRVL
|
||||
engine = PaddleOCRVL(
|
||||
pipeline_version="v1.6", use_layout_detection=True,
|
||||
use_doc_orientation_classify=False, use_doc_unwarping=False,
|
||||
device=args.device,
|
||||
)
|
||||
list(engine.predict(str(image)))
|
||||
|
||||
models = ROOT / "official_models"
|
||||
checks = []
|
||||
if args.classic:
|
||||
checks.extend((models / name / "inference.pdiparams" for name in
|
||||
("PP-OCRv5_mobile_det", "latin_PP-OCRv5_mobile_rec")))
|
||||
if args.vision:
|
||||
checks.extend((models / "PP-DocLayoutV3" / "inference.pdiparams",
|
||||
models / "PaddleOCR-VL-1.6" / "model.safetensors"))
|
||||
if any(not path.is_file() for path in checks):
|
||||
raise SystemExit("Official model download did not create all expected files")
|
||||
print("Official models ready; only synthetic pixels were used")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,72 @@
|
||||
:root {font-family:system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;color:#16233a;background:#f3f6f8;line-height:1.45}
|
||||
* {box-sizing:border-box}
|
||||
body {margin:0}
|
||||
.wrap {width:min(1240px,calc(100% - 32px));margin:auto}
|
||||
.top {background:#133345;color:white;padding:24px 0 28px}
|
||||
h1,h2,h3,h4,p {margin:0}
|
||||
h1 {font-size:clamp(1.6rem,3vw,2.5rem);letter-spacing:-.035em}
|
||||
h2 {font-size:1.35rem;letter-spacing:-.025em}
|
||||
h3 {font-size:1.1rem}
|
||||
h4 {font-size:.95rem;margin:22px 0 8px}
|
||||
.top p:last-child {color:#d6e6e9;margin-top:5px}
|
||||
.eyebrow {font-size:.72rem;text-transform:uppercase;font-weight:800;letter-spacing:.13em;color:#238476;margin-bottom:5px}
|
||||
.top .eyebrow {color:#8be0c8}
|
||||
main {padding:22px 0 50px}
|
||||
.panel {background:white;border:1px solid #dbe5e9;border-radius:14px;padding:22px;margin-bottom:20px;box-shadow:0 6px 30px rgba(20,46,64,.045)}
|
||||
.section-head {display:flex;align-items:start;justify-content:space-between;gap:18px;margin-bottom:18px}
|
||||
.privacy {font-size:.76rem;font-weight:700;color:#177868;background:#e9f6f1;border-radius:100px;padding:6px 11px;white-space:nowrap}
|
||||
.upload-label {display:block;font-weight:700;margin-bottom:6px}
|
||||
input[type=file] {display:block;width:100%;padding:12px;border:1px dashed #9eb5c0;background:#f8fbfc;border-radius:9px;color:#20384a}
|
||||
.muted {color:#617585;font-size:.87rem;margin-top:9px}
|
||||
.image-wrap,.processed-wrap {position:relative;display:block;width:100%;margin-top:18px;background:#e9eff1;border:1px solid #ccd9df;border-radius:8px;overflow:hidden}
|
||||
.image-wrap[hidden],#comparison[hidden] {display:none}
|
||||
.image-wrap img,.processed-wrap img {display:block;width:100%;height:auto}
|
||||
.image-wrap canvas,.processed-wrap canvas {position:absolute;inset:0;width:100%;height:100%}
|
||||
.image-wrap canvas {touch-action:none;cursor:crosshair}
|
||||
.processed-wrap canvas {pointer-events:none}
|
||||
.toolbar {display:flex;align-items:center;gap:10px;flex-wrap:wrap;margin-top:15px}
|
||||
.toolbar label {font-size:.88rem;font-weight:700;margin-left:auto}
|
||||
button,select,input[type=text] {font:inherit}
|
||||
button,select {border-radius:8px;border:1px solid #9bb4bf;background:white;color:#193949;padding:9px 12px}
|
||||
button {cursor:pointer;font-weight:650}
|
||||
button:hover:not(:disabled) {background:#e8f2f3}
|
||||
button:disabled {cursor:not-allowed;opacity:.52}
|
||||
.primary {background:#007d73;color:white;border-color:#007d73}
|
||||
.primary:hover:not(:disabled) {background:#00695f}
|
||||
.status {min-height:1.4em;margin-top:12px;font-weight:650}
|
||||
.status.error,.error {color:#ac273c}
|
||||
.status.ok {color:#08786d}
|
||||
.toggle {font-size:.85rem;color:#3d5667;white-space:nowrap}
|
||||
.results {display:grid;grid-template-columns:repeat(2,minmax(0,1fr));gap:16px;margin-top:20px}
|
||||
.method {border:1px solid #d4e2e6;border-radius:11px;padding:17px;min-width:0}
|
||||
.method:first-child {border-top:4px solid #087c75}
|
||||
.method:last-child {border-top:4px solid #465fc7}
|
||||
.meta {font-size:.84rem;color:#526d7b;margin-top:5px}
|
||||
.note {font-size:.82rem;color:#815b13;background:#fff8e5;padding:9px;border-radius:6px;margin-top:10px}
|
||||
.assignment-note {font-size:.82rem;color:#385469;background:#eef5f8;padding:9px;border-radius:6px;margin-top:10px}
|
||||
.field-list {display:grid;gap:7px}
|
||||
.field-item {display:grid;grid-template-columns:minmax(0,1fr) auto;gap:6px;padding:3px;background:#f8fbfc;border:1px solid #dce7ea;border-radius:8px;overflow-wrap:anywhere}
|
||||
.field-item.needs-review {background:#fff9eb;border-color:#e4c886}
|
||||
.field-show {display:grid;grid-template-columns:minmax(88px,110px) minmax(0,1fr);gap:2px 8px;width:100%;text-align:left;border:0;background:transparent;padding:6px 7px}
|
||||
.field-heading {grid-row:1 / 3;display:flex;flex-direction:column;gap:2px}
|
||||
.field-heading strong {color:#126d65}
|
||||
.field-heading small,.field-method {font-size:.73rem;color:#607783}
|
||||
.field-value {font-weight:650;overflow-wrap:anywhere;white-space:pre-line}
|
||||
.priority-fields {display:flex;flex-wrap:wrap;gap:6px;margin:8px 0 12px}
|
||||
.priority-fields h5 {flex-basis:100%;margin:0;font-size:.9rem}
|
||||
.priority-item {min-width:0;padding:3px 7px;border-radius:6px;background:#e8f6ef;color:#155d3d;font-size:.83rem;overflow-wrap:anywhere}
|
||||
.priority-item.missing {background:#fff2e9;color:#8a431d}
|
||||
.field-method {grid-column:2}
|
||||
.field-dismiss {align-self:center;font-size:.73rem;padding:6px 7px;color:#6d4050;border-color:#d9c9ce;background:white}
|
||||
.raw-lines {max-height:280px;overflow:auto;border:1px solid #e3ecef;border-radius:8px;padding:6px;display:grid;gap:4px}
|
||||
.raw-entry {display:grid;grid-template-columns:minmax(0,1fr) auto;gap:4px}
|
||||
.raw-line {display:flex;align-items:start;text-align:left;gap:9px;width:100%;padding:8px 10px;background:#f8fbfc;border-color:#dce7ea;overflow-wrap:anywhere;font-weight:400;font-size:.82rem}
|
||||
.raw-line span {color:#617783;min-width:28px}
|
||||
.raw-line span:last-child {color:#193949;min-width:0}
|
||||
.raw-pick {font-size:.72rem;padding:5px 7px}
|
||||
.manual {display:grid;grid-template-columns:minmax(100px,1fr) minmax(100px,2fr);gap:7px}
|
||||
.manual input {grid-column:1 / -1;border:1px solid #9bb4bf;border-radius:8px;padding:9px;width:100%}
|
||||
.manual button {justify-self:start}
|
||||
.manual-hint {grid-column:1 / -1;color:#617585;font-size:.78rem}
|
||||
@media(max-width:800px) {.results {grid-template-columns:1fr}.privacy {white-space:normal}.toolbar label {margin-left:0}}
|
||||
@media(max-width:520px) {.wrap {width:min(100% - 20px,1240px)}.panel {padding:14px}.section-head {display:block}.privacy {display:inline-block;margin-top:10px}.toolbar {display:grid;grid-template-columns:1fr 1fr}.toolbar .primary {grid-column:1/-1}.toolbar label {align-self:center}.toggle {display:block;margin-top:10px}.field-show {grid-template-columns:1fr}.field-heading {grid-row:auto}.field-method {grid-column:1}.raw-entry {grid-template-columns:minmax(0,1fr)}.raw-pick {justify-self:end}}
|
||||
@@ -0,0 +1,310 @@
|
||||
"""Offline worker for the full official PaddleOCR-VL layout pipeline.
|
||||
|
||||
The caller sends JPEG bytes on stdin. This process uses explicit local model
|
||||
directories; it writes a mode-0600 temporary file and deletes it afterward.
|
||||
No private images or OCR results are saved to the model cache or logs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
from html.parser import HTMLParser
|
||||
from pathlib import Path
|
||||
import re
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
from PIL import Image
|
||||
|
||||
|
||||
ROOT = Path(os.environ.get("OCR_MODEL_HOME", Path(__file__).resolve().parents[1] / ".cache")).expanduser().resolve()
|
||||
os.environ["PADDLE_PDX_CACHE_HOME"] = str(ROOT)
|
||||
os.environ["HF_HOME"] = str(ROOT / "hf")
|
||||
os.environ["HF_HUB_OFFLINE"] = "1"
|
||||
os.environ["TRANSFORMERS_OFFLINE"] = "1"
|
||||
os.environ["PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK"] = "True"
|
||||
os.environ["OMP_NUM_THREADS"] = "1"
|
||||
|
||||
|
||||
class _TableRows(HTMLParser):
|
||||
def __init__(self):
|
||||
super().__init__()
|
||||
self.rows = []
|
||||
self.row = None
|
||||
self.cell = None
|
||||
|
||||
def handle_starttag(self, tag, attrs):
|
||||
if tag == "tr":
|
||||
self.row = []
|
||||
elif tag in ("td", "th") and self.row is not None:
|
||||
self.cell = []
|
||||
elif tag == "br" and self.cell is not None:
|
||||
self.cell.append(" ")
|
||||
|
||||
def handle_data(self, data):
|
||||
if self.cell is not None:
|
||||
self.cell.append(data)
|
||||
|
||||
def handle_endtag(self, tag):
|
||||
if tag in ("td", "th") and self.cell is not None and self.row is not None:
|
||||
self.row.append(" ".join("".join(self.cell).split())[:500])
|
||||
self.cell = None
|
||||
elif tag == "tr" and self.row is not None:
|
||||
if self.row:
|
||||
self.rows.append(self.row)
|
||||
self.row = None
|
||||
|
||||
|
||||
def _table_rows(content: str) -> list[list[str]]:
|
||||
parser = _TableRows()
|
||||
parser.feed(content)
|
||||
parser.close()
|
||||
return parser.rows
|
||||
|
||||
|
||||
def _box(value, width: int, height: int, offset=(0, 0)):
|
||||
if value is None or len(value) != 4:
|
||||
return None
|
||||
try:
|
||||
x1, y1, x2, y2 = (float(item) for item in value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if not (0 <= x1 < x2 <= width and 0 <= y1 < y2 <= height):
|
||||
return None
|
||||
ox, oy = offset
|
||||
return [round(x1 + ox, 1), round(y1 + oy, 1),
|
||||
round(x2 + ox, 1), round(y2 + oy, 1)]
|
||||
|
||||
|
||||
def extract(payload: dict, offset=(0, 0)) -> dict:
|
||||
"""Keep original block text and its actual layout region, never fake line boxes."""
|
||||
width, height = int(payload["width"]), int(payload["height"])
|
||||
lines, regions = [], []
|
||||
for block in payload.get("parsing_res_list", []):
|
||||
box = _box(block.get("block_bbox"), width, height, offset)
|
||||
label = str(block.get("block_label", "region"))[:80]
|
||||
content = block.get("block_content", "")
|
||||
if not isinstance(content, str):
|
||||
content = str(content)
|
||||
if box:
|
||||
regions.append({"label": label, "box": box})
|
||||
if label == "table" and "<tr" in content.lower():
|
||||
for cells in _table_rows(content):
|
||||
text = " | ".join(cells)
|
||||
if text.strip():
|
||||
lines.append({"text": text[:1000], "cells": cells,
|
||||
"box": box, "source": "PaddleOCR-VL-Tabellenzeile; nur Tabellenregion als Bildbeleg"})
|
||||
else:
|
||||
for part in content.splitlines():
|
||||
text = part.strip()
|
||||
if text:
|
||||
lines.append({"text": text[:1000], "box": box,
|
||||
"source": f"PaddleOCR-VL-Layoutblock {label}; keine eigene Zeilenbox"})
|
||||
return {
|
||||
"model": "PaddleOCR-VL 1.6 · vollständige lokale Layoutpipeline",
|
||||
"lines": lines[:500], "regions": regions[:200],
|
||||
"layout_note": "Vision-Bildbelege markieren echte Layoutregionen. Mehrere Textzeilen können dieselbe Region teilen; die Region ist keine präzise Feldbox.",
|
||||
}
|
||||
|
||||
|
||||
def _retry_table(block: dict, width: int, height: int) -> bool:
|
||||
if block.get("block_label") != "table":
|
||||
return False
|
||||
box = _box(block.get("block_bbox"), width, height)
|
||||
if not box:
|
||||
return False
|
||||
x1, y1, x2, y2 = box
|
||||
area = (x2 - x1) * (y2 - y1) / (width * height)
|
||||
content = block.get("block_content", "")
|
||||
if not isinstance(content, str):
|
||||
return False
|
||||
visible_text = re.sub(r"<[^>]*>", "", content).strip()
|
||||
return area > .45 and len(visible_text) < 80
|
||||
|
||||
|
||||
def _split_table(box: list[float], width: int, height: int) -> list[tuple[int, int, int, int]]:
|
||||
x1, y1, x2, y2 = box
|
||||
left = max(0, int(x1) - 20)
|
||||
top = max(0, int(y1) - 20)
|
||||
right = min(width, int(x2) + 20)
|
||||
bottom = min(height, int(y2) + 20)
|
||||
middle = (left + right) // 2
|
||||
overlap = min(120, max(40, round((right - left) * .045)))
|
||||
return [(left, top, min(right, middle + overlap), bottom),
|
||||
(max(left, middle - overlap), top, right, bottom)]
|
||||
|
||||
|
||||
def _missing_neighbor_box(block: dict, blocks: list[dict], width: int, height: int):
|
||||
"""Find an uncovered page column directly left of a detected right table."""
|
||||
if block.get("block_label") != "table":
|
||||
return None
|
||||
box = _box(block.get("block_bbox"), width, height)
|
||||
if not box:
|
||||
return None
|
||||
x1, y1, x2, y2 = box
|
||||
table_width, table_height = x2 - x1, y2 - y1
|
||||
if x1 < .52 * width or table_width > .4 * width or table_height < .2 * height:
|
||||
return None
|
||||
left = max(0, round(x1 - 1.2 * table_width))
|
||||
right = min(width, round(x1 + .07 * table_width))
|
||||
top = max(0, round(y1 - .1 * table_height))
|
||||
bottom = min(height, round(y2 + .3 * table_height))
|
||||
if right - left < 300 or bottom - top < 300:
|
||||
return None
|
||||
for other in blocks:
|
||||
if other is block or other.get("block_label") != "table":
|
||||
continue
|
||||
other_box = _box(other.get("block_bbox"), width, height)
|
||||
if not other_box:
|
||||
continue
|
||||
a, b, c, d = other_box
|
||||
overlap = max(0, min(x1, c) - max(left, a)) * max(0, min(bottom, d) - max(top, b))
|
||||
if overlap > .15 * (x1 - left) * (bottom - top):
|
||||
return None
|
||||
return (left, top, right, bottom)
|
||||
|
||||
|
||||
def _run_tile(pipeline, image: Image.Image, crop_box, private: Path):
|
||||
descriptor, name = tempfile.mkstemp(prefix="table-", suffix=".jpg", dir=private)
|
||||
tile_path = Path(name)
|
||||
try:
|
||||
with os.fdopen(descriptor, "wb") as stream:
|
||||
image.crop(crop_box).save(stream, format="JPEG", quality=92)
|
||||
results = list(pipeline.predict(str(tile_path)))
|
||||
if len(results) != 1:
|
||||
return None
|
||||
payload = results[0].json
|
||||
return extract(payload.get("res", payload), crop_box[:2])
|
||||
except Exception:
|
||||
# Keep the original page result if a diagnostic retry fails.
|
||||
return None
|
||||
finally:
|
||||
tile_path.unlink(missing_ok=True)
|
||||
|
||||
|
||||
def predict_tiled(pipeline, path: Path, private: Path) -> dict:
|
||||
"""Generic full-page coverage for a wide page whose full parsing timed out."""
|
||||
with Image.open(path) as image:
|
||||
width, height = image.size
|
||||
boxes = [(0, 0, round(width * .34), height),
|
||||
(round(width * .25), 0, round(width * .66), height),
|
||||
(round(width * .61), 0, width, height)]
|
||||
tiles = [_run_tile(pipeline, image, box, private) for box in boxes]
|
||||
valid = [tile for tile in tiles if tile and tile["lines"]]
|
||||
if not valid:
|
||||
raise RuntimeError("tile_inference_failed")
|
||||
result = extract({"width": width, "height": height, "parsing_res_list": []})
|
||||
for tile in valid:
|
||||
result["lines"].extend(tile["lines"])
|
||||
result["regions"].extend(tile["regions"])
|
||||
result["lines"] = result["lines"][:500]
|
||||
result["regions"] = result["regions"][:200]
|
||||
result["retry_used"] = True
|
||||
result["layout_note"] = (
|
||||
f"Die Gesamtansicht brauchte zu lange. {len(valid)} von 3 überlappenden "
|
||||
"Ansichten derselben Seite wurden lokal gelesen; widersprüchliche "
|
||||
"Feldwerte bleiben ohne Zuordnung. Tabellenbelege sind grobe Regionen.")
|
||||
return result
|
||||
|
||||
|
||||
def predict_adaptive(pipeline, path: Path, private: Path) -> dict:
|
||||
results = list(pipeline.predict(str(path)))
|
||||
if len(results) != 1:
|
||||
raise RuntimeError("invalid_page_count")
|
||||
original = results[0].json
|
||||
payload = original.get("res", original)
|
||||
width, height = int(payload["width"]), int(payload["height"])
|
||||
blocks = payload.get("parsing_res_list", [])
|
||||
large_table = next((block for block in blocks if _retry_table(block, width, height)), None)
|
||||
neighbor_box = next((box for block in blocks
|
||||
if (box := _missing_neighbor_box(block, blocks, width, height))), None)
|
||||
if large_table is None and neighbor_box is None:
|
||||
return extract(payload)
|
||||
|
||||
# A large, almost empty table is a measurable layout/scale failure. Retry
|
||||
# only that detected region in two smaller views; all pixels still come
|
||||
# from the same upload and all coordinates map back to the common image.
|
||||
tiles = []
|
||||
large_recovered = False
|
||||
neighbor_recovered = False
|
||||
with Image.open(path) as image:
|
||||
if large_table is not None:
|
||||
box = _box(large_table["block_bbox"], width, height)
|
||||
for crop_box in _split_table(box, width, height):
|
||||
parsed = _run_tile(pipeline, image, crop_box, private)
|
||||
if parsed and parsed["lines"]:
|
||||
tiles.append(parsed)
|
||||
large_recovered = True
|
||||
if neighbor_box is not None:
|
||||
parsed = _run_tile(pipeline, image, neighbor_box, private)
|
||||
if parsed and parsed["lines"]:
|
||||
tiles.append(parsed)
|
||||
neighbor_recovered = True
|
||||
if not tiles:
|
||||
return extract(payload)
|
||||
kept = [block for block in blocks if block is not large_table] if large_recovered else blocks
|
||||
result = extract({**payload, "parsing_res_list": kept})
|
||||
for tile in tiles:
|
||||
result["lines"].extend(tile["lines"])
|
||||
result["regions"].extend(tile["regions"])
|
||||
result["lines"] = result["lines"][:500]
|
||||
result["regions"] = result["regions"][:200]
|
||||
result["retry_used"] = True
|
||||
notes = []
|
||||
if large_recovered:
|
||||
notes.append("Ein übergroßer, fast leerer Tabellenblock wurde in zwei kleineren Ansichten erneut gelesen.")
|
||||
if neighbor_recovered:
|
||||
notes.append("Eine von der Layoutstufe ausgelassene Nachbarspalte wurde zusätzlich gelesen.")
|
||||
notes.append("Tabellenzeilen haben weiterhin nur einen groben Regionsbeleg.")
|
||||
result["layout_note"] = " ".join(notes)
|
||||
return result
|
||||
|
||||
|
||||
def main() -> None:
|
||||
data = sys.stdin.buffer.read(16 * 1024 * 1024 + 1)
|
||||
if not data or len(data) > 16 * 1024 * 1024:
|
||||
print(json.dumps({"error": "invalid_input"}))
|
||||
return
|
||||
models = ROOT / "official_models"
|
||||
layout = models / "PP-DocLayoutV3"
|
||||
vlm = models / "PaddleOCR-VL-1.6"
|
||||
if not (layout / "inference.pdiparams").is_file() or not (vlm / "model.safetensors").is_file():
|
||||
print(json.dumps({"error": "models_missing"}))
|
||||
return
|
||||
private = Path(os.environ.get("OCR_PRIVATE_DIR", ROOT / "private"))
|
||||
private.mkdir(mode=0o700, exist_ok=True)
|
||||
descriptor, name = tempfile.mkstemp(prefix="page-", suffix=".jpg", dir=private)
|
||||
path = Path(name)
|
||||
try:
|
||||
with os.fdopen(descriptor, "wb") as stream:
|
||||
stream.write(data)
|
||||
# Paddle logs should never expose text or private file paths through
|
||||
# the SSH result channel. The app suppresses stderr as well.
|
||||
stdout_copy = os.dup(1)
|
||||
null_fd = os.open(os.devnull, os.O_WRONLY)
|
||||
os.dup2(null_fd, 1)
|
||||
os.close(null_fd)
|
||||
try:
|
||||
from paddleocr import PaddleOCRVL
|
||||
pipeline = PaddleOCRVL(pipeline_version="v1.6", use_layout_detection=True,
|
||||
layout_detection_model_dir=str(layout), vl_rec_model_dir=str(vlm),
|
||||
use_doc_orientation_classify=False,
|
||||
use_doc_unwarping=False,
|
||||
device=os.environ.get("OCR_VISION_DEVICE", "cpu"))
|
||||
result = (predict_tiled(pipeline, path, private)
|
||||
if "--tiles-only" in sys.argv[1:] else
|
||||
predict_adaptive(pipeline, path, private))
|
||||
finally:
|
||||
os.dup2(stdout_copy, 1)
|
||||
os.close(stdout_copy)
|
||||
print(json.dumps(result, ensure_ascii=False, separators=(",", ":")))
|
||||
except Exception:
|
||||
# No exception text: model libraries may include recognized content.
|
||||
print(json.dumps({"error": "inference_failed"}))
|
||||
finally:
|
||||
path.unlink(missing_ok=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user