Add private local OCR comparison package

This commit is contained in:
OCR Team
2026-10-07 21:00:21 +02:00
commit 87a1d114da
15 changed files with 2044 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
"""Local comparison UI for real-document OCR evaluation."""
+318
View File
@@ -0,0 +1,318 @@
"use strict";
const $ = id => document.getElementById(id);
const input = $("image-file"), original = $("original"), cornersCanvas = $("corners");
const processed = $("processed"), evidenceCanvas = $("evidence");
const run = $("run"), status = $("status"), rotation = $("rotation");
const initialCorners = () => [[0, 0], [1, 0], [1, 1], [0, 1]];
let corners = initialCorners(), edited = false, busy = false, ready = false;
let objectUrl = null, drag = null, result = null, selectedEvidence = null;
const manual = {classic: new Map(), vision: new Map()};
const suppressed = {classic: new Set(), vision: new Set()};
const priorityCodes = ["C.1.1", "C.1.2", "C.1.3", "B", "2.1", "2.2"];
const reviewMethods = new Set(["merged_code", "vl_table_context_code", "caption_region_next_line", "surname_above_forename"]);
const fieldNames = {"A":"Kennzeichen","C.1.1":"Name / Firmenname","C.1.2":"Vorname",
"C.1.3":"Anschrift","B":"Erstzulassung","2.1":"Herstellerschlüssel (HSN)",
"2.2":"Typschlüssel (TSN)","2":"Hersteller","D.1":"Marke","D.2":"Typ / Variante",
"D.3":"Handelsbezeichnung","E":"Fahrzeug-Identifizierungsnummer","P.3":"Kraftstoffart",
"R":"Farbe","17":"Merkmal zur Betriebserlaubnis"};
function setStatus(message, kind = "") {
status.textContent = message;
status.className = `status ${kind}`;
}
function validCorners() {
if (!ready) return false;
if (!edited) return true;
const w = original.naturalWidth, h = original.naturalHeight;
const points = corners.map(([x, y]) => [x * w, y * h]);
const edges = points.map((p, i) => [points[(i + 1) % 4][0] - p[0], points[(i + 1) % 4][1] - p[1]]);
const crosses = edges.map((a, i) => a[0] * edges[(i + 1) % 4][1] - a[1] * edges[(i + 1) % 4][0]);
const area = Math.abs(points.reduce((sum, p, i) => sum + p[0] * points[(i + 1) % 4][1] - p[1] * points[(i + 1) % 4][0], 0)) / 2;
return crosses.every(value => value > 0) && area > .02 * w * h &&
edges.every(([x, y]) => Math.hypot(x, y) >= 30);
}
function updateControls() {
input.disabled = busy;
$("reset-corners").disabled = !ready || busy;
rotation.disabled = busy;
run.disabled = busy || !validCorners();
}
function drawCorners() {
const w = original.naturalWidth, h = original.naturalHeight;
if (!w || !h) return;
cornersCanvas.width = w;
cornersCanvas.height = h;
const ctx = cornersCanvas.getContext("2d");
const pts = corners.map(([x, y]) => [x * (w - 1), y * (h - 1)]);
ctx.strokeStyle = "#1459dd";
ctx.lineWidth = Math.max(5, w / 550);
ctx.beginPath();
pts.forEach(([x, y], i) => i ? ctx.lineTo(x, y) : ctx.moveTo(x, y));
ctx.closePath();ctx.stroke();
const radius = Math.max(25, w / 110);
pts.forEach(([x, y], i) => {
ctx.beginPath();ctx.arc(x, y, radius, 0, 2 * Math.PI);
ctx.fillStyle = "#fff";ctx.fill();ctx.stroke();
ctx.fillStyle = "#1547a5";ctx.font = `bold ${Math.max(22, w / 125)}px system-ui`;
ctx.fillText(String(i + 1), Math.min(w - radius, x + radius * .9), Math.max(radius, y - radius * .65));
});
}
function pointInCanvas(event) {
const b = cornersCanvas.getBoundingClientRect();
return [(event.clientX - b.left) / b.width, (event.clientY - b.top) / b.height];
}
cornersCanvas.addEventListener("pointerdown", event => {
if (!ready || busy) return;
const b = cornersCanvas.getBoundingClientRect();
const [x, y] = pointInCanvas(event);
const distances = corners.map(([cx, cy]) => Math.hypot((cx - x) * b.width, (cy - y) * b.height));
const index = distances.indexOf(Math.min(...distances));
if (distances[index] > 32) return;
drag = index;
cornersCanvas.setPointerCapture(event.pointerId);
event.preventDefault();
});
cornersCanvas.addEventListener("pointermove", event => {
if (drag === null || busy) return;
const [x, y] = pointInCanvas(event);
corners[drag] = [Math.max(0, Math.min(1, x)), Math.max(0, Math.min(1, y))];
edited = true; result = null; $("comparison").hidden = true;
drawCorners();updateControls();
setStatus(validCorners() ? "Ecken geändert. Bitte beide Verfahren neu starten." :
"Ecken müssen ein nicht gekreuztes Viereck um das Dokument bilden.", validCorners() ? "" : "error");
});
function finishDrag() {drag = null;}
cornersCanvas.addEventListener("pointerup", finishDrag);
cornersCanvas.addEventListener("pointercancel", finishDrag);
$("reset-corners").addEventListener("click", () => {
corners = initialCorners(); edited = false; result = null;
$("comparison").hidden = true;drawCorners();updateControls();setStatus("Ganzes Bild ausgewählt.");
});
rotation.addEventListener("change", () => {
result = null;$("comparison").hidden = true;setStatus("Leserichtung geändert. Bitte erneut vergleichen.");
});
input.addEventListener("change", () => {
if (objectUrl) URL.revokeObjectURL(objectUrl);
objectUrl = null; ready = false; edited = false; corners = initialCorners();
drag = null; result = null; selectedEvidence = null;
manual.classic.clear();manual.vision.clear();
suppressed.classic.clear();suppressed.vision.clear();
$("comparison").hidden = true;rotation.value = "0";
const file = input.files?.[0];
$("original-wrap").hidden = !file;
$("file-name").textContent = file ? `${file.name} · ${(file.size / 1048576).toFixed(2)} MiB` : "JPEG oder PNG, maximal 16 MiB.";
if (file) {objectUrl = URL.createObjectURL(file);original.src = objectUrl;}
else original.removeAttribute("src");
setStatus("");updateControls();
});
original.addEventListener("load", () => {
ready = true;drawCorners();updateControls();
$("file-name").textContent += ` · ${original.naturalWidth} × ${original.naturalHeight} Pixel`;
});
original.addEventListener("error", () => {
ready = false;updateControls();setStatus("Das Bild kann nicht geöffnet werden. Bitte JPEG oder PNG verwenden.", "error");
});
window.addEventListener("pagehide", () => {if (objectUrl) URL.revokeObjectURL(objectUrl);});
const errorText = {
busy: "Ein Vergleich läuft bereits.", unsupported_media_type: "Nur JPEG und PNG sind erlaubt.",
upload_too_large: "Das Bild ist größer als 16 MiB.", upload_timeout: "Der Upload hat zu lange gedauert.",
invalid_image: "Die Bilddatei lässt sich nicht vollständig lesen.",
unsupported_encoding: "Der Bildinhalt ist kein unterstütztes JPEG oder PNG.",
image_too_large: "Die Auflösung überschreitet 24 Megapixel oder 10.000 Pixel pro Seite.",
invalid_corners: "Die vier Ecken bilden kein gültiges Dokumentviereck.",
invalid_rotation: "Die gewählte Drehung ist ungültig.",
classic_timeout: "Die klassische OCR hat das Zeitlimit überschritten.",
classic_models_missing: "Die lokalen OCR-Modelle fehlen. Zuerst die Modell-Einrichtung aus der README ausführen.",
classic_failed: "Die klassische OCR ist lokal fehlgeschlagen.",
classic_invalid: "Die klassische OCR hat ein ungültiges Ergebnis geliefert.",
vision_timeout: "PaddleOCR-VL hat das Zeitlimit überschritten.",
vision_failed: "PaddleOCR-VL konnte auf der konfigurierten eigenen Hardware nicht ausgeführt werden.",
vision_disabled: "Vision ist in dieser Installation deaktiviert.",
vision_config_error: "Die Vision-Konfiguration ist unvollständig oder ungültig.",
vision_models_missing: "Die lokalen Vision-Modelle fehlen. Zuerst die Modell-Einrichtung aus der README ausführen.",
vision_invalid: "PaddleOCR-VL hat ein ungültiges Ergebnis geliefert.",
};
function element(tag, value, className = "") {
const node = document.createElement(tag);
node.textContent = value;
if (className) node.className = className;
return node;
}
function drawEvidence() {
if (!result || !processed.naturalWidth) return;
const w = processed.naturalWidth, h = processed.naturalHeight;
evidenceCanvas.width = w;evidenceCanvas.height = h;
const ctx = evidenceCanvas.getContext("2d");
if ($( "all-boxes" ).checked) {
for (const [engine, branch] of Object.entries(result.results)) {
if (branch.status !== "ok") continue;
ctx.strokeStyle = engine === "classic" ? "#008778" : "#425edf";
ctx.lineWidth = Math.max(2, w / 900);
for (const line of branch.lines) if (line.box) {
const [x1, y1, x2, y2] = line.box;
ctx.strokeRect(x1, y1, x2 - x1, y2 - y1);
}
}
}
if (selectedEvidence?.anchor && selectedEvidence.anchor.join() !== selectedEvidence.box.join()) {
const [x1, y1, x2, y2] = selectedEvidence.anchor;
ctx.strokeStyle = "#d88614";ctx.fillStyle = "rgba(216,134,20,.13)";
ctx.lineWidth = Math.max(4, w / 400);
ctx.fillRect(x1, y1, x2 - x1, y2 - y1);
ctx.strokeRect(x1, y1, x2 - x1, y2 - y1);
}
if (selectedEvidence?.box) {
const [x1, y1, x2, y2] = selectedEvidence.box;
ctx.strokeStyle = "#d72b4a";ctx.fillStyle = "rgba(215,43,74,.16)";
ctx.lineWidth = Math.max(4, w / 400);
ctx.fillRect(x1, y1, x2 - x1, y2 - y1);
ctx.strokeRect(x1, y1, x2 - x1, y2 - y1);
}
}
function focusEvidence(box, anchor = null) {
if (!box) return;
selectedEvidence = {box, anchor};drawEvidence();
$("comparison").scrollIntoView({behavior:"smooth",block:"start"});
}
processed.addEventListener("load", drawEvidence);
$("all-boxes").addEventListener("change", drawEvidence);
function renderCard(engine) {
const card = $(`${engine}-card`), branch = result.results[engine];
card.replaceChildren(element("h3", engine === "classic" ? "A · Klassische OCR" : "B · Vision-Dokumentmodell"));
if (branch.status !== "ok") {
card.append(element("p", errorText[branch.error] || "Dieses Verfahren ist fehlgeschlagen.", "error"),
element("p", `${(branch.elapsed_ms / 1000).toFixed(1)} s`, "meta"));
return;
}
const automatic = branch.fields.filter(field => !suppressed[engine].has(field.code));
card.append(element("p", `${branch.model} · ${(branch.elapsed_ms / 1000).toFixed(1)} s · ${branch.lines.length} Textbereiche · ${automatic.length} automatische Feldvorschläge`, "meta"));
if (branch.layout_note) card.append(element("p", branch.layout_note, "note"));
card.append(element("p", "Die Zuordnung ist ein Vorschlag. Orange markiert den gelesenen Feldcode, Rot den Wert. Falsche Vorschläge lassen sich verwerfen; der Rohtext bleibt erhalten.", "assignment-note"));
card.append(element("h4", "Strukturierte Feldvorschläge"));
const merged = new Map(automatic.map(field => [field.code, field]));
for (const [code, field] of manual[engine]) merged.set(code, field);
const priority = element("div", "", "priority-fields");
priority.append(element("h5", "Wichtige Angaben"));
for (const code of priorityCodes) {
const field = merged.get(code);
priority.append(element("span", `${code} ${fieldNames[code]}: ${field ? (field.manual ? "manuell" : reviewMethods.has(field.method) ? "prüfen" : "Vorschlag") : "offen"}`,
`priority-item${field ? "" : " missing"}`));
}
card.append(priority);
if (!merged.size) card.append(element("p", "Keine sicher belegbaren Felder. Rohtext prüfen und bei Bedarf manuell zuordnen.", "meta"));
const fieldList = element("div", "", "field-list");
for (const field of merged.values()) {
const row = element("div", "", `field-item${reviewMethods.has(field.method) ? " needs-review" : ""}`);
const show = element("button", "", "field-show");show.type = "button";
const heading = element("span", "", "field-heading");
heading.append(element("strong", field.code), element("small", fieldNames[field.code] || "Feldcode"));
const method = field.manual ? "Manuell aus OCR-Text" :
field.method === "adjacent_code" ? "Code und Wert in Nachbarboxen" :
field.method === "below_caption" ? "Wert unter gedruckter Beschriftung" :
field.method === "caption_region_next_line" ? "Nächste Textzeile derselben Layoutregion · bitte prüfen" :
field.method === "surname_above_forename" ? "C.1.1-Code nicht gelesen · Wert oberhalb von C.1.2 · bitte prüfen" :
field.method === "vl_table_adjacent_cell" ? "Benachbarte Modellzellen · grober Tabellenbeleg" :
field.method === "vl_table_context_code" ? "Punkt im Feldcode fehlt · Tabellenfolge B/2.1/2.2 · bitte prüfen" :
field.method === "merged_code" ? "Code und Wert verschmolzen · bitte prüfen" :
"Code und Wert in einer Textzeile";
show.append(heading, element("span", field.value, "field-value"), element("small", method, "field-method"));
show.disabled = !field.box;
show.addEventListener("click", () => focusEvidence(field.box, field.anchor_box));
const dismiss = element("button", field.manual ? "Entfernen" : "Verwerfen", "field-dismiss");
dismiss.type = "button";
dismiss.setAttribute("aria-label", `${field.code} ${field.manual ? "entfernen" : "verwerfen"}`);
dismiss.addEventListener("click", () => {
if (field.manual) manual[engine].delete(field.code);
else suppressed[engine].add(field.code);
renderCard(engine);
});
row.append(show, dismiss);
fieldList.append(row);
}
card.append(fieldList, element("h4", "Rohtext mit Bildbelegen"));
if (!branch.lines.length) card.append(element("p", "Kein Text erkannt.", "meta"));
const raw = element("div", "", "raw-lines");
let code, lineSelect, value, form;
branch.lines.forEach((line, index) => {
const entry = element("div", "", "raw-entry");
const row = element("button", "", "raw-line");row.type = "button";
row.append(element("span", `${index + 1}.`), element("span", line.text));
row.disabled = !line.box;
row.title = line.source || "Bildbereich markieren";
row.addEventListener("click", () => {
lineSelect.value = String(index);value.value = line.text;
focusEvidence(line.box);
});
const pick = element("button", "Zuordnen", "raw-pick");pick.type = "button";
pick.disabled = !line.box;
pick.addEventListener("click", () => {
lineSelect.value = String(index);value.value = line.text;
selectedEvidence = {box:line.box,anchor:null};drawEvidence();
form.scrollIntoView({behavior:"smooth",block:"center"});code.focus();
});
entry.append(row,pick);raw.append(entry);
});
card.append(raw);
if (!branch.lines.length) return;
card.append(element("h4", "Zeile manuell zuordnen"));
form = element("div", "", "manual");
code = document.createElement("select");lineSelect = document.createElement("select");
code.setAttribute("aria-label", "Feldcode");lineSelect.setAttribute("aria-label", "OCR-Zeile");
code.append(new Option("Feldcode", ""));
result.field_codes.forEach(fieldCode => code.append(new Option(`${fieldCode} · ${fieldNames[fieldCode] || "Feldcode"}`, fieldCode)));
lineSelect.append(new Option("OCR-Zeile wählen", ""));
branch.lines.forEach((row, index) => lineSelect.append(new Option(`${index + 1}: ${row.text.slice(0, 55)}`, String(index))));
value = document.createElement("input");value.type = "text";value.maxLength = 500;
value.setAttribute("aria-label", "Wörtlicher Ausschnitt aus der OCR-Zeile");
value.placeholder = "OCR-Zeile wählen, dann bei Bedarf auf den Wert kürzen";
lineSelect.addEventListener("change", () => {value.value = lineSelect.value === "" ? "" : branch.lines[Number(lineSelect.value)].text;});
const add = element("button", "Zuordnen");add.type = "button";
const hint = element("p", "Nur wörtlich erkannter Text mit Bildbereich ist zulässig.", "manual-hint");
add.addEventListener("click", () => {
const row = lineSelect.value === "" ? null : branch.lines[Number(lineSelect.value)];
const chosen = value.value.trim();
if (!code.value || !chosen || !row?.box || !row.text.includes(chosen)) {
hint.textContent = "Feldcode und wörtlichen Ausschnitt aus der gewählten Zeile angeben.";
hint.className = "manual-hint error";return;
}
manual[engine].set(code.value, {code:code.value,value:chosen,box:row.box,manual:true});
suppressed[engine].delete(code.value);
renderCard(engine);
});
const clear = element("button", "Manuelle Zuordnungen löschen");clear.type = "button";
clear.addEventListener("click", () => {manual[engine].clear();renderCard(engine);});
form.append(code,lineSelect,value,add,clear,hint);card.append(form);
}
run.addEventListener("click", async () => {
const file = input.files?.[0];
if (!file || !validCorners() || busy) return;
if (file.size > 16 * 1048576) {setStatus(errorText.upload_too_large, "error");return;}
const mime = file.type || (/\.jpe?g$/i.test(file.name) ? "image/jpeg" : /\.png$/i.test(file.name) ? "image/png" : "");
if (!["image/jpeg", "image/png"].includes(mime)) {setStatus(errorText.unsupported_media_type, "error");return;}
busy = true;updateControls();result = null;selectedEvidence = null;
manual.classic.clear();manual.vision.clear();$("comparison").hidden = true;
suppressed.classic.clear();suppressed.vision.clear();
setStatus("Die lokalen Verfahren verarbeiten dasselbe Bild …");
try {
const geometry = {rotation:Number(rotation.value)};
if (edited) geometry.corners = corners;
const response = await fetch("/api/compare", {method:"POST",body:file,
headers:{"Content-Type":mime,"X-OCR-Geometry":JSON.stringify(geometry)},cache:"no-store"});
const data = await response.json();
if (!response.ok) {setStatus(errorText[data.error] || `Vergleich fehlgeschlagen (${response.status}).`,"error");return;}
result = data;processed.src = data.image;
$("image-size").textContent = `Gemeinsames OCR-Bild: ${data.image_size.width} × ${data.image_size.height} Pixel. Klick auf Wert oder Rohtext markiert seinen Bildbeleg.`;
renderCard("classic");renderCard("vision");$("comparison").hidden = false;
const both = Object.values(data.results).every(branch => branch.status === "ok");
setStatus(both ? "Beide Ergebnisse bereit. Feldvorschläge anhand der Bildbelege prüfen." :
"Ein Verfahren ist fehlgeschlagen; das andere Ergebnis und der Fehler sind sichtbar.",both ? "ok" : "error");
} catch {
setStatus(result ? "Die Antwort des lokalen Vergleichs konnte nicht angezeigt werden. Seite neu laden." :
"Der lokale Dienst antwortet nicht. Prozess und /api/health prüfen.","error");
} finally {busy = false;updateControls();}
});
+48
View File
@@ -0,0 +1,48 @@
"""PP-OCRv5 Latin inference using preloaded local model directories only."""
from __future__ import annotations
import json
import os
from pathlib import Path
import sys
ROOT = Path(os.environ.get("OCR_MODEL_HOME", Path(__file__).resolve().parents[1] / ".cache")).expanduser().resolve()
MODELS = ROOT / "official_models"
def main() -> None:
if len(sys.argv) != 3:
raise SystemExit(2)
image, output = map(Path, sys.argv[1:])
detector = MODELS / "PP-OCRv5_mobile_det"
recognizer = MODELS / "latin_PP-OCRv5_mobile_rec"
if not (detector / "inference.pdiparams").is_file() or not (recognizer / "inference.pdiparams").is_file():
raise SystemExit(3)
from paddleocr import PaddleOCR
engine = PaddleOCR(
text_detection_model_name="PP-OCRv5_mobile_det",
text_detection_model_dir=str(detector),
text_recognition_model_name="latin_PP-OCRv5_mobile_rec",
text_recognition_model_dir=str(recognizer),
use_doc_orientation_classify=False,
use_doc_unwarping=False,
use_textline_orientation=False,
device="cpu",
)
results = list(engine.predict(str(image)))
if len(results) != 1:
raise RuntimeError("expected_one_page")
raw = results[0].json["res"]
lines = []
for text, polygon in zip(raw["rec_texts"], raw["rec_polys"]):
points = [[int(x), int(y)] for x, y in polygon]
xs, ys = zip(*points)
lines.append({"text": str(text), "bbox": [min(xs), min(ys), max(xs), max(ys)]})
output.write_text(json.dumps({"lines": lines}, ensure_ascii=False))
if __name__ == "__main__":
main()
+356
View File
@@ -0,0 +1,356 @@
"""Assign observed OCR text to ZB Teil I fields with image evidence.
No value is generated. Weak code inferences from neighboring observed fields
are explicit and remain reviewable in the UI.
"""
from __future__ import annotations
from difflib import SequenceMatcher
import re
FIELD_CODES = tuple("""A C.1.1 C.1.2 C.1.3 X I B 2.1 2.2 J 4 E 3 D.1 D.2 D.3 2 5
V.9 14 P.3 10 14.1 P.1 L 9 P.2/P.4 T 18 19 20 G 12 13 Q V.7 F.1 F.2
7.1 7.2 7.3 8.1 8.2 8.3 U.1 U.2 U.3 O.1 O.2 S.1 S.2 15.1 15.2 15.3
R 11 K 6 17 16 21 22""".split())
def _norm(text: str) -> str:
return re.sub(r"[^A-Z0-9]", "", text.upper())
def _code(text: str) -> str | None:
# Exact dotted labels win. Dot loss is accepted only when it is
# unambiguous; arbitrary words and strings with values are not labels.
literal = text.strip().upper().strip("() ")
if literal in ("21", "22"):
# Without the dot, 2.1/2.2 and 21/22 cannot be distinguished.
return None
if literal in FIELD_CODES:
return literal
token = _norm(text)
matches = [code for code in FIELD_CODES if _norm(code) == token]
return matches[0] if len(matches) == 1 else None
_OWNER_CAPTIONS = {
"C.1.1": re.compile(r"\b(?:name|firmenname)\b", re.I),
"C.1.2": re.compile(r"\bvorname", re.I),
"C.1.3": re.compile(r"\b(?:anschrift|adresse)\b", re.I),
}
_OWNER_PRINTED = {"C.1.1": "NAMEODERFIRMENNAME", "C.1.2": "VORNAMEN",
"C.1.3": "ANSCHRIFT"}
def _owner_caption(text: str) -> str | None:
match = re.match(r"^\s*\(?\s*C\s*\.?\s*1\s*\.?\s*([123])\s*\)?\s*(.*)$", text, re.I)
if match:
code = f"C.1.{match[1]}"
tail = match[2].strip(" \t:;=-")
if not tail or _OWNER_CAPTIONS[code].search(tail):
return code
letters = re.sub(r"[^A-Z]", "", tail.upper())
if len(letters) >= 6 and SequenceMatcher(None, letters, _OWNER_PRINTED[code]).ratio() >= .62:
return code
# Small printed codes can be lost while the full caption survives.
# Use only the distinctive short printed wording, never a person's name.
if len(text) < 55:
if re.search(r"\bName\s+oder\s+Firmenname\b", text, re.I):
return "C.1.1"
if re.search(r"\bVorname\b", text, re.I):
return "C.1.2"
if re.search(r"\bAnschrift\b", text, re.I):
return "C.1.3"
return None
def _height(box: list[float]) -> float:
return box[3] - box[1]
def _vertical_overlap(a: list[float], b: list[float]) -> float:
return max(0, min(a[3], b[3]) - max(a[1], b[1]))
def _nearby_value(anchor: dict, lines: list[dict], labels: set[int]) -> tuple[int, float] | None:
ax1, ay1, ax2, ay2 = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if (index in labels or line is anchor or not line.get("box") or
not line["text"].strip(" \t-–—.,")):
continue
bx1, by1, bx2, by2 = line["box"]
bh = _height(line["box"])
if bx2 <= (ax1 + ax2) / 2 or bx1 < ax1 - 0.2 * ah:
continue
if bx1 - ax2 > max(120, 5 * ah):
continue
overlap = _vertical_overlap(anchor["box"], line["box"])
if overlap < 0.45 * min(ah, bh):
continue
center_gap = abs((by1 + by2) / 2 - (ay1 + ay2) / 2)
# Prefer the closest value to the right; reject rows that only touch
# because large OCR boxes extend into a neighbouring printed row.
candidates.append((max(0, bx1 - ax2) + center_gap * 1.5, index))
if not candidates:
return None
candidates.sort()
return candidates[0][1], candidates[0][0]
def _owner_below(anchor_index: int, lines: list[dict], labels: set[int]) -> dict | None:
"""Read holder value rows below an observed caption, within its column."""
anchor = lines[anchor_index]
code = _owner_caption(anchor["text"])
if "Layoutblock" in anchor.get("source", ""):
# The VL parser sometimes puts a printed caption and its value in
# consecutive text lines of one coarse region. Their box is shared;
# line order is the only honest evidence for this relation.
same_region = []
for index in range(anchor_index + 1, min(len(lines), anchor_index + (3 if code == "C.1.3" else 2))):
line = lines[index]
if index in labels or line.get("box") != anchor["box"] or not line["text"].strip():
break
same_region.append(index)
if same_region:
value = "\n".join(lines[index]["text"].strip() for index in same_region)
return {"value": value[:500], "raw_text": value, "box": anchor["box"],
"anchor_box": anchor["box"], "line_indices": [anchor_index, *same_region],
"method": "caption_region_next_line", "confidence": None}
ax1, _, _, ay2 = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if index in labels or not line.get("box") or not line["text"].strip():
continue
x1, y1, x2, y2 = line["box"]
if not (ay2 - .15 * ah <= y1 <= ay2 + max(85, 2.8 * ah)):
continue
if not (ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)):
continue
if x2 <= ax1 or re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I):
continue
candidates.append((max(0, y1 - ay2), abs(x1 - ax1), index))
if not candidates:
return None
candidates.sort()
first = lines[candidates[0][2]]["box"]
first_indices = [index for index, line in enumerate(lines)
if (index not in labels or line["text"].strip().isdigit())
and line.get("box") and line["text"].strip()
and line["box"][2] > ax1
and _vertical_overlap(first, line["box"]) >=
.35 * min(_height(first), _height(line["box"]))]
first_indices.sort(key=lambda index: lines[index]["box"][0])
# Only join fragments that touch the same printed value row. A remote
# table column with a large gap is not part of the holder's name.
row = [min((index for index in first_indices
if ax1 - max(65, 1.5 * ah) <= lines[index]["box"][0]
<= ax1 + max(170, 3 * ah)),
key=lambda index: lines[index]["box"][0], default=candidates[0][2])]
for index in first_indices:
if index == row[0] or lines[index]["box"][0] < lines[row[0]]["box"][0]:
continue
previous = lines[row[-1]]["box"]
box = lines[index]["box"]
if box[0] - previous[2] <= max(45, .8 * _height(first)):
row.append(index)
rows = [row]
if code == "C.1.3" and "Layoutblock" not in anchor.get("source", ""):
bottom = max(lines[index]["box"][3] for index in row)
row_height = max(_height(lines[index]["box"]) for index in row)
next_row = []
for index, line in enumerate(lines):
if ((index in labels and not line["text"].strip().isdigit())
or index in row or not line.get("box") or not line["text"].strip()):
continue
x1, y1, x2, _ = line["box"]
if (bottom - .1 * row_height <= y1 <= bottom + max(25, .55 * row_height)
and x1 >= ax1 - max(65, 1.5 * ah)
and x2 > ax1
and not re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I)):
next_row.append(index)
next_row.sort(key=lambda index: lines[index]["box"][0])
if next_row:
starter = next((index for index in next_row
if lines[index]["box"][0] <= ax1 + max(170, 3 * ah)), None)
contiguous = [starter] if starter is not None else []
for index in next_row:
if not contiguous or index == starter or lines[index]["box"][0] < lines[starter]["box"][0]:
continue
if lines[index]["box"][0] - lines[contiguous[-1]]["box"][2] <= max(45, .8 * row_height):
contiguous.append(index)
if contiguous:
rows.append(contiguous)
indices = [index for row in rows for index in row]
value = "\n".join("\n".join(lines[index]["text"].strip() for index in row)
if len(row) > 1 and all(lines[index]["box"] == lines[row[0]]["box"] for index in row)
else " ".join(lines[index]["text"].strip() for index in row)
for row in rows)
boxes = [lines[index]["box"] for index in indices]
box = [min(b[0] for b in boxes), min(b[1] for b in boxes),
max(b[2] for b in boxes), max(b[3] for b in boxes)]
return {"value": value[:500], "raw_text": value, "box": box,
"anchor_box": anchor["box"], "line_indices": [anchor_index, *indices],
"method": "below_caption", "confidence": None}
def _surname_above_forename(lines: list[dict], owner_labels: dict[int, str],
labels: set[int]) -> dict | None:
"""Weak suggestion when C.1.1 is unreadable but C.1.2 is observed."""
if "C.1.1" in owner_labels.values():
return None
forename = [index for index, code in owner_labels.items() if code == "C.1.2"]
if len(forename) != 1:
return None
anchor_index = forename[0]
anchor = lines[anchor_index]
ax1, ay1, _, _ = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if index in labels or not line.get("box"):
continue
text = line["text"].strip()
if len(re.sub(r"[^A-ZÄÖÜ]", "", text.upper())) < 3 or any(c.isdigit() for c in text):
continue
x1, _, x2, y2 = line["box"]
gap = ay1 - y2
if (0 <= gap <= max(150, 3 * ah)
and ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)
and x2 > ax1):
candidates.append((gap, abs(x1 - ax1), index))
candidates.sort()
if not candidates or (len(candidates) > 1 and candidates[1][0] - candidates[0][0] < ah):
return None
value_index = candidates[0][2]
value_line = lines[value_index]
return {"code": "C.1.1", "value": value_line["text"].strip()[:500],
"raw_text": value_line["text"], "box": value_line["box"],
"anchor_box": anchor["box"], "line_indices": [anchor_index, value_index],
"method": "surname_above_forename", "confidence": None}
def assign_fields(lines: list[dict], *, allow_adjacent: bool = True) -> list[dict]:
"""Return only observed, localized values; duplicates stay in raw lines."""
proposed: dict[str, list[dict]] = {}
labels = {index: code for index, line in enumerate(lines)
if line.get("box") and (code := _code(line["text"]))}
owner_labels = {index: code for index, line in enumerate(lines)
if line.get("box") and (code := _owner_caption(line["text"]))}
label_indices = set(labels) | set(owner_labels)
for index, line in enumerate(lines):
box = line.get("box")
if not box:
continue
if isinstance(line.get("cells"), list):
cells = line["cells"]
contextual = {}
if (len(cells) >= 6 and _code(cells[0]) == "B"
and _norm(cells[2]) == "21" and _norm(cells[4]) == "22"
and _code(cells[2]) is None and _code(cells[4]) is None):
# In one observed table row, the printed B / 2.1 / 2.2
# sequence can lose both tiny dots. Keep this weaker reading
# explicit and reviewable; never reinterpret a lone 21/22.
contextual = {2: "2.1", 4: "2.2"}
consumed_values: set[int] = set()
for cell_index, cell in enumerate(cells[:-1]):
if cell_index in consumed_values:
continue
code = (contextual.get(cell_index) or _code(cell)) if isinstance(cell, str) else None
value = cells[cell_index + 1] if isinstance(cells[cell_index + 1], str) else ""
value = value.strip()
if (code and value and value not in ("-", "–", "—")
and _code(value) is None
and not (code in _OWNER_CAPTIONS and _OWNER_CAPTIONS[code].search(value))):
consumed_values.add(cell_index + 1)
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": line["text"],
"box": box, "anchor_box": box, "line_indices": [index],
"method": "vl_table_context_code" if cell_index in contextual else "vl_table_adjacent_cell",
"confidence": None,
})
continue
text = line["text"].strip()
# Code and value in the same OCR line.
matched = False
for code in sorted(FIELD_CODES, key=len, reverse=True):
if code.startswith("C.1."):
# These printed captions precede the holder data on another row.
continue
printed = re.escape(code)
marker = (rf"(?:\({printed}\)|{printed}(?=\s*[:;=\-]))"
if len(code) == 1 or code.isdigit() else
rf"(?:\({printed}\)|{printed}(?=\s|[:;=\-]))")
match = re.match(rf"^\s*{marker}\s*[:;=\-]?\s*", text, flags=re.I)
if match and (value := text[match.end():].strip(" \t:;=-")) and value.strip("-–—.,"):
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": text,
"box": box, "anchor_box": box, "line_indices": [index],
"method": "same_line_code", "confidence": None,
})
matched = True
break
if not matched:
# Small printed codes can merge with a value in one OCR box.
# Keep the original box and mark this weaker reading for review.
for code in sorted((c for c in FIELD_CODES if re.fullmatch(r"[A-Z]\.\d(?:\.\d)?", c)),
key=len, reverse=True):
marker = re.escape(code).replace(r"\.", r"\.?" )
match = re.match(rf"^\s*{marker}(?=[A-Z0-9])", text, flags=re.I)
if match and (value := text[match.end():].strip(" \t:;=-")):
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": text,
"box": box, "anchor_box": box, "line_indices": [index],
"method": "merged_code", "confidence": None,
})
break
for index, code in (labels.items() if allow_adjacent else []):
if code in _OWNER_CAPTIONS:
continue
nearby = _nearby_value(lines[index], lines, label_indices)
if nearby is None:
continue
value_index, distance = nearby
value_line = lines[value_index]
proposed.setdefault(code, []).append({
"code": code, "value": value_line["text"].strip()[:500],
"raw_text": value_line["text"], "box": value_line["box"],
"anchor_box": lines[index]["box"], "line_indices": [index, value_index],
"method": "adjacent_code", "confidence": None, "_distance": distance,
})
for index, code in owner_labels.items():
below = _owner_below(index, lines, label_indices)
if below:
proposed.setdefault(code, []).append({"code": code, **below})
if allow_adjacent and (weak_surname := _surname_above_forename(lines, owner_labels, label_indices)):
proposed.setdefault("C.1.1", []).append(weak_surname)
selected = {}
for code in FIELD_CODES:
unique = {}
for entry in proposed.get(code, []):
# Overlapping local image tiles may observe the same printed value
# more than once. Identical readings are one proposal, even when
# their honest layout regions have different sizes.
unique.setdefault(_norm(entry["value"]), entry)
if len(unique) == 1:
selected[code] = next(iter(unique.values()))
# One OCR value box cannot prove two different adjacent fields. Prefer a
# clearly nearer printed code; otherwise leave both assignments open.
by_value: dict[int, list[dict]] = {}
for entry in selected.values():
if entry["method"] == "adjacent_code":
by_value.setdefault(entry["line_indices"][-1], []).append(entry)
for collisions in by_value.values():
if len(collisions) < 2:
continue
collisions.sort(key=lambda item: item["_distance"])
keep = collisions[0] if collisions[0]["_distance"] < .7 * collisions[1]["_distance"] else None
for entry in collisions:
if entry is not keep:
selected.pop(entry["code"], None)
for entry in selected.values():
entry.pop("_distance", None)
return [selected[code] for code in FIELD_CODES if code in selected]
+51
View File
@@ -0,0 +1,51 @@
<!doctype html>
<html lang="de">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Fahrzeugschein · lokaler OCR-Vergleich</title>
<link rel="stylesheet" href="/style.css">
<script src="/app.js" defer></script>
</head>
<body>
<header class="top"><div class="wrap">
<p class="eyebrow">Lokaler Dokumenttest</p>
<h1>Fahrzeugschein-OCR im Vergleich</h1>
<p>Ein Bild, zwei unabhängige Verfahren. Werte erst anhand des Bildbelegs bestätigen.</p>
</div></header>
<main class="wrap">
<section class="panel input-panel" aria-labelledby="input-title">
<div class="section-head"><div><p class="eyebrow">01 / Eingabe</p><h2 id="input-title">Bild vorbereiten</h2></div>
<span class="privacy">Lokale Verarbeitung; Vision optional auf eigener Hardware</span></div>
<label class="upload-label" for="image-file">Fahrzeugschein auswählen</label>
<input id="image-file" type="file" accept="image/jpeg,image/png,.jpg,.jpeg,.png">
<p id="file-name" class="muted">JPEG oder PNG, maximal 16 MiB. Die Originaldatei wird nicht verändert.</p>
<div id="original-wrap" class="image-wrap" hidden>
<img id="original" alt="Ausgewähltes Originalbild mit veränderbaren Ecken">
<canvas id="corners" aria-label="Vier Dokumentecken"></canvas>
</div>
<div class="toolbar">
<button id="reset-corners" type="button" disabled>Ganzes Bild</button>
<label for="rotation">Leserichtung</label>
<select id="rotation"><option value="0">0°</option><option value="90">90° rechts</option>
<option value="180">180°</option><option value="270">90° links</option></select>
<button id="run" class="primary" type="button" disabled>Beide Verfahren vergleichen</button>
</div>
<p class="muted">Die vier blauen Punkte lassen sich mit Maus oder Finger an die äußeren Dokumentecken ziehen. Ohne Änderung läuft das ganze Bild. Beide Verfahren erhalten exakt dieselben aufbereiteten Pixel.</p>
<p id="status" class="status" role="status" aria-live="polite"></p>
</section>
<section id="comparison" class="panel" hidden aria-labelledby="compare-title">
<div class="section-head"><div><p class="eyebrow">02 / Belege</p><h2 id="compare-title">Ergebnis am Bild prüfen</h2></div>
<label class="toggle"><input id="all-boxes" type="checkbox"> Alle Textbereiche zeigen</label></div>
<div class="processed-wrap"><img id="processed" alt="Aufbereitetes Bild für beide OCR-Verfahren">
<canvas id="evidence" aria-label="Markierte Bildbelege"></canvas></div>
<p id="image-size" class="muted"></p>
<div class="results">
<article id="classic-card" class="method" aria-labelledby="classic-title"><h3 id="classic-title">A · Klassische OCR</h3></article>
<article id="vision-card" class="method" aria-labelledby="vision-title"><h3 id="vision-title">B · Vision-Dokumentmodell</h3></article>
</div>
</section>
</main>
</body>
</html>
+304
View File
@@ -0,0 +1,304 @@
"""Local-only, privacy-preserving comparison of classic OCR and PaddleOCR-VL."""
from __future__ import annotations
import asyncio
from io import BytesIO
import json
import os
from pathlib import Path
import shlex
import subprocess
import sys
import tempfile
import time
import warnings
import cv2
import numpy as np
from PIL import Image, ImageOps, UnidentifiedImageError
from starlette.applications import Starlette
from starlette.requests import Request
from starlette.responses import FileResponse, JSONResponse
from starlette.routing import Route
from .fields import FIELD_CODES, assign_fields
HERE = Path(__file__).resolve().parent
PROJECT = HERE.parent
MODEL_HOME = Path(os.environ.get("OCR_MODEL_HOME", PROJECT / ".cache")).expanduser().resolve()
def _remote_command(seconds: int = 180, *, tiles_only: bool = False) -> str:
remote_dir = os.environ["OCR_VISION_REMOTE_DIR"]
remote_python = os.environ.get("OCR_VISION_REMOTE_PYTHON", "./.venv-vision/bin/python")
runner = f'{shlex.quote(remote_python)} -m ocr_compare.vision_worker'
if tiles_only:
runner += ' --tiles-only'
isolation = 'unshare -Urn ' if os.environ.get("OCR_VISION_REMOTE_UNSHARE") == "1" else ""
device = shlex.quote(os.environ.get("OCR_VISION_REMOTE_DEVICE", "gpu:0"))
model_home = os.environ.get("OCR_VISION_REMOTE_MODEL_HOME")
model_env = f'OCR_MODEL_HOME={shlex.quote(model_home)} ' if model_home else ""
return (
f'set -eu; cd {shlex.quote(remote_dir)}; directory=$(mktemp -d); '
'trap \'rm -rf -- "$directory"\' EXIT; '
f'OCR_PRIVATE_DIR="$directory" OCR_VISION_DEVICE={device} {model_env}'
f'timeout -k 5s {seconds}s {isolation}{runner}'
)
MAX_UPLOAD = 16 * 1024 * 1024
MAX_PIXELS = 24_000_000
MAX_SIDE = 10_000
OCR_SIDE = 4000
Image.MAX_IMAGE_PIXELS = MAX_PIXELS
_slot = asyncio.Semaphore(1)
class InputError(ValueError):
pass
def _geometry(image: Image.Image, header: str | None) -> Image.Image:
try:
spec = json.loads(header) if header else {}
rotation = int(spec.get("rotation", 0))
if rotation not in (0, 90, 180, 270):
raise InputError("invalid_rotation")
points = spec.get("corners")
if points is not None:
if (not isinstance(points, list) or len(points) != 4 or
any(not isinstance(p, list) or len(p) != 2 or
any(not isinstance(v, (int, float)) or isinstance(v, bool) or
not 0 <= v <= 1 for v in p) for p in points)):
raise InputError("invalid_corners")
w, h = image.size
source = np.array([[x * (w - 1), y * (h - 1)] for x, y in points], np.float32)
area = cv2.contourArea(source)
edges = [float(np.linalg.norm(source[(i + 1) % 4] - source[i])) for i in range(4)]
turns = []
for i in range(4):
ax, ay = source[(i + 1) % 4] - source[i]
bx, by = source[(i + 2) % 4] - source[(i + 1) % 4]
turns.append(float(ax * by - ay * bx))
if area < .02 * w * h or min(edges) < 30 or min(turns) <= 0:
raise InputError("invalid_corners")
out_w = round(max(edges[0], edges[2]))
out_h = round(max(edges[1], edges[3]))
if out_w < 30 or out_h < 30 or out_w * out_h > MAX_PIXELS:
raise InputError("invalid_corners")
# The four positions are always TL, TR, BR, BL. This avoids the
# previous fixed-template aspect-ratio and rotation ambiguity.
target = np.array([[0, 0], [out_w - 1, 0],
[out_w - 1, out_h - 1], [0, out_h - 1]], np.float32)
matrix = cv2.getPerspectiveTransform(source, target)
warped = cv2.warpPerspective(np.asarray(image), matrix, (out_w, out_h),
flags=cv2.INTER_LINEAR,
borderMode=cv2.BORDER_CONSTANT,
borderValue=(255, 255, 255))
image = Image.fromarray(warped)
if rotation:
image = image.rotate(-rotation, expand=True)
if max(image.size) > OCR_SIDE:
image.thumbnail((OCR_SIDE, OCR_SIDE), Image.Resampling.LANCZOS)
return image
except (ValueError, TypeError, KeyError, json.JSONDecodeError) as exc:
if isinstance(exc, InputError):
raise
raise InputError("invalid_corners") from None
def _prepare(data: bytes, header: str | None) -> tuple[bytes, tuple[int, int]]:
try:
with warnings.catch_warnings():
warnings.simplefilter("error", Image.DecompressionBombWarning)
with Image.open(BytesIO(data)) as source:
if source.format not in ("JPEG", "PNG", "MPO"):
raise InputError("unsupported_encoding")
w, h = source.size
if min(w, h) < 1 or max(w, h) > MAX_SIDE or w * h > MAX_PIXELS:
raise InputError("image_too_large")
source.load()
image = ImageOps.exif_transpose(source).convert("RGB")
image = _geometry(image, header)
output = BytesIO()
image.save(output, format="JPEG", quality=92, subsampling=0)
return output.getvalue(), image.size
except InputError:
raise
except (OSError, ValueError, UnidentifiedImageError, Image.DecompressionBombError,
Image.DecompressionBombWarning):
raise InputError("invalid_image") from None
def _classic(data: bytes, directory: Path) -> dict:
started = time.monotonic()
models = MODEL_HOME / "official_models"
if any(not (models / name / "inference.pdiparams").is_file() for name in
("PP-OCRv5_mobile_det", "latin_PP-OCRv5_mobile_rec")):
return {"status": "error", "error": "classic_models_missing", "elapsed_ms": 0}
image_path = directory / "same-input.jpg"
output_path = directory / "classic.json"
image_path.write_bytes(data)
env = os.environ.copy()
env.update(PADDLE_PDX_CACHE_HOME=str(MODEL_HOME),
HF_HOME=str(MODEL_HOME / "hf"), HF_HUB_OFFLINE="1",
TRANSFORMERS_OFFLINE="1", PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK="True",
OMP_NUM_THREADS="1", OPENBLAS_NUM_THREADS="1", OMP_THREAD_LIMIT="1")
try:
process = subprocess.run(
[sys.executable, "-m", "ocr_compare.classic_worker", str(image_path), str(output_path)],
cwd=PROJECT, env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
timeout=120, check=False,
)
except subprocess.TimeoutExpired:
return {"status": "error", "error": "classic_timeout", "elapsed_ms": round((time.monotonic()-started)*1000)}
except OSError:
return {"status": "error", "error": "classic_failed", "elapsed_ms": round((time.monotonic()-started)*1000)}
elapsed = round((time.monotonic() - started) * 1000)
if process.returncode or not output_path.is_file():
return {"status": "error", "error": "classic_failed", "elapsed_ms": elapsed}
try:
raw = json.loads(output_path.read_text())
lines = [{"text": row["text"], "box": row["bbox"], "source": "OCR-Zeile"}
for row in raw["lines"] if row.get("bbox")]
except (OSError, ValueError, KeyError, TypeError):
return {"status": "error", "error": "classic_invalid", "elapsed_ms": elapsed}
return {"status": "ok", "model": "PP-OCRv5 mobile (Latin), lokal",
"elapsed_ms": elapsed, "lines": lines, "fields": assign_fields(lines),
"regions": []}
def _vision(data: bytes) -> dict:
started = time.monotonic()
mode = os.environ.get("OCR_VISION_MODE", "local").lower()
if mode == "off":
return {"status": "error", "error": "vision_disabled", "elapsed_ms": 0}
if mode not in ("local", "ssh"):
return {"status": "error", "error": "vision_config_error", "elapsed_ms": 0}
if mode == "local" and any(not path.is_file() for path in (
MODEL_HOME / "official_models" / "PP-DocLayoutV3" / "inference.pdiparams",
MODEL_HOME / "official_models" / "PaddleOCR-VL-1.6" / "model.safetensors")):
return {"status": "error", "error": "vision_models_missing", "elapsed_ms": 0}
if mode == "ssh" and (not os.environ.get("OCR_VISION_SSH_TARGET") or
not os.environ.get("OCR_VISION_REMOTE_DIR")):
return {"status": "error", "error": "vision_config_error", "elapsed_ms": 0}
with Image.open(BytesIO(data)) as page:
wide = page.width > 2500 and page.width / page.height > 1.65
limit = 90 if wide else 180
def run_worker(seconds: int, *, tiles_only: bool = False):
if mode == "ssh":
command = ["ssh", "-T", "-o", "BatchMode=yes", "-o", "StrictHostKeyChecking=yes",
"-o", "ConnectTimeout=10"]
identity = os.environ.get("OCR_VISION_SSH_IDENTITY")
if identity:
command += ["-i", str(Path(identity).expanduser()), "-o", "IdentitiesOnly=yes"]
command += [os.environ["OCR_VISION_SSH_TARGET"],
_remote_command(seconds, tiles_only=tiles_only)]
env = None
else:
command = [os.environ.get("OCR_VISION_PYTHON", sys.executable),
"-m", "ocr_compare.vision_worker"]
if tiles_only:
command.append("--tiles-only")
env = os.environ.copy()
env.update(OCR_MODEL_HOME=str(MODEL_HOME),
OCR_VISION_DEVICE=os.environ.get("OCR_VISION_DEVICE", "cpu"))
try:
with tempfile.TemporaryDirectory(prefix="ocr-vision-") as private:
if env is not None:
env["OCR_PRIVATE_DIR"] = private
return subprocess.run(command, input=data, stdout=subprocess.PIPE,
stderr=subprocess.DEVNULL, timeout=seconds + 15,
check=False, cwd=PROJECT if mode == "local" else None, env=env)
except subprocess.TimeoutExpired:
return None
except OSError:
return subprocess.CompletedProcess(command, 127, b"")
process = run_worker(limit)
if wide and (process is None or process.returncode == 124):
# Wide documents can form one prohibitively expensive layout block.
# The three generic views cover the same uploaded page without a
# document-specific field template.
process = run_worker(90, tiles_only=True)
elapsed = round((time.monotonic() - started) * 1000)
if process is None or process.returncode == 124:
return {"status": "error", "error": "vision_timeout", "elapsed_ms": elapsed}
if process.returncode or len(process.stdout) > 2_000_000:
return {"status": "error", "error": "vision_failed", "elapsed_ms": elapsed}
try:
raw = json.loads(process.stdout)
if raw.get("error"):
code = "vision_models_missing" if raw["error"] == "models_missing" else "vision_failed"
return {"status": "error", "error": code, "elapsed_ms": elapsed}
lines = raw["lines"]
regions = raw["regions"]
if not isinstance(lines, list) or not isinstance(regions, list):
raise ValueError("invalid collections")
except (ValueError, TypeError, KeyError):
return {"status": "error", "error": "vision_invalid", "elapsed_ms": elapsed}
return {"status": "ok", "model": raw.get("model", "PaddleOCR-VL"),
"elapsed_ms": elapsed, "lines": lines,
"fields": assign_fields(lines, allow_adjacent=False),
"regions": regions, "layout_note": raw.get("layout_note", ""),
"retry_used": bool(raw.get("retry_used", False))}
async def index(_: Request):
return FileResponse(HERE / "index.html", media_type="text/html; charset=utf-8")
async def script(_: Request):
return FileResponse(HERE / "app.js", media_type="text/javascript; charset=utf-8")
async def style(_: Request):
return FileResponse(HERE / "style.css", media_type="text/css; charset=utf-8")
async def health(_: Request):
return JSONResponse({"ready": True, "scope": "localhost only",
"vision_mode": os.environ.get("OCR_VISION_MODE", "local")})
async def compare(request: Request):
if _slot.locked():
return JSONResponse({"error": "busy"}, status_code=429)
if request.headers.get("content-type", "").split(";", 1)[0] not in ("image/jpeg", "image/png"):
return JSONResponse({"error": "unsupported_media_type"}, status_code=415)
async with _slot:
data = bytearray()
try:
async with asyncio.timeout(30):
async for chunk in request.stream():
data.extend(chunk)
if len(data) > MAX_UPLOAD:
return JSONResponse({"error": "upload_too_large"}, status_code=413)
except TimeoutError:
return JSONResponse({"error": "upload_timeout"}, status_code=408)
try:
image, (width, height) = await asyncio.to_thread(
_prepare, bytes(data), request.headers.get("x-ocr-geometry"))
except InputError as exc:
return JSONResponse({"error": str(exc)}, status_code=422)
with tempfile.TemporaryDirectory(prefix="ocr-compare-") as path:
directory = Path(path)
classic, vision = await asyncio.gather(
asyncio.to_thread(_classic, image, directory),
asyncio.to_thread(_vision, image),
)
# Exact JPEG bytes sent to both pipelines are returned for evidence
# highlighting. They never enter a report, log, or external OCR API.
import base64
return JSONResponse({"image": "data:image/jpeg;base64," + base64.b64encode(image).decode(),
"image_size": {"width": width, "height": height},
"field_codes": FIELD_CODES,
"results": {"classic": classic, "vision": vision}})
app = Starlette(routes=[Route("/", index), Route("/app.js", script),
Route("/style.css", style), Route("/api/health", health),
Route("/api/compare", compare, methods=["POST"])])
+71
View File
@@ -0,0 +1,71 @@
"""Download official model weights and verify them on a synthetic page."""
from __future__ import annotations
import argparse
import os
from pathlib import Path
import tempfile
from PIL import Image, ImageDraw
ROOT = Path(os.environ.get("OCR_MODEL_HOME", Path(__file__).resolve().parents[1] / ".cache")).expanduser().resolve()
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--classic", action="store_true")
parser.add_argument("--vision", action="store_true")
parser.add_argument("--device", default="cpu", help="Vision device, for example cpu or gpu:0")
args = parser.parse_args()
if not args.classic and not args.vision:
parser.error("choose --classic and/or --vision")
ROOT.mkdir(parents=True, exist_ok=True)
os.environ["PADDLE_PDX_CACHE_HOME"] = str(ROOT)
os.environ["HF_HOME"] = str(ROOT / "hf")
os.environ.pop("HF_HUB_OFFLINE", None)
os.environ.pop("TRANSFORMERS_OFFLINE", None)
with tempfile.TemporaryDirectory(prefix="ocr-synthetic-setup-") as temporary:
image = Path(temporary) / "synthetic.png"
page = Image.new("RGB", (1000, 600), "white")
draw = ImageDraw.Draw(page)
draw.text((50, 50), "SYNTHETIC TEST / NO PERSONAL DATA", fill="black")
draw.text((50, 130), "D.3 SAMPLE 123", fill="black")
page.save(image)
if args.classic:
from paddleocr import PaddleOCR
engine = PaddleOCR(
text_detection_model_name="PP-OCRv5_mobile_det",
text_recognition_model_name="latin_PP-OCRv5_mobile_rec",
use_doc_orientation_classify=False,
use_doc_unwarping=False,
use_textline_orientation=False,
device="cpu",
)
list(engine.predict(str(image)))
if args.vision:
from paddleocr import PaddleOCRVL
engine = PaddleOCRVL(
pipeline_version="v1.6", use_layout_detection=True,
use_doc_orientation_classify=False, use_doc_unwarping=False,
device=args.device,
)
list(engine.predict(str(image)))
models = ROOT / "official_models"
checks = []
if args.classic:
checks.extend((models / name / "inference.pdiparams" for name in
("PP-OCRv5_mobile_det", "latin_PP-OCRv5_mobile_rec")))
if args.vision:
checks.extend((models / "PP-DocLayoutV3" / "inference.pdiparams",
models / "PaddleOCR-VL-1.6" / "model.safetensors"))
if any(not path.is_file() for path in checks):
raise SystemExit("Official model download did not create all expected files")
print("Official models ready; only synthetic pixels were used")
if __name__ == "__main__":
main()
+72
View File
@@ -0,0 +1,72 @@
:root {font-family:system-ui,-apple-system,BlinkMacSystemFont,"Segoe UI",sans-serif;color:#16233a;background:#f3f6f8;line-height:1.45}
* {box-sizing:border-box}
body {margin:0}
.wrap {width:min(1240px,calc(100% - 32px));margin:auto}
.top {background:#133345;color:white;padding:24px 0 28px}
h1,h2,h3,h4,p {margin:0}
h1 {font-size:clamp(1.6rem,3vw,2.5rem);letter-spacing:-.035em}
h2 {font-size:1.35rem;letter-spacing:-.025em}
h3 {font-size:1.1rem}
h4 {font-size:.95rem;margin:22px 0 8px}
.top p:last-child {color:#d6e6e9;margin-top:5px}
.eyebrow {font-size:.72rem;text-transform:uppercase;font-weight:800;letter-spacing:.13em;color:#238476;margin-bottom:5px}
.top .eyebrow {color:#8be0c8}
main {padding:22px 0 50px}
.panel {background:white;border:1px solid #dbe5e9;border-radius:14px;padding:22px;margin-bottom:20px;box-shadow:0 6px 30px rgba(20,46,64,.045)}
.section-head {display:flex;align-items:start;justify-content:space-between;gap:18px;margin-bottom:18px}
.privacy {font-size:.76rem;font-weight:700;color:#177868;background:#e9f6f1;border-radius:100px;padding:6px 11px;white-space:nowrap}
.upload-label {display:block;font-weight:700;margin-bottom:6px}
input[type=file] {display:block;width:100%;padding:12px;border:1px dashed #9eb5c0;background:#f8fbfc;border-radius:9px;color:#20384a}
.muted {color:#617585;font-size:.87rem;margin-top:9px}
.image-wrap,.processed-wrap {position:relative;display:block;width:100%;margin-top:18px;background:#e9eff1;border:1px solid #ccd9df;border-radius:8px;overflow:hidden}
.image-wrap[hidden],#comparison[hidden] {display:none}
.image-wrap img,.processed-wrap img {display:block;width:100%;height:auto}
.image-wrap canvas,.processed-wrap canvas {position:absolute;inset:0;width:100%;height:100%}
.image-wrap canvas {touch-action:none;cursor:crosshair}
.processed-wrap canvas {pointer-events:none}
.toolbar {display:flex;align-items:center;gap:10px;flex-wrap:wrap;margin-top:15px}
.toolbar label {font-size:.88rem;font-weight:700;margin-left:auto}
button,select,input[type=text] {font:inherit}
button,select {border-radius:8px;border:1px solid #9bb4bf;background:white;color:#193949;padding:9px 12px}
button {cursor:pointer;font-weight:650}
button:hover:not(:disabled) {background:#e8f2f3}
button:disabled {cursor:not-allowed;opacity:.52}
.primary {background:#007d73;color:white;border-color:#007d73}
.primary:hover:not(:disabled) {background:#00695f}
.status {min-height:1.4em;margin-top:12px;font-weight:650}
.status.error,.error {color:#ac273c}
.status.ok {color:#08786d}
.toggle {font-size:.85rem;color:#3d5667;white-space:nowrap}
.results {display:grid;grid-template-columns:repeat(2,minmax(0,1fr));gap:16px;margin-top:20px}
.method {border:1px solid #d4e2e6;border-radius:11px;padding:17px;min-width:0}
.method:first-child {border-top:4px solid #087c75}
.method:last-child {border-top:4px solid #465fc7}
.meta {font-size:.84rem;color:#526d7b;margin-top:5px}
.note {font-size:.82rem;color:#815b13;background:#fff8e5;padding:9px;border-radius:6px;margin-top:10px}
.assignment-note {font-size:.82rem;color:#385469;background:#eef5f8;padding:9px;border-radius:6px;margin-top:10px}
.field-list {display:grid;gap:7px}
.field-item {display:grid;grid-template-columns:minmax(0,1fr) auto;gap:6px;padding:3px;background:#f8fbfc;border:1px solid #dce7ea;border-radius:8px;overflow-wrap:anywhere}
.field-item.needs-review {background:#fff9eb;border-color:#e4c886}
.field-show {display:grid;grid-template-columns:minmax(88px,110px) minmax(0,1fr);gap:2px 8px;width:100%;text-align:left;border:0;background:transparent;padding:6px 7px}
.field-heading {grid-row:1 / 3;display:flex;flex-direction:column;gap:2px}
.field-heading strong {color:#126d65}
.field-heading small,.field-method {font-size:.73rem;color:#607783}
.field-value {font-weight:650;overflow-wrap:anywhere;white-space:pre-line}
.priority-fields {display:flex;flex-wrap:wrap;gap:6px;margin:8px 0 12px}
.priority-fields h5 {flex-basis:100%;margin:0;font-size:.9rem}
.priority-item {min-width:0;padding:3px 7px;border-radius:6px;background:#e8f6ef;color:#155d3d;font-size:.83rem;overflow-wrap:anywhere}
.priority-item.missing {background:#fff2e9;color:#8a431d}
.field-method {grid-column:2}
.field-dismiss {align-self:center;font-size:.73rem;padding:6px 7px;color:#6d4050;border-color:#d9c9ce;background:white}
.raw-lines {max-height:280px;overflow:auto;border:1px solid #e3ecef;border-radius:8px;padding:6px;display:grid;gap:4px}
.raw-entry {display:grid;grid-template-columns:minmax(0,1fr) auto;gap:4px}
.raw-line {display:flex;align-items:start;text-align:left;gap:9px;width:100%;padding:8px 10px;background:#f8fbfc;border-color:#dce7ea;overflow-wrap:anywhere;font-weight:400;font-size:.82rem}
.raw-line span {color:#617783;min-width:28px}
.raw-line span:last-child {color:#193949;min-width:0}
.raw-pick {font-size:.72rem;padding:5px 7px}
.manual {display:grid;grid-template-columns:minmax(100px,1fr) minmax(100px,2fr);gap:7px}
.manual input {grid-column:1 / -1;border:1px solid #9bb4bf;border-radius:8px;padding:9px;width:100%}
.manual button {justify-self:start}
.manual-hint {grid-column:1 / -1;color:#617585;font-size:.78rem}
@media(max-width:800px) {.results {grid-template-columns:1fr}.privacy {white-space:normal}.toolbar label {margin-left:0}}
@media(max-width:520px) {.wrap {width:min(100% - 20px,1240px)}.panel {padding:14px}.section-head {display:block}.privacy {display:inline-block;margin-top:10px}.toolbar {display:grid;grid-template-columns:1fr 1fr}.toolbar .primary {grid-column:1/-1}.toolbar label {align-self:center}.toggle {display:block;margin-top:10px}.field-show {grid-template-columns:1fr}.field-heading {grid-row:auto}.field-method {grid-column:1}.raw-entry {grid-template-columns:minmax(0,1fr)}.raw-pick {justify-self:end}}
+310
View File
@@ -0,0 +1,310 @@
"""Offline worker for the full official PaddleOCR-VL layout pipeline.
The caller sends JPEG bytes on stdin. This process uses explicit local model
directories; it writes a mode-0600 temporary file and deletes it afterward.
No private images or OCR results are saved to the model cache or logs.
"""
from __future__ import annotations
import json
import os
from html.parser import HTMLParser
from pathlib import Path
import re
import sys
import tempfile
from PIL import Image
ROOT = Path(os.environ.get("OCR_MODEL_HOME", Path(__file__).resolve().parents[1] / ".cache")).expanduser().resolve()
os.environ["PADDLE_PDX_CACHE_HOME"] = str(ROOT)
os.environ["HF_HOME"] = str(ROOT / "hf")
os.environ["HF_HUB_OFFLINE"] = "1"
os.environ["TRANSFORMERS_OFFLINE"] = "1"
os.environ["PADDLE_PDX_DISABLE_MODEL_SOURCE_CHECK"] = "True"
os.environ["OMP_NUM_THREADS"] = "1"
class _TableRows(HTMLParser):
def __init__(self):
super().__init__()
self.rows = []
self.row = None
self.cell = None
def handle_starttag(self, tag, attrs):
if tag == "tr":
self.row = []
elif tag in ("td", "th") and self.row is not None:
self.cell = []
elif tag == "br" and self.cell is not None:
self.cell.append(" ")
def handle_data(self, data):
if self.cell is not None:
self.cell.append(data)
def handle_endtag(self, tag):
if tag in ("td", "th") and self.cell is not None and self.row is not None:
self.row.append(" ".join("".join(self.cell).split())[:500])
self.cell = None
elif tag == "tr" and self.row is not None:
if self.row:
self.rows.append(self.row)
self.row = None
def _table_rows(content: str) -> list[list[str]]:
parser = _TableRows()
parser.feed(content)
parser.close()
return parser.rows
def _box(value, width: int, height: int, offset=(0, 0)):
if value is None or len(value) != 4:
return None
try:
x1, y1, x2, y2 = (float(item) for item in value)
except (TypeError, ValueError):
return None
if not (0 <= x1 < x2 <= width and 0 <= y1 < y2 <= height):
return None
ox, oy = offset
return [round(x1 + ox, 1), round(y1 + oy, 1),
round(x2 + ox, 1), round(y2 + oy, 1)]
def extract(payload: dict, offset=(0, 0)) -> dict:
"""Keep original block text and its actual layout region, never fake line boxes."""
width, height = int(payload["width"]), int(payload["height"])
lines, regions = [], []
for block in payload.get("parsing_res_list", []):
box = _box(block.get("block_bbox"), width, height, offset)
label = str(block.get("block_label", "region"))[:80]
content = block.get("block_content", "")
if not isinstance(content, str):
content = str(content)
if box:
regions.append({"label": label, "box": box})
if label == "table" and "<tr" in content.lower():
for cells in _table_rows(content):
text = " | ".join(cells)
if text.strip():
lines.append({"text": text[:1000], "cells": cells,
"box": box, "source": "PaddleOCR-VL-Tabellenzeile; nur Tabellenregion als Bildbeleg"})
else:
for part in content.splitlines():
text = part.strip()
if text:
lines.append({"text": text[:1000], "box": box,
"source": f"PaddleOCR-VL-Layoutblock {label}; keine eigene Zeilenbox"})
return {
"model": "PaddleOCR-VL 1.6 · vollständige lokale Layoutpipeline",
"lines": lines[:500], "regions": regions[:200],
"layout_note": "Vision-Bildbelege markieren echte Layoutregionen. Mehrere Textzeilen können dieselbe Region teilen; die Region ist keine präzise Feldbox.",
}
def _retry_table(block: dict, width: int, height: int) -> bool:
if block.get("block_label") != "table":
return False
box = _box(block.get("block_bbox"), width, height)
if not box:
return False
x1, y1, x2, y2 = box
area = (x2 - x1) * (y2 - y1) / (width * height)
content = block.get("block_content", "")
if not isinstance(content, str):
return False
visible_text = re.sub(r"<[^>]*>", "", content).strip()
return area > .45 and len(visible_text) < 80
def _split_table(box: list[float], width: int, height: int) -> list[tuple[int, int, int, int]]:
x1, y1, x2, y2 = box
left = max(0, int(x1) - 20)
top = max(0, int(y1) - 20)
right = min(width, int(x2) + 20)
bottom = min(height, int(y2) + 20)
middle = (left + right) // 2
overlap = min(120, max(40, round((right - left) * .045)))
return [(left, top, min(right, middle + overlap), bottom),
(max(left, middle - overlap), top, right, bottom)]
def _missing_neighbor_box(block: dict, blocks: list[dict], width: int, height: int):
"""Find an uncovered page column directly left of a detected right table."""
if block.get("block_label") != "table":
return None
box = _box(block.get("block_bbox"), width, height)
if not box:
return None
x1, y1, x2, y2 = box
table_width, table_height = x2 - x1, y2 - y1
if x1 < .52 * width or table_width > .4 * width or table_height < .2 * height:
return None
left = max(0, round(x1 - 1.2 * table_width))
right = min(width, round(x1 + .07 * table_width))
top = max(0, round(y1 - .1 * table_height))
bottom = min(height, round(y2 + .3 * table_height))
if right - left < 300 or bottom - top < 300:
return None
for other in blocks:
if other is block or other.get("block_label") != "table":
continue
other_box = _box(other.get("block_bbox"), width, height)
if not other_box:
continue
a, b, c, d = other_box
overlap = max(0, min(x1, c) - max(left, a)) * max(0, min(bottom, d) - max(top, b))
if overlap > .15 * (x1 - left) * (bottom - top):
return None
return (left, top, right, bottom)
def _run_tile(pipeline, image: Image.Image, crop_box, private: Path):
descriptor, name = tempfile.mkstemp(prefix="table-", suffix=".jpg", dir=private)
tile_path = Path(name)
try:
with os.fdopen(descriptor, "wb") as stream:
image.crop(crop_box).save(stream, format="JPEG", quality=92)
results = list(pipeline.predict(str(tile_path)))
if len(results) != 1:
return None
payload = results[0].json
return extract(payload.get("res", payload), crop_box[:2])
except Exception:
# Keep the original page result if a diagnostic retry fails.
return None
finally:
tile_path.unlink(missing_ok=True)
def predict_tiled(pipeline, path: Path, private: Path) -> dict:
"""Generic full-page coverage for a wide page whose full parsing timed out."""
with Image.open(path) as image:
width, height = image.size
boxes = [(0, 0, round(width * .34), height),
(round(width * .25), 0, round(width * .66), height),
(round(width * .61), 0, width, height)]
tiles = [_run_tile(pipeline, image, box, private) for box in boxes]
valid = [tile for tile in tiles if tile and tile["lines"]]
if not valid:
raise RuntimeError("tile_inference_failed")
result = extract({"width": width, "height": height, "parsing_res_list": []})
for tile in valid:
result["lines"].extend(tile["lines"])
result["regions"].extend(tile["regions"])
result["lines"] = result["lines"][:500]
result["regions"] = result["regions"][:200]
result["retry_used"] = True
result["layout_note"] = (
f"Die Gesamtansicht brauchte zu lange. {len(valid)} von 3 überlappenden "
"Ansichten derselben Seite wurden lokal gelesen; widersprüchliche "
"Feldwerte bleiben ohne Zuordnung. Tabellenbelege sind grobe Regionen.")
return result
def predict_adaptive(pipeline, path: Path, private: Path) -> dict:
results = list(pipeline.predict(str(path)))
if len(results) != 1:
raise RuntimeError("invalid_page_count")
original = results[0].json
payload = original.get("res", original)
width, height = int(payload["width"]), int(payload["height"])
blocks = payload.get("parsing_res_list", [])
large_table = next((block for block in blocks if _retry_table(block, width, height)), None)
neighbor_box = next((box for block in blocks
if (box := _missing_neighbor_box(block, blocks, width, height))), None)
if large_table is None and neighbor_box is None:
return extract(payload)
# A large, almost empty table is a measurable layout/scale failure. Retry
# only that detected region in two smaller views; all pixels still come
# from the same upload and all coordinates map back to the common image.
tiles = []
large_recovered = False
neighbor_recovered = False
with Image.open(path) as image:
if large_table is not None:
box = _box(large_table["block_bbox"], width, height)
for crop_box in _split_table(box, width, height):
parsed = _run_tile(pipeline, image, crop_box, private)
if parsed and parsed["lines"]:
tiles.append(parsed)
large_recovered = True
if neighbor_box is not None:
parsed = _run_tile(pipeline, image, neighbor_box, private)
if parsed and parsed["lines"]:
tiles.append(parsed)
neighbor_recovered = True
if not tiles:
return extract(payload)
kept = [block for block in blocks if block is not large_table] if large_recovered else blocks
result = extract({**payload, "parsing_res_list": kept})
for tile in tiles:
result["lines"].extend(tile["lines"])
result["regions"].extend(tile["regions"])
result["lines"] = result["lines"][:500]
result["regions"] = result["regions"][:200]
result["retry_used"] = True
notes = []
if large_recovered:
notes.append("Ein übergroßer, fast leerer Tabellenblock wurde in zwei kleineren Ansichten erneut gelesen.")
if neighbor_recovered:
notes.append("Eine von der Layoutstufe ausgelassene Nachbarspalte wurde zusätzlich gelesen.")
notes.append("Tabellenzeilen haben weiterhin nur einen groben Regionsbeleg.")
result["layout_note"] = " ".join(notes)
return result
def main() -> None:
data = sys.stdin.buffer.read(16 * 1024 * 1024 + 1)
if not data or len(data) > 16 * 1024 * 1024:
print(json.dumps({"error": "invalid_input"}))
return
models = ROOT / "official_models"
layout = models / "PP-DocLayoutV3"
vlm = models / "PaddleOCR-VL-1.6"
if not (layout / "inference.pdiparams").is_file() or not (vlm / "model.safetensors").is_file():
print(json.dumps({"error": "models_missing"}))
return
private = Path(os.environ.get("OCR_PRIVATE_DIR", ROOT / "private"))
private.mkdir(mode=0o700, exist_ok=True)
descriptor, name = tempfile.mkstemp(prefix="page-", suffix=".jpg", dir=private)
path = Path(name)
try:
with os.fdopen(descriptor, "wb") as stream:
stream.write(data)
# Paddle logs should never expose text or private file paths through
# the SSH result channel. The app suppresses stderr as well.
stdout_copy = os.dup(1)
null_fd = os.open(os.devnull, os.O_WRONLY)
os.dup2(null_fd, 1)
os.close(null_fd)
try:
from paddleocr import PaddleOCRVL
pipeline = PaddleOCRVL(pipeline_version="v1.6", use_layout_detection=True,
layout_detection_model_dir=str(layout), vl_rec_model_dir=str(vlm),
use_doc_orientation_classify=False,
use_doc_unwarping=False,
device=os.environ.get("OCR_VISION_DEVICE", "cpu"))
result = (predict_tiled(pipeline, path, private)
if "--tiles-only" in sys.argv[1:] else
predict_adaptive(pipeline, path, private))
finally:
os.dup2(stdout_copy, 1)
os.close(stdout_copy)
print(json.dumps(result, ensure_ascii=False, separators=(",", ":")))
except Exception:
# No exception text: model libraries may include recognized content.
print(json.dumps({"error": "inference_failed"}))
finally:
path.unlink(missing_ok=True)
if __name__ == "__main__":
main()