Files

357 lines
17 KiB
Python
Raw Permalink Normal View History

2026-10-07 21:00:21 +02:00
"""Assign observed OCR text to ZB Teil I fields with image evidence.
No value is generated. Weak code inferences from neighboring observed fields
are explicit and remain reviewable in the UI.
"""
from __future__ import annotations
from difflib import SequenceMatcher
import re
FIELD_CODES = tuple("""A C.1.1 C.1.2 C.1.3 X I B 2.1 2.2 J 4 E 3 D.1 D.2 D.3 2 5
V.9 14 P.3 10 14.1 P.1 L 9 P.2/P.4 T 18 19 20 G 12 13 Q V.7 F.1 F.2
7.1 7.2 7.3 8.1 8.2 8.3 U.1 U.2 U.3 O.1 O.2 S.1 S.2 15.1 15.2 15.3
R 11 K 6 17 16 21 22""".split())
def _norm(text: str) -> str:
return re.sub(r"[^A-Z0-9]", "", text.upper())
def _code(text: str) -> str | None:
# Exact dotted labels win. Dot loss is accepted only when it is
# unambiguous; arbitrary words and strings with values are not labels.
literal = text.strip().upper().strip("() ")
if literal in ("21", "22"):
# Without the dot, 2.1/2.2 and 21/22 cannot be distinguished.
return None
if literal in FIELD_CODES:
return literal
token = _norm(text)
matches = [code for code in FIELD_CODES if _norm(code) == token]
return matches[0] if len(matches) == 1 else None
_OWNER_CAPTIONS = {
"C.1.1": re.compile(r"\b(?:name|firmenname)\b", re.I),
"C.1.2": re.compile(r"\bvorname", re.I),
"C.1.3": re.compile(r"\b(?:anschrift|adresse)\b", re.I),
}
_OWNER_PRINTED = {"C.1.1": "NAMEODERFIRMENNAME", "C.1.2": "VORNAMEN",
"C.1.3": "ANSCHRIFT"}
def _owner_caption(text: str) -> str | None:
match = re.match(r"^\s*\(?\s*C\s*\.?\s*1\s*\.?\s*([123])\s*\)?\s*(.*)$", text, re.I)
if match:
code = f"C.1.{match[1]}"
tail = match[2].strip(" \t:;=-")
if not tail or _OWNER_CAPTIONS[code].search(tail):
return code
letters = re.sub(r"[^A-Z]", "", tail.upper())
if len(letters) >= 6 and SequenceMatcher(None, letters, _OWNER_PRINTED[code]).ratio() >= .62:
return code
# Small printed codes can be lost while the full caption survives.
# Use only the distinctive short printed wording, never a person's name.
if len(text) < 55:
if re.search(r"\bName\s+oder\s+Firmenname\b", text, re.I):
return "C.1.1"
if re.search(r"\bVorname\b", text, re.I):
return "C.1.2"
if re.search(r"\bAnschrift\b", text, re.I):
return "C.1.3"
return None
def _height(box: list[float]) -> float:
return box[3] - box[1]
def _vertical_overlap(a: list[float], b: list[float]) -> float:
return max(0, min(a[3], b[3]) - max(a[1], b[1]))
def _nearby_value(anchor: dict, lines: list[dict], labels: set[int]) -> tuple[int, float] | None:
ax1, ay1, ax2, ay2 = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if (index in labels or line is anchor or not line.get("box") or
not line["text"].strip(" \t-–—.,")):
continue
bx1, by1, bx2, by2 = line["box"]
bh = _height(line["box"])
if bx2 <= (ax1 + ax2) / 2 or bx1 < ax1 - 0.2 * ah:
continue
if bx1 - ax2 > max(120, 5 * ah):
continue
overlap = _vertical_overlap(anchor["box"], line["box"])
if overlap < 0.45 * min(ah, bh):
continue
center_gap = abs((by1 + by2) / 2 - (ay1 + ay2) / 2)
# Prefer the closest value to the right; reject rows that only touch
# because large OCR boxes extend into a neighbouring printed row.
candidates.append((max(0, bx1 - ax2) + center_gap * 1.5, index))
if not candidates:
return None
candidates.sort()
return candidates[0][1], candidates[0][0]
def _owner_below(anchor_index: int, lines: list[dict], labels: set[int]) -> dict | None:
"""Read holder value rows below an observed caption, within its column."""
anchor = lines[anchor_index]
code = _owner_caption(anchor["text"])
if "Layoutblock" in anchor.get("source", ""):
# The VL parser sometimes puts a printed caption and its value in
# consecutive text lines of one coarse region. Their box is shared;
# line order is the only honest evidence for this relation.
same_region = []
for index in range(anchor_index + 1, min(len(lines), anchor_index + (3 if code == "C.1.3" else 2))):
line = lines[index]
if index in labels or line.get("box") != anchor["box"] or not line["text"].strip():
break
same_region.append(index)
if same_region:
value = "\n".join(lines[index]["text"].strip() for index in same_region)
return {"value": value[:500], "raw_text": value, "box": anchor["box"],
"anchor_box": anchor["box"], "line_indices": [anchor_index, *same_region],
"method": "caption_region_next_line", "confidence": None}
ax1, _, _, ay2 = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if index in labels or not line.get("box") or not line["text"].strip():
continue
x1, y1, x2, y2 = line["box"]
if not (ay2 - .15 * ah <= y1 <= ay2 + max(85, 2.8 * ah)):
continue
if not (ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)):
continue
if x2 <= ax1 or re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I):
continue
candidates.append((max(0, y1 - ay2), abs(x1 - ax1), index))
if not candidates:
return None
candidates.sort()
first = lines[candidates[0][2]]["box"]
first_indices = [index for index, line in enumerate(lines)
if (index not in labels or line["text"].strip().isdigit())
and line.get("box") and line["text"].strip()
and line["box"][2] > ax1
and _vertical_overlap(first, line["box"]) >=
.35 * min(_height(first), _height(line["box"]))]
first_indices.sort(key=lambda index: lines[index]["box"][0])
# Only join fragments that touch the same printed value row. A remote
# table column with a large gap is not part of the holder's name.
row = [min((index for index in first_indices
if ax1 - max(65, 1.5 * ah) <= lines[index]["box"][0]
<= ax1 + max(170, 3 * ah)),
key=lambda index: lines[index]["box"][0], default=candidates[0][2])]
for index in first_indices:
if index == row[0] or lines[index]["box"][0] < lines[row[0]]["box"][0]:
continue
previous = lines[row[-1]]["box"]
box = lines[index]["box"]
if box[0] - previous[2] <= max(45, .8 * _height(first)):
row.append(index)
rows = [row]
if code == "C.1.3" and "Layoutblock" not in anchor.get("source", ""):
bottom = max(lines[index]["box"][3] for index in row)
row_height = max(_height(lines[index]["box"]) for index in row)
next_row = []
for index, line in enumerate(lines):
if ((index in labels and not line["text"].strip().isdigit())
or index in row or not line.get("box") or not line["text"].strip()):
continue
x1, y1, x2, _ = line["box"]
if (bottom - .1 * row_height <= y1 <= bottom + max(25, .55 * row_height)
and x1 >= ax1 - max(65, 1.5 * ah)
and x2 > ax1
and not re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I)):
next_row.append(index)
next_row.sort(key=lambda index: lines[index]["box"][0])
if next_row:
starter = next((index for index in next_row
if lines[index]["box"][0] <= ax1 + max(170, 3 * ah)), None)
contiguous = [starter] if starter is not None else []
for index in next_row:
if not contiguous or index == starter or lines[index]["box"][0] < lines[starter]["box"][0]:
continue
if lines[index]["box"][0] - lines[contiguous[-1]]["box"][2] <= max(45, .8 * row_height):
contiguous.append(index)
if contiguous:
rows.append(contiguous)
indices = [index for row in rows for index in row]
value = "\n".join("\n".join(lines[index]["text"].strip() for index in row)
if len(row) > 1 and all(lines[index]["box"] == lines[row[0]]["box"] for index in row)
else " ".join(lines[index]["text"].strip() for index in row)
for row in rows)
boxes = [lines[index]["box"] for index in indices]
box = [min(b[0] for b in boxes), min(b[1] for b in boxes),
max(b[2] for b in boxes), max(b[3] for b in boxes)]
return {"value": value[:500], "raw_text": value, "box": box,
"anchor_box": anchor["box"], "line_indices": [anchor_index, *indices],
"method": "below_caption", "confidence": None}
def _surname_above_forename(lines: list[dict], owner_labels: dict[int, str],
labels: set[int]) -> dict | None:
"""Weak suggestion when C.1.1 is unreadable but C.1.2 is observed."""
if "C.1.1" in owner_labels.values():
return None
forename = [index for index, code in owner_labels.items() if code == "C.1.2"]
if len(forename) != 1:
return None
anchor_index = forename[0]
anchor = lines[anchor_index]
ax1, ay1, _, _ = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if index in labels or not line.get("box"):
continue
text = line["text"].strip()
if len(re.sub(r"[^A-ZÄÖÜ]", "", text.upper())) < 3 or any(c.isdigit() for c in text):
continue
x1, _, x2, y2 = line["box"]
gap = ay1 - y2
if (0 <= gap <= max(150, 3 * ah)
and ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)
and x2 > ax1):
candidates.append((gap, abs(x1 - ax1), index))
candidates.sort()
if not candidates or (len(candidates) > 1 and candidates[1][0] - candidates[0][0] < ah):
return None
value_index = candidates[0][2]
value_line = lines[value_index]
return {"code": "C.1.1", "value": value_line["text"].strip()[:500],
"raw_text": value_line["text"], "box": value_line["box"],
"anchor_box": anchor["box"], "line_indices": [anchor_index, value_index],
"method": "surname_above_forename", "confidence": None}
def assign_fields(lines: list[dict], *, allow_adjacent: bool = True) -> list[dict]:
"""Return only observed, localized values; duplicates stay in raw lines."""
proposed: dict[str, list[dict]] = {}
labels = {index: code for index, line in enumerate(lines)
if line.get("box") and (code := _code(line["text"]))}
owner_labels = {index: code for index, line in enumerate(lines)
if line.get("box") and (code := _owner_caption(line["text"]))}
label_indices = set(labels) | set(owner_labels)
for index, line in enumerate(lines):
box = line.get("box")
if not box:
continue
if isinstance(line.get("cells"), list):
cells = line["cells"]
contextual = {}
if (len(cells) >= 6 and _code(cells[0]) == "B"
and _norm(cells[2]) == "21" and _norm(cells[4]) == "22"
and _code(cells[2]) is None and _code(cells[4]) is None):
# In one observed table row, the printed B / 2.1 / 2.2
# sequence can lose both tiny dots. Keep this weaker reading
# explicit and reviewable; never reinterpret a lone 21/22.
contextual = {2: "2.1", 4: "2.2"}
consumed_values: set[int] = set()
for cell_index, cell in enumerate(cells[:-1]):
if cell_index in consumed_values:
continue
code = (contextual.get(cell_index) or _code(cell)) if isinstance(cell, str) else None
value = cells[cell_index + 1] if isinstance(cells[cell_index + 1], str) else ""
value = value.strip()
if (code and value and value not in ("-", "–", "—")
and _code(value) is None
and not (code in _OWNER_CAPTIONS and _OWNER_CAPTIONS[code].search(value))):
consumed_values.add(cell_index + 1)
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": line["text"],
"box": box, "anchor_box": box, "line_indices": [index],
"method": "vl_table_context_code" if cell_index in contextual else "vl_table_adjacent_cell",
"confidence": None,
})
continue
text = line["text"].strip()
# Code and value in the same OCR line.
matched = False
for code in sorted(FIELD_CODES, key=len, reverse=True):
if code.startswith("C.1."):
# These printed captions precede the holder data on another row.
continue
printed = re.escape(code)
marker = (rf"(?:\({printed}\)|{printed}(?=\s*[:;=\-]))"
if len(code) == 1 or code.isdigit() else
rf"(?:\({printed}\)|{printed}(?=\s|[:;=\-]))")
match = re.match(rf"^\s*{marker}\s*[:;=\-]?\s*", text, flags=re.I)
if match and (value := text[match.end():].strip(" \t:;=-")) and value.strip("-–—.,"):
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": text,
"box": box, "anchor_box": box, "line_indices": [index],
"method": "same_line_code", "confidence": None,
})
matched = True
break
if not matched:
# Small printed codes can merge with a value in one OCR box.
# Keep the original box and mark this weaker reading for review.
for code in sorted((c for c in FIELD_CODES if re.fullmatch(r"[A-Z]\.\d(?:\.\d)?", c)),
key=len, reverse=True):
marker = re.escape(code).replace(r"\.", r"\.?" )
match = re.match(rf"^\s*{marker}(?=[A-Z0-9])", text, flags=re.I)
if match and (value := text[match.end():].strip(" \t:;=-")):
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": text,
"box": box, "anchor_box": box, "line_indices": [index],
"method": "merged_code", "confidence": None,
})
break
for index, code in (labels.items() if allow_adjacent else []):
if code in _OWNER_CAPTIONS:
continue
nearby = _nearby_value(lines[index], lines, label_indices)
if nearby is None:
continue
value_index, distance = nearby
value_line = lines[value_index]
proposed.setdefault(code, []).append({
"code": code, "value": value_line["text"].strip()[:500],
"raw_text": value_line["text"], "box": value_line["box"],
"anchor_box": lines[index]["box"], "line_indices": [index, value_index],
"method": "adjacent_code", "confidence": None, "_distance": distance,
})
for index, code in owner_labels.items():
below = _owner_below(index, lines, label_indices)
if below:
proposed.setdefault(code, []).append({"code": code, **below})
if allow_adjacent and (weak_surname := _surname_above_forename(lines, owner_labels, label_indices)):
proposed.setdefault("C.1.1", []).append(weak_surname)
selected = {}
for code in FIELD_CODES:
unique = {}
for entry in proposed.get(code, []):
# Overlapping local image tiles may observe the same printed value
# more than once. Identical readings are one proposal, even when
# their honest layout regions have different sizes.
unique.setdefault(_norm(entry["value"]), entry)
if len(unique) == 1:
selected[code] = next(iter(unique.values()))
# One OCR value box cannot prove two different adjacent fields. Prefer a
# clearly nearer printed code; otherwise leave both assignments open.
by_value: dict[int, list[dict]] = {}
for entry in selected.values():
if entry["method"] == "adjacent_code":
by_value.setdefault(entry["line_indices"][-1], []).append(entry)
for collisions in by_value.values():
if len(collisions) < 2:
continue
collisions.sort(key=lambda item: item["_distance"])
keep = collisions[0] if collisions[0]["_distance"] < .7 * collisions[1]["_distance"] else None
for entry in collisions:
if entry is not keep:
selected.pop(entry["code"], None)
for entry in selected.values():
entry.pop("_distance", None)
return [selected[code] for code in FIELD_CODES if code in selected]