Files
2026-10-07 21:00:21 +02:00

357 lines
17 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Assign observed OCR text to ZB Teil I fields with image evidence.
No value is generated. Weak code inferences from neighboring observed fields
are explicit and remain reviewable in the UI.
"""
from __future__ import annotations
from difflib import SequenceMatcher
import re
FIELD_CODES = tuple("""A C.1.1 C.1.2 C.1.3 X I B 2.1 2.2 J 4 E 3 D.1 D.2 D.3 2 5
V.9 14 P.3 10 14.1 P.1 L 9 P.2/P.4 T 18 19 20 G 12 13 Q V.7 F.1 F.2
7.1 7.2 7.3 8.1 8.2 8.3 U.1 U.2 U.3 O.1 O.2 S.1 S.2 15.1 15.2 15.3
R 11 K 6 17 16 21 22""".split())
def _norm(text: str) -> str:
return re.sub(r"[^A-Z0-9]", "", text.upper())
def _code(text: str) -> str | None:
# Exact dotted labels win. Dot loss is accepted only when it is
# unambiguous; arbitrary words and strings with values are not labels.
literal = text.strip().upper().strip("() ")
if literal in ("21", "22"):
# Without the dot, 2.1/2.2 and 21/22 cannot be distinguished.
return None
if literal in FIELD_CODES:
return literal
token = _norm(text)
matches = [code for code in FIELD_CODES if _norm(code) == token]
return matches[0] if len(matches) == 1 else None
_OWNER_CAPTIONS = {
"C.1.1": re.compile(r"\b(?:name|firmenname)\b", re.I),
"C.1.2": re.compile(r"\bvorname", re.I),
"C.1.3": re.compile(r"\b(?:anschrift|adresse)\b", re.I),
}
_OWNER_PRINTED = {"C.1.1": "NAMEODERFIRMENNAME", "C.1.2": "VORNAMEN",
"C.1.3": "ANSCHRIFT"}
def _owner_caption(text: str) -> str | None:
match = re.match(r"^\s*\(?\s*C\s*\.?\s*1\s*\.?\s*([123])\s*\)?\s*(.*)$", text, re.I)
if match:
code = f"C.1.{match[1]}"
tail = match[2].strip(" \t:;=-")
if not tail or _OWNER_CAPTIONS[code].search(tail):
return code
letters = re.sub(r"[^A-Z]", "", tail.upper())
if len(letters) >= 6 and SequenceMatcher(None, letters, _OWNER_PRINTED[code]).ratio() >= .62:
return code
# Small printed codes can be lost while the full caption survives.
# Use only the distinctive short printed wording, never a person's name.
if len(text) < 55:
if re.search(r"\bName\s+oder\s+Firmenname\b", text, re.I):
return "C.1.1"
if re.search(r"\bVorname\b", text, re.I):
return "C.1.2"
if re.search(r"\bAnschrift\b", text, re.I):
return "C.1.3"
return None
def _height(box: list[float]) -> float:
return box[3] - box[1]
def _vertical_overlap(a: list[float], b: list[float]) -> float:
return max(0, min(a[3], b[3]) - max(a[1], b[1]))
def _nearby_value(anchor: dict, lines: list[dict], labels: set[int]) -> tuple[int, float] | None:
ax1, ay1, ax2, ay2 = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if (index in labels or line is anchor or not line.get("box") or
not line["text"].strip(" \t-–—.,")):
continue
bx1, by1, bx2, by2 = line["box"]
bh = _height(line["box"])
if bx2 <= (ax1 + ax2) / 2 or bx1 < ax1 - 0.2 * ah:
continue
if bx1 - ax2 > max(120, 5 * ah):
continue
overlap = _vertical_overlap(anchor["box"], line["box"])
if overlap < 0.45 * min(ah, bh):
continue
center_gap = abs((by1 + by2) / 2 - (ay1 + ay2) / 2)
# Prefer the closest value to the right; reject rows that only touch
# because large OCR boxes extend into a neighbouring printed row.
candidates.append((max(0, bx1 - ax2) + center_gap * 1.5, index))
if not candidates:
return None
candidates.sort()
return candidates[0][1], candidates[0][0]
def _owner_below(anchor_index: int, lines: list[dict], labels: set[int]) -> dict | None:
"""Read holder value rows below an observed caption, within its column."""
anchor = lines[anchor_index]
code = _owner_caption(anchor["text"])
if "Layoutblock" in anchor.get("source", ""):
# The VL parser sometimes puts a printed caption and its value in
# consecutive text lines of one coarse region. Their box is shared;
# line order is the only honest evidence for this relation.
same_region = []
for index in range(anchor_index + 1, min(len(lines), anchor_index + (3 if code == "C.1.3" else 2))):
line = lines[index]
if index in labels or line.get("box") != anchor["box"] or not line["text"].strip():
break
same_region.append(index)
if same_region:
value = "\n".join(lines[index]["text"].strip() for index in same_region)
return {"value": value[:500], "raw_text": value, "box": anchor["box"],
"anchor_box": anchor["box"], "line_indices": [anchor_index, *same_region],
"method": "caption_region_next_line", "confidence": None}
ax1, _, _, ay2 = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if index in labels or not line.get("box") or not line["text"].strip():
continue
x1, y1, x2, y2 = line["box"]
if not (ay2 - .15 * ah <= y1 <= ay2 + max(85, 2.8 * ah)):
continue
if not (ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)):
continue
if x2 <= ax1 or re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I):
continue
candidates.append((max(0, y1 - ay2), abs(x1 - ax1), index))
if not candidates:
return None
candidates.sort()
first = lines[candidates[0][2]]["box"]
first_indices = [index for index, line in enumerate(lines)
if (index not in labels or line["text"].strip().isdigit())
and line.get("box") and line["text"].strip()
and line["box"][2] > ax1
and _vertical_overlap(first, line["box"]) >=
.35 * min(_height(first), _height(line["box"]))]
first_indices.sort(key=lambda index: lines[index]["box"][0])
# Only join fragments that touch the same printed value row. A remote
# table column with a large gap is not part of the holder's name.
row = [min((index for index in first_indices
if ax1 - max(65, 1.5 * ah) <= lines[index]["box"][0]
<= ax1 + max(170, 3 * ah)),
key=lambda index: lines[index]["box"][0], default=candidates[0][2])]
for index in first_indices:
if index == row[0] or lines[index]["box"][0] < lines[row[0]]["box"][0]:
continue
previous = lines[row[-1]]["box"]
box = lines[index]["box"]
if box[0] - previous[2] <= max(45, .8 * _height(first)):
row.append(index)
rows = [row]
if code == "C.1.3" and "Layoutblock" not in anchor.get("source", ""):
bottom = max(lines[index]["box"][3] for index in row)
row_height = max(_height(lines[index]["box"]) for index in row)
next_row = []
for index, line in enumerate(lines):
if ((index in labels and not line["text"].strip().isdigit())
or index in row or not line.get("box") or not line["text"].strip()):
continue
x1, y1, x2, _ = line["box"]
if (bottom - .1 * row_height <= y1 <= bottom + max(25, .55 * row_height)
and x1 >= ax1 - max(65, 1.5 * ah)
and x2 > ax1
and not re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I)):
next_row.append(index)
next_row.sort(key=lambda index: lines[index]["box"][0])
if next_row:
starter = next((index for index in next_row
if lines[index]["box"][0] <= ax1 + max(170, 3 * ah)), None)
contiguous = [starter] if starter is not None else []
for index in next_row:
if not contiguous or index == starter or lines[index]["box"][0] < lines[starter]["box"][0]:
continue
if lines[index]["box"][0] - lines[contiguous[-1]]["box"][2] <= max(45, .8 * row_height):
contiguous.append(index)
if contiguous:
rows.append(contiguous)
indices = [index for row in rows for index in row]
value = "\n".join("\n".join(lines[index]["text"].strip() for index in row)
if len(row) > 1 and all(lines[index]["box"] == lines[row[0]]["box"] for index in row)
else " ".join(lines[index]["text"].strip() for index in row)
for row in rows)
boxes = [lines[index]["box"] for index in indices]
box = [min(b[0] for b in boxes), min(b[1] for b in boxes),
max(b[2] for b in boxes), max(b[3] for b in boxes)]
return {"value": value[:500], "raw_text": value, "box": box,
"anchor_box": anchor["box"], "line_indices": [anchor_index, *indices],
"method": "below_caption", "confidence": None}
def _surname_above_forename(lines: list[dict], owner_labels: dict[int, str],
labels: set[int]) -> dict | None:
"""Weak suggestion when C.1.1 is unreadable but C.1.2 is observed."""
if "C.1.1" in owner_labels.values():
return None
forename = [index for index, code in owner_labels.items() if code == "C.1.2"]
if len(forename) != 1:
return None
anchor_index = forename[0]
anchor = lines[anchor_index]
ax1, ay1, _, _ = anchor["box"]
ah = _height(anchor["box"])
candidates = []
for index, line in enumerate(lines):
if index in labels or not line.get("box"):
continue
text = line["text"].strip()
if len(re.sub(r"[^A-ZÄÖÜ]", "", text.upper())) < 3 or any(c.isdigit() for c in text):
continue
x1, _, x2, y2 = line["box"]
gap = ay1 - y2
if (0 <= gap <= max(150, 3 * ah)
and ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)
and x2 > ax1):
candidates.append((gap, abs(x1 - ax1), index))
candidates.sort()
if not candidates or (len(candidates) > 1 and candidates[1][0] - candidates[0][0] < ah):
return None
value_index = candidates[0][2]
value_line = lines[value_index]
return {"code": "C.1.1", "value": value_line["text"].strip()[:500],
"raw_text": value_line["text"], "box": value_line["box"],
"anchor_box": anchor["box"], "line_indices": [anchor_index, value_index],
"method": "surname_above_forename", "confidence": None}
def assign_fields(lines: list[dict], *, allow_adjacent: bool = True) -> list[dict]:
"""Return only observed, localized values; duplicates stay in raw lines."""
proposed: dict[str, list[dict]] = {}
labels = {index: code for index, line in enumerate(lines)
if line.get("box") and (code := _code(line["text"]))}
owner_labels = {index: code for index, line in enumerate(lines)
if line.get("box") and (code := _owner_caption(line["text"]))}
label_indices = set(labels) | set(owner_labels)
for index, line in enumerate(lines):
box = line.get("box")
if not box:
continue
if isinstance(line.get("cells"), list):
cells = line["cells"]
contextual = {}
if (len(cells) >= 6 and _code(cells[0]) == "B"
and _norm(cells[2]) == "21" and _norm(cells[4]) == "22"
and _code(cells[2]) is None and _code(cells[4]) is None):
# In one observed table row, the printed B / 2.1 / 2.2
# sequence can lose both tiny dots. Keep this weaker reading
# explicit and reviewable; never reinterpret a lone 21/22.
contextual = {2: "2.1", 4: "2.2"}
consumed_values: set[int] = set()
for cell_index, cell in enumerate(cells[:-1]):
if cell_index in consumed_values:
continue
code = (contextual.get(cell_index) or _code(cell)) if isinstance(cell, str) else None
value = cells[cell_index + 1] if isinstance(cells[cell_index + 1], str) else ""
value = value.strip()
if (code and value and value not in ("-", "–", "—")
and _code(value) is None
and not (code in _OWNER_CAPTIONS and _OWNER_CAPTIONS[code].search(value))):
consumed_values.add(cell_index + 1)
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": line["text"],
"box": box, "anchor_box": box, "line_indices": [index],
"method": "vl_table_context_code" if cell_index in contextual else "vl_table_adjacent_cell",
"confidence": None,
})
continue
text = line["text"].strip()
# Code and value in the same OCR line.
matched = False
for code in sorted(FIELD_CODES, key=len, reverse=True):
if code.startswith("C.1."):
# These printed captions precede the holder data on another row.
continue
printed = re.escape(code)
marker = (rf"(?:\({printed}\)|{printed}(?=\s*[:;=\-]))"
if len(code) == 1 or code.isdigit() else
rf"(?:\({printed}\)|{printed}(?=\s|[:;=\-]))")
match = re.match(rf"^\s*{marker}\s*[:;=\-]?\s*", text, flags=re.I)
if match and (value := text[match.end():].strip(" \t:;=-")) and value.strip("-–—.,"):
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": text,
"box": box, "anchor_box": box, "line_indices": [index],
"method": "same_line_code", "confidence": None,
})
matched = True
break
if not matched:
# Small printed codes can merge with a value in one OCR box.
# Keep the original box and mark this weaker reading for review.
for code in sorted((c for c in FIELD_CODES if re.fullmatch(r"[A-Z]\.\d(?:\.\d)?", c)),
key=len, reverse=True):
marker = re.escape(code).replace(r"\.", r"\.?" )
match = re.match(rf"^\s*{marker}(?=[A-Z0-9])", text, flags=re.I)
if match and (value := text[match.end():].strip(" \t:;=-")):
proposed.setdefault(code, []).append({
"code": code, "value": value[:500], "raw_text": text,
"box": box, "anchor_box": box, "line_indices": [index],
"method": "merged_code", "confidence": None,
})
break
for index, code in (labels.items() if allow_adjacent else []):
if code in _OWNER_CAPTIONS:
continue
nearby = _nearby_value(lines[index], lines, label_indices)
if nearby is None:
continue
value_index, distance = nearby
value_line = lines[value_index]
proposed.setdefault(code, []).append({
"code": code, "value": value_line["text"].strip()[:500],
"raw_text": value_line["text"], "box": value_line["box"],
"anchor_box": lines[index]["box"], "line_indices": [index, value_index],
"method": "adjacent_code", "confidence": None, "_distance": distance,
})
for index, code in owner_labels.items():
below = _owner_below(index, lines, label_indices)
if below:
proposed.setdefault(code, []).append({"code": code, **below})
if allow_adjacent and (weak_surname := _surname_above_forename(lines, owner_labels, label_indices)):
proposed.setdefault("C.1.1", []).append(weak_surname)
selected = {}
for code in FIELD_CODES:
unique = {}
for entry in proposed.get(code, []):
# Overlapping local image tiles may observe the same printed value
# more than once. Identical readings are one proposal, even when
# their honest layout regions have different sizes.
unique.setdefault(_norm(entry["value"]), entry)
if len(unique) == 1:
selected[code] = next(iter(unique.values()))
# One OCR value box cannot prove two different adjacent fields. Prefer a
# clearly nearer printed code; otherwise leave both assignments open.
by_value: dict[int, list[dict]] = {}
for entry in selected.values():
if entry["method"] == "adjacent_code":
by_value.setdefault(entry["line_indices"][-1], []).append(entry)
for collisions in by_value.values():
if len(collisions) < 2:
continue
collisions.sort(key=lambda item: item["_distance"])
keep = collisions[0] if collisions[0]["_distance"] < .7 * collisions[1]["_distance"] else None
for entry in collisions:
if entry is not keep:
selected.pop(entry["code"], None)
for entry in selected.values():
entry.pop("_distance", None)
return [selected[code] for code in FIELD_CODES if code in selected]