357 lines
17 KiB
Python
357 lines
17 KiB
Python
"""Assign observed OCR text to ZB Teil I fields with image evidence.
|
||
|
||
No value is generated. Weak code inferences from neighboring observed fields
|
||
are explicit and remain reviewable in the UI.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from difflib import SequenceMatcher
|
||
import re
|
||
|
||
|
||
FIELD_CODES = tuple("""A C.1.1 C.1.2 C.1.3 X I B 2.1 2.2 J 4 E 3 D.1 D.2 D.3 2 5
|
||
V.9 14 P.3 10 14.1 P.1 L 9 P.2/P.4 T 18 19 20 G 12 13 Q V.7 F.1 F.2
|
||
7.1 7.2 7.3 8.1 8.2 8.3 U.1 U.2 U.3 O.1 O.2 S.1 S.2 15.1 15.2 15.3
|
||
R 11 K 6 17 16 21 22""".split())
|
||
|
||
|
||
def _norm(text: str) -> str:
|
||
return re.sub(r"[^A-Z0-9]", "", text.upper())
|
||
|
||
|
||
def _code(text: str) -> str | None:
|
||
# Exact dotted labels win. Dot loss is accepted only when it is
|
||
# unambiguous; arbitrary words and strings with values are not labels.
|
||
literal = text.strip().upper().strip("() ")
|
||
if literal in ("21", "22"):
|
||
# Without the dot, 2.1/2.2 and 21/22 cannot be distinguished.
|
||
return None
|
||
if literal in FIELD_CODES:
|
||
return literal
|
||
token = _norm(text)
|
||
matches = [code for code in FIELD_CODES if _norm(code) == token]
|
||
return matches[0] if len(matches) == 1 else None
|
||
|
||
|
||
_OWNER_CAPTIONS = {
|
||
"C.1.1": re.compile(r"\b(?:name|firmenname)\b", re.I),
|
||
"C.1.2": re.compile(r"\bvorname", re.I),
|
||
"C.1.3": re.compile(r"\b(?:anschrift|adresse)\b", re.I),
|
||
}
|
||
_OWNER_PRINTED = {"C.1.1": "NAMEODERFIRMENNAME", "C.1.2": "VORNAMEN",
|
||
"C.1.3": "ANSCHRIFT"}
|
||
|
||
|
||
def _owner_caption(text: str) -> str | None:
|
||
match = re.match(r"^\s*\(?\s*C\s*\.?\s*1\s*\.?\s*([123])\s*\)?\s*(.*)$", text, re.I)
|
||
if match:
|
||
code = f"C.1.{match[1]}"
|
||
tail = match[2].strip(" \t:;=-")
|
||
if not tail or _OWNER_CAPTIONS[code].search(tail):
|
||
return code
|
||
letters = re.sub(r"[^A-Z]", "", tail.upper())
|
||
if len(letters) >= 6 and SequenceMatcher(None, letters, _OWNER_PRINTED[code]).ratio() >= .62:
|
||
return code
|
||
# Small printed codes can be lost while the full caption survives.
|
||
# Use only the distinctive short printed wording, never a person's name.
|
||
if len(text) < 55:
|
||
if re.search(r"\bName\s+oder\s+Firmenname\b", text, re.I):
|
||
return "C.1.1"
|
||
if re.search(r"\bVorname\b", text, re.I):
|
||
return "C.1.2"
|
||
if re.search(r"\bAnschrift\b", text, re.I):
|
||
return "C.1.3"
|
||
return None
|
||
|
||
|
||
def _height(box: list[float]) -> float:
|
||
return box[3] - box[1]
|
||
|
||
|
||
def _vertical_overlap(a: list[float], b: list[float]) -> float:
|
||
return max(0, min(a[3], b[3]) - max(a[1], b[1]))
|
||
|
||
|
||
def _nearby_value(anchor: dict, lines: list[dict], labels: set[int]) -> tuple[int, float] | None:
|
||
ax1, ay1, ax2, ay2 = anchor["box"]
|
||
ah = _height(anchor["box"])
|
||
candidates = []
|
||
for index, line in enumerate(lines):
|
||
if (index in labels or line is anchor or not line.get("box") or
|
||
not line["text"].strip(" \t-–—.,")):
|
||
continue
|
||
bx1, by1, bx2, by2 = line["box"]
|
||
bh = _height(line["box"])
|
||
if bx2 <= (ax1 + ax2) / 2 or bx1 < ax1 - 0.2 * ah:
|
||
continue
|
||
if bx1 - ax2 > max(120, 5 * ah):
|
||
continue
|
||
overlap = _vertical_overlap(anchor["box"], line["box"])
|
||
if overlap < 0.45 * min(ah, bh):
|
||
continue
|
||
center_gap = abs((by1 + by2) / 2 - (ay1 + ay2) / 2)
|
||
# Prefer the closest value to the right; reject rows that only touch
|
||
# because large OCR boxes extend into a neighbouring printed row.
|
||
candidates.append((max(0, bx1 - ax2) + center_gap * 1.5, index))
|
||
if not candidates:
|
||
return None
|
||
candidates.sort()
|
||
return candidates[0][1], candidates[0][0]
|
||
|
||
|
||
def _owner_below(anchor_index: int, lines: list[dict], labels: set[int]) -> dict | None:
|
||
"""Read holder value rows below an observed caption, within its column."""
|
||
anchor = lines[anchor_index]
|
||
code = _owner_caption(anchor["text"])
|
||
if "Layoutblock" in anchor.get("source", ""):
|
||
# The VL parser sometimes puts a printed caption and its value in
|
||
# consecutive text lines of one coarse region. Their box is shared;
|
||
# line order is the only honest evidence for this relation.
|
||
same_region = []
|
||
for index in range(anchor_index + 1, min(len(lines), anchor_index + (3 if code == "C.1.3" else 2))):
|
||
line = lines[index]
|
||
if index in labels or line.get("box") != anchor["box"] or not line["text"].strip():
|
||
break
|
||
same_region.append(index)
|
||
if same_region:
|
||
value = "\n".join(lines[index]["text"].strip() for index in same_region)
|
||
return {"value": value[:500], "raw_text": value, "box": anchor["box"],
|
||
"anchor_box": anchor["box"], "line_indices": [anchor_index, *same_region],
|
||
"method": "caption_region_next_line", "confidence": None}
|
||
ax1, _, _, ay2 = anchor["box"]
|
||
ah = _height(anchor["box"])
|
||
candidates = []
|
||
for index, line in enumerate(lines):
|
||
if index in labels or not line.get("box") or not line["text"].strip():
|
||
continue
|
||
x1, y1, x2, y2 = line["box"]
|
||
if not (ay2 - .15 * ah <= y1 <= ay2 + max(85, 2.8 * ah)):
|
||
continue
|
||
if not (ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)):
|
||
continue
|
||
if x2 <= ax1 or re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I):
|
||
continue
|
||
candidates.append((max(0, y1 - ay2), abs(x1 - ax1), index))
|
||
if not candidates:
|
||
return None
|
||
candidates.sort()
|
||
first = lines[candidates[0][2]]["box"]
|
||
first_indices = [index for index, line in enumerate(lines)
|
||
if (index not in labels or line["text"].strip().isdigit())
|
||
and line.get("box") and line["text"].strip()
|
||
and line["box"][2] > ax1
|
||
and _vertical_overlap(first, line["box"]) >=
|
||
.35 * min(_height(first), _height(line["box"]))]
|
||
first_indices.sort(key=lambda index: lines[index]["box"][0])
|
||
# Only join fragments that touch the same printed value row. A remote
|
||
# table column with a large gap is not part of the holder's name.
|
||
row = [min((index for index in first_indices
|
||
if ax1 - max(65, 1.5 * ah) <= lines[index]["box"][0]
|
||
<= ax1 + max(170, 3 * ah)),
|
||
key=lambda index: lines[index]["box"][0], default=candidates[0][2])]
|
||
for index in first_indices:
|
||
if index == row[0] or lines[index]["box"][0] < lines[row[0]]["box"][0]:
|
||
continue
|
||
previous = lines[row[-1]]["box"]
|
||
box = lines[index]["box"]
|
||
if box[0] - previous[2] <= max(45, .8 * _height(first)):
|
||
row.append(index)
|
||
rows = [row]
|
||
if code == "C.1.3" and "Layoutblock" not in anchor.get("source", ""):
|
||
bottom = max(lines[index]["box"][3] for index in row)
|
||
row_height = max(_height(lines[index]["box"]) for index in row)
|
||
next_row = []
|
||
for index, line in enumerate(lines):
|
||
if ((index in labels and not line["text"].strip().isdigit())
|
||
or index in row or not line.get("box") or not line["text"].strip()):
|
||
continue
|
||
x1, y1, x2, _ = line["box"]
|
||
if (bottom - .1 * row_height <= y1 <= bottom + max(25, .55 * row_height)
|
||
and x1 >= ax1 - max(65, 1.5 * ah)
|
||
and x2 > ax1
|
||
and not re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I)):
|
||
next_row.append(index)
|
||
next_row.sort(key=lambda index: lines[index]["box"][0])
|
||
if next_row:
|
||
starter = next((index for index in next_row
|
||
if lines[index]["box"][0] <= ax1 + max(170, 3 * ah)), None)
|
||
contiguous = [starter] if starter is not None else []
|
||
for index in next_row:
|
||
if not contiguous or index == starter or lines[index]["box"][0] < lines[starter]["box"][0]:
|
||
continue
|
||
if lines[index]["box"][0] - lines[contiguous[-1]]["box"][2] <= max(45, .8 * row_height):
|
||
contiguous.append(index)
|
||
if contiguous:
|
||
rows.append(contiguous)
|
||
indices = [index for row in rows for index in row]
|
||
value = "\n".join("\n".join(lines[index]["text"].strip() for index in row)
|
||
if len(row) > 1 and all(lines[index]["box"] == lines[row[0]]["box"] for index in row)
|
||
else " ".join(lines[index]["text"].strip() for index in row)
|
||
for row in rows)
|
||
boxes = [lines[index]["box"] for index in indices]
|
||
box = [min(b[0] for b in boxes), min(b[1] for b in boxes),
|
||
max(b[2] for b in boxes), max(b[3] for b in boxes)]
|
||
return {"value": value[:500], "raw_text": value, "box": box,
|
||
"anchor_box": anchor["box"], "line_indices": [anchor_index, *indices],
|
||
"method": "below_caption", "confidence": None}
|
||
|
||
|
||
def _surname_above_forename(lines: list[dict], owner_labels: dict[int, str],
|
||
labels: set[int]) -> dict | None:
|
||
"""Weak suggestion when C.1.1 is unreadable but C.1.2 is observed."""
|
||
if "C.1.1" in owner_labels.values():
|
||
return None
|
||
forename = [index for index, code in owner_labels.items() if code == "C.1.2"]
|
||
if len(forename) != 1:
|
||
return None
|
||
anchor_index = forename[0]
|
||
anchor = lines[anchor_index]
|
||
ax1, ay1, _, _ = anchor["box"]
|
||
ah = _height(anchor["box"])
|
||
candidates = []
|
||
for index, line in enumerate(lines):
|
||
if index in labels or not line.get("box"):
|
||
continue
|
||
text = line["text"].strip()
|
||
if len(re.sub(r"[^A-ZÄÖÜ]", "", text.upper())) < 3 or any(c.isdigit() for c in text):
|
||
continue
|
||
x1, _, x2, y2 = line["box"]
|
||
gap = ay1 - y2
|
||
if (0 <= gap <= max(150, 3 * ah)
|
||
and ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)
|
||
and x2 > ax1):
|
||
candidates.append((gap, abs(x1 - ax1), index))
|
||
candidates.sort()
|
||
if not candidates or (len(candidates) > 1 and candidates[1][0] - candidates[0][0] < ah):
|
||
return None
|
||
value_index = candidates[0][2]
|
||
value_line = lines[value_index]
|
||
return {"code": "C.1.1", "value": value_line["text"].strip()[:500],
|
||
"raw_text": value_line["text"], "box": value_line["box"],
|
||
"anchor_box": anchor["box"], "line_indices": [anchor_index, value_index],
|
||
"method": "surname_above_forename", "confidence": None}
|
||
|
||
|
||
def assign_fields(lines: list[dict], *, allow_adjacent: bool = True) -> list[dict]:
|
||
"""Return only observed, localized values; duplicates stay in raw lines."""
|
||
proposed: dict[str, list[dict]] = {}
|
||
labels = {index: code for index, line in enumerate(lines)
|
||
if line.get("box") and (code := _code(line["text"]))}
|
||
owner_labels = {index: code for index, line in enumerate(lines)
|
||
if line.get("box") and (code := _owner_caption(line["text"]))}
|
||
label_indices = set(labels) | set(owner_labels)
|
||
for index, line in enumerate(lines):
|
||
box = line.get("box")
|
||
if not box:
|
||
continue
|
||
if isinstance(line.get("cells"), list):
|
||
cells = line["cells"]
|
||
contextual = {}
|
||
if (len(cells) >= 6 and _code(cells[0]) == "B"
|
||
and _norm(cells[2]) == "21" and _norm(cells[4]) == "22"
|
||
and _code(cells[2]) is None and _code(cells[4]) is None):
|
||
# In one observed table row, the printed B / 2.1 / 2.2
|
||
# sequence can lose both tiny dots. Keep this weaker reading
|
||
# explicit and reviewable; never reinterpret a lone 21/22.
|
||
contextual = {2: "2.1", 4: "2.2"}
|
||
consumed_values: set[int] = set()
|
||
for cell_index, cell in enumerate(cells[:-1]):
|
||
if cell_index in consumed_values:
|
||
continue
|
||
code = (contextual.get(cell_index) or _code(cell)) if isinstance(cell, str) else None
|
||
value = cells[cell_index + 1] if isinstance(cells[cell_index + 1], str) else ""
|
||
value = value.strip()
|
||
if (code and value and value not in ("-", "–", "—")
|
||
and _code(value) is None
|
||
and not (code in _OWNER_CAPTIONS and _OWNER_CAPTIONS[code].search(value))):
|
||
consumed_values.add(cell_index + 1)
|
||
proposed.setdefault(code, []).append({
|
||
"code": code, "value": value[:500], "raw_text": line["text"],
|
||
"box": box, "anchor_box": box, "line_indices": [index],
|
||
"method": "vl_table_context_code" if cell_index in contextual else "vl_table_adjacent_cell",
|
||
"confidence": None,
|
||
})
|
||
continue
|
||
text = line["text"].strip()
|
||
# Code and value in the same OCR line.
|
||
matched = False
|
||
for code in sorted(FIELD_CODES, key=len, reverse=True):
|
||
if code.startswith("C.1."):
|
||
# These printed captions precede the holder data on another row.
|
||
continue
|
||
printed = re.escape(code)
|
||
marker = (rf"(?:\({printed}\)|{printed}(?=\s*[:;=\-]))"
|
||
if len(code) == 1 or code.isdigit() else
|
||
rf"(?:\({printed}\)|{printed}(?=\s|[:;=\-]))")
|
||
match = re.match(rf"^\s*{marker}\s*[:;=\-]?\s*", text, flags=re.I)
|
||
if match and (value := text[match.end():].strip(" \t:;=-")) and value.strip("-–—.,"):
|
||
proposed.setdefault(code, []).append({
|
||
"code": code, "value": value[:500], "raw_text": text,
|
||
"box": box, "anchor_box": box, "line_indices": [index],
|
||
"method": "same_line_code", "confidence": None,
|
||
})
|
||
matched = True
|
||
break
|
||
if not matched:
|
||
# Small printed codes can merge with a value in one OCR box.
|
||
# Keep the original box and mark this weaker reading for review.
|
||
for code in sorted((c for c in FIELD_CODES if re.fullmatch(r"[A-Z]\.\d(?:\.\d)?", c)),
|
||
key=len, reverse=True):
|
||
marker = re.escape(code).replace(r"\.", r"\.?" )
|
||
match = re.match(rf"^\s*{marker}(?=[A-Z0-9])", text, flags=re.I)
|
||
if match and (value := text[match.end():].strip(" \t:;=-")):
|
||
proposed.setdefault(code, []).append({
|
||
"code": code, "value": value[:500], "raw_text": text,
|
||
"box": box, "anchor_box": box, "line_indices": [index],
|
||
"method": "merged_code", "confidence": None,
|
||
})
|
||
break
|
||
for index, code in (labels.items() if allow_adjacent else []):
|
||
if code in _OWNER_CAPTIONS:
|
||
continue
|
||
nearby = _nearby_value(lines[index], lines, label_indices)
|
||
if nearby is None:
|
||
continue
|
||
value_index, distance = nearby
|
||
value_line = lines[value_index]
|
||
proposed.setdefault(code, []).append({
|
||
"code": code, "value": value_line["text"].strip()[:500],
|
||
"raw_text": value_line["text"], "box": value_line["box"],
|
||
"anchor_box": lines[index]["box"], "line_indices": [index, value_index],
|
||
"method": "adjacent_code", "confidence": None, "_distance": distance,
|
||
})
|
||
for index, code in owner_labels.items():
|
||
below = _owner_below(index, lines, label_indices)
|
||
if below:
|
||
proposed.setdefault(code, []).append({"code": code, **below})
|
||
if allow_adjacent and (weak_surname := _surname_above_forename(lines, owner_labels, label_indices)):
|
||
proposed.setdefault("C.1.1", []).append(weak_surname)
|
||
selected = {}
|
||
for code in FIELD_CODES:
|
||
unique = {}
|
||
for entry in proposed.get(code, []):
|
||
# Overlapping local image tiles may observe the same printed value
|
||
# more than once. Identical readings are one proposal, even when
|
||
# their honest layout regions have different sizes.
|
||
unique.setdefault(_norm(entry["value"]), entry)
|
||
if len(unique) == 1:
|
||
selected[code] = next(iter(unique.values()))
|
||
# One OCR value box cannot prove two different adjacent fields. Prefer a
|
||
# clearly nearer printed code; otherwise leave both assignments open.
|
||
by_value: dict[int, list[dict]] = {}
|
||
for entry in selected.values():
|
||
if entry["method"] == "adjacent_code":
|
||
by_value.setdefault(entry["line_indices"][-1], []).append(entry)
|
||
for collisions in by_value.values():
|
||
if len(collisions) < 2:
|
||
continue
|
||
collisions.sort(key=lambda item: item["_distance"])
|
||
keep = collisions[0] if collisions[0]["_distance"] < .7 * collisions[1]["_distance"] else None
|
||
for entry in collisions:
|
||
if entry is not keep:
|
||
selected.pop(entry["code"], None)
|
||
for entry in selected.values():
|
||
entry.pop("_distance", None)
|
||
return [selected[code] for code in FIELD_CODES if code in selected]
|