"""Assign observed OCR text to ZB Teil I fields with image evidence. No value is generated. Weak code inferences from neighboring observed fields are explicit and remain reviewable in the UI. """ from __future__ import annotations from difflib import SequenceMatcher import re FIELD_CODES = tuple("""A C.1.1 C.1.2 C.1.3 X I B 2.1 2.2 J 4 E 3 D.1 D.2 D.3 2 5 V.9 14 P.3 10 14.1 P.1 L 9 P.2/P.4 T 18 19 20 G 12 13 Q V.7 F.1 F.2 7.1 7.2 7.3 8.1 8.2 8.3 U.1 U.2 U.3 O.1 O.2 S.1 S.2 15.1 15.2 15.3 R 11 K 6 17 16 21 22""".split()) def _norm(text: str) -> str: return re.sub(r"[^A-Z0-9]", "", text.upper()) def _code(text: str) -> str | None: # Exact dotted labels win. Dot loss is accepted only when it is # unambiguous; arbitrary words and strings with values are not labels. literal = text.strip().upper().strip("() ") if literal in ("21", "22"): # Without the dot, 2.1/2.2 and 21/22 cannot be distinguished. return None if literal in FIELD_CODES: return literal token = _norm(text) matches = [code for code in FIELD_CODES if _norm(code) == token] return matches[0] if len(matches) == 1 else None _OWNER_CAPTIONS = { "C.1.1": re.compile(r"\b(?:name|firmenname)\b", re.I), "C.1.2": re.compile(r"\bvorname", re.I), "C.1.3": re.compile(r"\b(?:anschrift|adresse)\b", re.I), } _OWNER_PRINTED = {"C.1.1": "NAMEODERFIRMENNAME", "C.1.2": "VORNAMEN", "C.1.3": "ANSCHRIFT"} def _owner_caption(text: str) -> str | None: match = re.match(r"^\s*\(?\s*C\s*\.?\s*1\s*\.?\s*([123])\s*\)?\s*(.*)$", text, re.I) if match: code = f"C.1.{match[1]}" tail = match[2].strip(" \t:;=-") if not tail or _OWNER_CAPTIONS[code].search(tail): return code letters = re.sub(r"[^A-Z]", "", tail.upper()) if len(letters) >= 6 and SequenceMatcher(None, letters, _OWNER_PRINTED[code]).ratio() >= .62: return code # Small printed codes can be lost while the full caption survives. # Use only the distinctive short printed wording, never a person's name. if len(text) < 55: if re.search(r"\bName\s+oder\s+Firmenname\b", text, re.I): return "C.1.1" if re.search(r"\bVorname\b", text, re.I): return "C.1.2" if re.search(r"\bAnschrift\b", text, re.I): return "C.1.3" return None def _height(box: list[float]) -> float: return box[3] - box[1] def _vertical_overlap(a: list[float], b: list[float]) -> float: return max(0, min(a[3], b[3]) - max(a[1], b[1])) def _nearby_value(anchor: dict, lines: list[dict], labels: set[int]) -> tuple[int, float] | None: ax1, ay1, ax2, ay2 = anchor["box"] ah = _height(anchor["box"]) candidates = [] for index, line in enumerate(lines): if (index in labels or line is anchor or not line.get("box") or not line["text"].strip(" \t-–—.,")): continue bx1, by1, bx2, by2 = line["box"] bh = _height(line["box"]) if bx2 <= (ax1 + ax2) / 2 or bx1 < ax1 - 0.2 * ah: continue if bx1 - ax2 > max(120, 5 * ah): continue overlap = _vertical_overlap(anchor["box"], line["box"]) if overlap < 0.45 * min(ah, bh): continue center_gap = abs((by1 + by2) / 2 - (ay1 + ay2) / 2) # Prefer the closest value to the right; reject rows that only touch # because large OCR boxes extend into a neighbouring printed row. candidates.append((max(0, bx1 - ax2) + center_gap * 1.5, index)) if not candidates: return None candidates.sort() return candidates[0][1], candidates[0][0] def _owner_below(anchor_index: int, lines: list[dict], labels: set[int]) -> dict | None: """Read holder value rows below an observed caption, within its column.""" anchor = lines[anchor_index] code = _owner_caption(anchor["text"]) if "Layoutblock" in anchor.get("source", ""): # The VL parser sometimes puts a printed caption and its value in # consecutive text lines of one coarse region. Their box is shared; # line order is the only honest evidence for this relation. same_region = [] for index in range(anchor_index + 1, min(len(lines), anchor_index + (3 if code == "C.1.3" else 2))): line = lines[index] if index in labels or line.get("box") != anchor["box"] or not line["text"].strip(): break same_region.append(index) if same_region: value = "\n".join(lines[index]["text"].strip() for index in same_region) return {"value": value[:500], "raw_text": value, "box": anchor["box"], "anchor_box": anchor["box"], "line_indices": [anchor_index, *same_region], "method": "caption_region_next_line", "confidence": None} ax1, _, _, ay2 = anchor["box"] ah = _height(anchor["box"]) candidates = [] for index, line in enumerate(lines): if index in labels or not line.get("box") or not line["text"].strip(): continue x1, y1, x2, y2 = line["box"] if not (ay2 - .15 * ah <= y1 <= ay2 + max(85, 2.8 * ah)): continue if not (ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah)): continue if x2 <= ax1 or re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I): continue candidates.append((max(0, y1 - ay2), abs(x1 - ax1), index)) if not candidates: return None candidates.sort() first = lines[candidates[0][2]]["box"] first_indices = [index for index, line in enumerate(lines) if (index not in labels or line["text"].strip().isdigit()) and line.get("box") and line["text"].strip() and line["box"][2] > ax1 and _vertical_overlap(first, line["box"]) >= .35 * min(_height(first), _height(line["box"]))] first_indices.sort(key=lambda index: lines[index]["box"][0]) # Only join fragments that touch the same printed value row. A remote # table column with a large gap is not part of the holder's name. row = [min((index for index in first_indices if ax1 - max(65, 1.5 * ah) <= lines[index]["box"][0] <= ax1 + max(170, 3 * ah)), key=lambda index: lines[index]["box"][0], default=candidates[0][2])] for index in first_indices: if index == row[0] or lines[index]["box"][0] < lines[row[0]]["box"][0]: continue previous = lines[row[-1]]["box"] box = lines[index]["box"] if box[0] - previous[2] <= max(45, .8 * _height(first)): row.append(index) rows = [row] if code == "C.1.3" and "Layoutblock" not in anchor.get("source", ""): bottom = max(lines[index]["box"][3] for index in row) row_height = max(_height(lines[index]["box"]) for index in row) next_row = [] for index, line in enumerate(lines): if ((index in labels and not line["text"].strip().isdigit()) or index in row or not line.get("box") or not line["text"].strip()): continue x1, y1, x2, _ = line["box"] if (bottom - .1 * row_height <= y1 <= bottom + max(25, .55 * row_height) and x1 >= ax1 - max(65, 1.5 * ah) and x2 > ax1 and not re.match(r"^\s*X\s*(?:Nächste|Naechste|HU)\b", line["text"], re.I)): next_row.append(index) next_row.sort(key=lambda index: lines[index]["box"][0]) if next_row: starter = next((index for index in next_row if lines[index]["box"][0] <= ax1 + max(170, 3 * ah)), None) contiguous = [starter] if starter is not None else [] for index in next_row: if not contiguous or index == starter or lines[index]["box"][0] < lines[starter]["box"][0]: continue if lines[index]["box"][0] - lines[contiguous[-1]]["box"][2] <= max(45, .8 * row_height): contiguous.append(index) if contiguous: rows.append(contiguous) indices = [index for row in rows for index in row] value = "\n".join("\n".join(lines[index]["text"].strip() for index in row) if len(row) > 1 and all(lines[index]["box"] == lines[row[0]]["box"] for index in row) else " ".join(lines[index]["text"].strip() for index in row) for row in rows) boxes = [lines[index]["box"] for index in indices] box = [min(b[0] for b in boxes), min(b[1] for b in boxes), max(b[2] for b in boxes), max(b[3] for b in boxes)] return {"value": value[:500], "raw_text": value, "box": box, "anchor_box": anchor["box"], "line_indices": [anchor_index, *indices], "method": "below_caption", "confidence": None} def _surname_above_forename(lines: list[dict], owner_labels: dict[int, str], labels: set[int]) -> dict | None: """Weak suggestion when C.1.1 is unreadable but C.1.2 is observed.""" if "C.1.1" in owner_labels.values(): return None forename = [index for index, code in owner_labels.items() if code == "C.1.2"] if len(forename) != 1: return None anchor_index = forename[0] anchor = lines[anchor_index] ax1, ay1, _, _ = anchor["box"] ah = _height(anchor["box"]) candidates = [] for index, line in enumerate(lines): if index in labels or not line.get("box"): continue text = line["text"].strip() if len(re.sub(r"[^A-ZÄÖÜ]", "", text.upper())) < 3 or any(c.isdigit() for c in text): continue x1, _, x2, y2 = line["box"] gap = ay1 - y2 if (0 <= gap <= max(150, 3 * ah) and ax1 - max(65, 1.5 * ah) <= x1 <= ax1 + max(170, 3 * ah) and x2 > ax1): candidates.append((gap, abs(x1 - ax1), index)) candidates.sort() if not candidates or (len(candidates) > 1 and candidates[1][0] - candidates[0][0] < ah): return None value_index = candidates[0][2] value_line = lines[value_index] return {"code": "C.1.1", "value": value_line["text"].strip()[:500], "raw_text": value_line["text"], "box": value_line["box"], "anchor_box": anchor["box"], "line_indices": [anchor_index, value_index], "method": "surname_above_forename", "confidence": None} def assign_fields(lines: list[dict], *, allow_adjacent: bool = True) -> list[dict]: """Return only observed, localized values; duplicates stay in raw lines.""" proposed: dict[str, list[dict]] = {} labels = {index: code for index, line in enumerate(lines) if line.get("box") and (code := _code(line["text"]))} owner_labels = {index: code for index, line in enumerate(lines) if line.get("box") and (code := _owner_caption(line["text"]))} label_indices = set(labels) | set(owner_labels) for index, line in enumerate(lines): box = line.get("box") if not box: continue if isinstance(line.get("cells"), list): cells = line["cells"] contextual = {} if (len(cells) >= 6 and _code(cells[0]) == "B" and _norm(cells[2]) == "21" and _norm(cells[4]) == "22" and _code(cells[2]) is None and _code(cells[4]) is None): # In one observed table row, the printed B / 2.1 / 2.2 # sequence can lose both tiny dots. Keep this weaker reading # explicit and reviewable; never reinterpret a lone 21/22. contextual = {2: "2.1", 4: "2.2"} consumed_values: set[int] = set() for cell_index, cell in enumerate(cells[:-1]): if cell_index in consumed_values: continue code = (contextual.get(cell_index) or _code(cell)) if isinstance(cell, str) else None value = cells[cell_index + 1] if isinstance(cells[cell_index + 1], str) else "" value = value.strip() if (code and value and value not in ("-", "–", "—") and _code(value) is None and not (code in _OWNER_CAPTIONS and _OWNER_CAPTIONS[code].search(value))): consumed_values.add(cell_index + 1) proposed.setdefault(code, []).append({ "code": code, "value": value[:500], "raw_text": line["text"], "box": box, "anchor_box": box, "line_indices": [index], "method": "vl_table_context_code" if cell_index in contextual else "vl_table_adjacent_cell", "confidence": None, }) continue text = line["text"].strip() # Code and value in the same OCR line. matched = False for code in sorted(FIELD_CODES, key=len, reverse=True): if code.startswith("C.1."): # These printed captions precede the holder data on another row. continue printed = re.escape(code) marker = (rf"(?:\({printed}\)|{printed}(?=\s*[:;=\-]))" if len(code) == 1 or code.isdigit() else rf"(?:\({printed}\)|{printed}(?=\s|[:;=\-]))") match = re.match(rf"^\s*{marker}\s*[:;=\-]?\s*", text, flags=re.I) if match and (value := text[match.end():].strip(" \t:;=-")) and value.strip("-–—.,"): proposed.setdefault(code, []).append({ "code": code, "value": value[:500], "raw_text": text, "box": box, "anchor_box": box, "line_indices": [index], "method": "same_line_code", "confidence": None, }) matched = True break if not matched: # Small printed codes can merge with a value in one OCR box. # Keep the original box and mark this weaker reading for review. for code in sorted((c for c in FIELD_CODES if re.fullmatch(r"[A-Z]\.\d(?:\.\d)?", c)), key=len, reverse=True): marker = re.escape(code).replace(r"\.", r"\.?" ) match = re.match(rf"^\s*{marker}(?=[A-Z0-9])", text, flags=re.I) if match and (value := text[match.end():].strip(" \t:;=-")): proposed.setdefault(code, []).append({ "code": code, "value": value[:500], "raw_text": text, "box": box, "anchor_box": box, "line_indices": [index], "method": "merged_code", "confidence": None, }) break for index, code in (labels.items() if allow_adjacent else []): if code in _OWNER_CAPTIONS: continue nearby = _nearby_value(lines[index], lines, label_indices) if nearby is None: continue value_index, distance = nearby value_line = lines[value_index] proposed.setdefault(code, []).append({ "code": code, "value": value_line["text"].strip()[:500], "raw_text": value_line["text"], "box": value_line["box"], "anchor_box": lines[index]["box"], "line_indices": [index, value_index], "method": "adjacent_code", "confidence": None, "_distance": distance, }) for index, code in owner_labels.items(): below = _owner_below(index, lines, label_indices) if below: proposed.setdefault(code, []).append({"code": code, **below}) if allow_adjacent and (weak_surname := _surname_above_forename(lines, owner_labels, label_indices)): proposed.setdefault("C.1.1", []).append(weak_surname) selected = {} for code in FIELD_CODES: unique = {} for entry in proposed.get(code, []): # Overlapping local image tiles may observe the same printed value # more than once. Identical readings are one proposal, even when # their honest layout regions have different sizes. unique.setdefault(_norm(entry["value"]), entry) if len(unique) == 1: selected[code] = next(iter(unique.values())) # One OCR value box cannot prove two different adjacent fields. Prefer a # clearly nearer printed code; otherwise leave both assignments open. by_value: dict[int, list[dict]] = {} for entry in selected.values(): if entry["method"] == "adjacent_code": by_value.setdefault(entry["line_indices"][-1], []).append(entry) for collisions in by_value.values(): if len(collisions) < 2: continue collisions.sort(key=lambda item: item["_distance"]) keep = collisions[0] if collisions[0]["_distance"] < .7 * collisions[1]["_distance"] else None for entry in collisions: if entry is not keep: selected.pop(entry["code"], None) for entry in selected.values(): entry.pop("_distance", None) return [selected[code] for code in FIELD_CODES if code in selected]