#!/usr/bin/env python3 """Reproduce the model's KNOWN FAILURE CASES (see KNOWN_LIMITATIONS.md). All inputs are SYNTHETIC — fictional names, no real personal data. The point is to show, honestly and reproducibly, where the model still misses or mislabels PERSON names so you can decide where extra safeguards (post-pass, review) are needed. pip install onnxruntime transformers numpy Run from the repo root: python examples/known_limitations.py """ import json from pathlib import Path import numpy as np import onnxruntime as ort from transformers import AutoTokenizer ROOT = Path(__file__).resolve().parent.parent PER_THRESHOLD = 0.2 tok = AutoTokenizer.from_pretrained(str(ROOT)) cfg = json.load(open(ROOT / "config.json", encoding="utf-8")) id2label = {int(k): v for k, v in cfg["id2label"].items()} label2id = cfg["label2id"] sess = ort.InferenceSession(str(ROOT / "onnx" / "model_quantized.onnx")) in_names = {i.name for i in sess.get_inputs()} def softmax(x): e = np.exp(x - x.max(-1, keepdims=True)); return e / e.sum(-1, keepdims=True) def predict_per(text): enc = tok(text, return_offsets_mapping=True, return_tensors="np", truncation=True, max_length=512) offs = enc["offset_mapping"][0] feeds = {"input_ids": enc["input_ids"].astype(np.int64), "attention_mask": enc["attention_mask"].astype(np.int64)} if "token_type_ids" in in_names: feeds["token_type_ids"] = np.zeros_like(enc["input_ids"], dtype=np.int64) probs = softmax(sess.run(None, feeds)[0][0]) ids = probs.argmax(-1) pb, pi = label2id["B-PER"], label2id["I-PER"] spans, cur = [], None for i, (s, e) in enumerate(offs): if s == e: if cur: spans.append(cur); cur = None continue lab = id2label[int(ids[i])] if lab == "O" and probs[i, pb] + probs[i, pi] >= PER_THRESHOLD: lab = "I-PER" if cur else "B-PER" if not lab.endswith("PER"): if cur: spans.append(cur); cur = None continue if lab == "B-PER" or cur is None: if cur: spans.append(cur) cur = [int(s), int(e)] else: cur[1] = int(e) if cur: spans.append(cur) return [text[a:b] for a, b in spans] # (input, fictional clean name behind the garble, failure mode) CASES = [ # 1) Heavy OCR garble -> missed entirely ("Pozwana ote tobodaaka wniosła sprzeciw.", "Bożena Łobodzińska", "heavy OCR garble -> MISSED"), ("Powód go Hea stawił się osobiście.", "Igor Heliasz", "heavy OCR garble -> MISSED"), ("Decyzję doręczono: ist yi daziak.", "Justyna Idziak", "heavy OCR garble -> MISSED"), # 2) Partial detection -> residue leak ("Wniosek złożył wiar Żoją w dniu 5 maja.", "Wiktor Żołądek", "partial -> only part caught (residue)"), ("WIESEAW KUE stawił się na rozprawie.", "Wiesław Kuc", "ALL-CAPS garble -> fragmented (gap leaks)"), ] if __name__ == "__main__": print(f"PER threshold = {PER_THRESHOLD}\n") for text, clean, mode in CASES: per = predict_per(text) print(f"# {mode} (fictional: {clean})") print(f" input : {text}") print(f" model : PER = {per if per else '— (nothing)'}") print()