herbert-polish-legal-ner / examples /known_limitations.py
i008's picture
HerBERT Polish legal NER (general): weights + ONNX + card + examples
399294e verified
Raw History Blame Contribute Delete
3.24 kB
#!/usr/bin/env python3
"""Reproduce the model's KNOWN FAILURE CASES (see KNOWN_LIMITATIONS.md).
All inputs are SYNTHETIC — fictional names, no real personal data. The point is
to show, honestly and reproducibly, where the model still misses or mislabels
PERSON names so you can decide where extra safeguards (post-pass, review) are
needed. pip install onnxruntime transformers numpy
Run from the repo root: python examples/known_limitations.py
"""
import json
from pathlib import Path
import numpy as np
import onnxruntime as ort
from transformers import AutoTokenizer
ROOT = Path(__file__).resolve().parent.parent
PER_THRESHOLD = 0.2
tok = AutoTokenizer.from_pretrained(str(ROOT))
cfg = json.load(open(ROOT / "config.json", encoding="utf-8"))
id2label = {int(k): v for k, v in cfg["id2label"].items()}
label2id = cfg["label2id"]
sess = ort.InferenceSession(str(ROOT / "onnx" / "model_quantized.onnx"))
in_names = {i.name for i in sess.get_inputs()}
def softmax(x):
e = np.exp(x - x.max(-1, keepdims=True)); return e / e.sum(-1, keepdims=True)
def predict_per(text):
enc = tok(text, return_offsets_mapping=True, return_tensors="np", truncation=True, max_length=512)
offs = enc["offset_mapping"][0]
feeds = {"input_ids": enc["input_ids"].astype(np.int64),
"attention_mask": enc["attention_mask"].astype(np.int64)}
if "token_type_ids" in in_names:
feeds["token_type_ids"] = np.zeros_like(enc["input_ids"], dtype=np.int64)
probs = softmax(sess.run(None, feeds)[0][0])
ids = probs.argmax(-1)
pb, pi = label2id["B-PER"], label2id["I-PER"]
spans, cur = [], None
for i, (s, e) in enumerate(offs):
if s == e:
if cur: spans.append(cur); cur = None
continue
lab = id2label[int(ids[i])]
if lab == "O" and probs[i, pb] + probs[i, pi] >= PER_THRESHOLD:
lab = "I-PER" if cur else "B-PER"
if not lab.endswith("PER"):
if cur: spans.append(cur); cur = None
continue
if lab == "B-PER" or cur is None:
if cur: spans.append(cur)
cur = [int(s), int(e)]
else:
cur[1] = int(e)
if cur: spans.append(cur)
return [text[a:b] for a, b in spans]
# (input, fictional clean name behind the garble, failure mode)
CASES = [
# 1) Heavy OCR garble -> missed entirely
("Pozwana ote tobodaaka wniosła sprzeciw.", "Bożena Łobodzińska", "heavy OCR garble -> MISSED"),
("Powód go Hea stawił się osobiście.", "Igor Heliasz", "heavy OCR garble -> MISSED"),
("Decyzję doręczono: ist yi daziak.", "Justyna Idziak", "heavy OCR garble -> MISSED"),
# 2) Partial detection -> residue leak
("Wniosek złożył wiar Żoją w dniu 5 maja.", "Wiktor Żołądek", "partial -> only part caught (residue)"),
("WIESEAW KUE stawił się na rozprawie.", "Wiesław Kuc", "ALL-CAPS garble -> fragmented (gap leaks)"),
]
if __name__ == "__main__":
print(f"PER threshold = {PER_THRESHOLD}\n")
for text, clean, mode in CASES:
per = predict_per(text)
print(f"# {mode} (fictional: {clean})")
print(f" input : {text}")
print(f" model : PER = {per if per else '— (nothing)'}")
print()