Token Classification
Transformers
ONNX
Safetensors
Polish
bert
named-entity-recognition
ner
pii
pii-detection
anonymization
privacy
gdpr
polish
legal
legal-nlp
herbert
Eval Results (legacy)
Instructions to use lexedit/herbert-polish-legal-ner with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use lexedit/herbert-polish-legal-ner with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("token-classification", model="lexedit/herbert-polish-legal-ner")# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForTokenClassification tokenizer = AutoTokenizer.from_pretrained("lexedit/herbert-polish-legal-ner") model = AutoModelForTokenClassification.from_pretrained("lexedit/herbert-polish-legal-ner", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Download examples/known_limitations.py from lexedit/herbert-polish-legal-ner: direct link, hf CLI and curl.
- Browser
- Download file 3.24 kB
-
https://huggingface.co/lexedit/herbert-polish-legal-ner/resolve/main/examples/known_limitations.py
- Command line
-
hf download hf://lexedit/herbert-polish-legal-ner/examples/known_limitations.py
-
curl -L -o known_limitations.py https://huggingface.co/lexedit/herbert-polish-legal-ner/resolve/main/examples/known_limitations.py
3.24 kB
| #!/usr/bin/env python3 | |
| """Reproduce the model's KNOWN FAILURE CASES (see KNOWN_LIMITATIONS.md). | |
| All inputs are SYNTHETIC — fictional names, no real personal data. The point is | |
| to show, honestly and reproducibly, where the model still misses or mislabels | |
| PERSON names so you can decide where extra safeguards (post-pass, review) are | |
| needed. pip install onnxruntime transformers numpy | |
| Run from the repo root: python examples/known_limitations.py | |
| """ | |
| import json | |
| from pathlib import Path | |
| import numpy as np | |
| import onnxruntime as ort | |
| from transformers import AutoTokenizer | |
| ROOT = Path(__file__).resolve().parent.parent | |
| PER_THRESHOLD = 0.2 | |
| tok = AutoTokenizer.from_pretrained(str(ROOT)) | |
| cfg = json.load(open(ROOT / "config.json", encoding="utf-8")) | |
| id2label = {int(k): v for k, v in cfg["id2label"].items()} | |
| label2id = cfg["label2id"] | |
| sess = ort.InferenceSession(str(ROOT / "onnx" / "model_quantized.onnx")) | |
| in_names = {i.name for i in sess.get_inputs()} | |
| def softmax(x): | |
| e = np.exp(x - x.max(-1, keepdims=True)); return e / e.sum(-1, keepdims=True) | |
| def predict_per(text): | |
| enc = tok(text, return_offsets_mapping=True, return_tensors="np", truncation=True, max_length=512) | |
| offs = enc["offset_mapping"][0] | |
| feeds = {"input_ids": enc["input_ids"].astype(np.int64), | |
| "attention_mask": enc["attention_mask"].astype(np.int64)} | |
| if "token_type_ids" in in_names: | |
| feeds["token_type_ids"] = np.zeros_like(enc["input_ids"], dtype=np.int64) | |
| probs = softmax(sess.run(None, feeds)[0][0]) | |
| ids = probs.argmax(-1) | |
| pb, pi = label2id["B-PER"], label2id["I-PER"] | |
| spans, cur = [], None | |
| for i, (s, e) in enumerate(offs): | |
| if s == e: | |
| if cur: spans.append(cur); cur = None | |
| continue | |
| lab = id2label[int(ids[i])] | |
| if lab == "O" and probs[i, pb] + probs[i, pi] >= PER_THRESHOLD: | |
| lab = "I-PER" if cur else "B-PER" | |
| if not lab.endswith("PER"): | |
| if cur: spans.append(cur); cur = None | |
| continue | |
| if lab == "B-PER" or cur is None: | |
| if cur: spans.append(cur) | |
| cur = [int(s), int(e)] | |
| else: | |
| cur[1] = int(e) | |
| if cur: spans.append(cur) | |
| return [text[a:b] for a, b in spans] | |
| # (input, fictional clean name behind the garble, failure mode) | |
| CASES = [ | |
| # 1) Heavy OCR garble -> missed entirely | |
| ("Pozwana ote tobodaaka wniosła sprzeciw.", "Bożena Łobodzińska", "heavy OCR garble -> MISSED"), | |
| ("Powód go Hea stawił się osobiście.", "Igor Heliasz", "heavy OCR garble -> MISSED"), | |
| ("Decyzję doręczono: ist yi daziak.", "Justyna Idziak", "heavy OCR garble -> MISSED"), | |
| # 2) Partial detection -> residue leak | |
| ("Wniosek złożył wiar Żoją w dniu 5 maja.", "Wiktor Żołądek", "partial -> only part caught (residue)"), | |
| ("WIESEAW KUE stawił się na rozprawie.", "Wiesław Kuc", "ALL-CAPS garble -> fragmented (gap leaks)"), | |
| ] | |
| if __name__ == "__main__": | |
| print(f"PER threshold = {PER_THRESHOLD}\n") | |
| for text, clean, mode in CASES: | |
| per = predict_per(text) | |
| print(f"# {mode} (fictional: {clean})") | |
| print(f" input : {text}") | |
| print(f" model : PER = {per if per else '— (nothing)'}") | |
| print() | |