#!/usr/bin/env python3 """PyTorch / transformers inference for the HerBERT Polish legal NER model. pip install torch transformers Run from the repo root: python examples/inference_pytorch.py """ from pathlib import Path from transformers import pipeline ROOT = Path(__file__).resolve().parent.parent ner = pipeline( "token-classification", model=str(ROOT), tokenizer=str(ROOT), aggregation_strategy="first", # group B-/I- subwords into whole entities ) SAMPLES = [ "Pozwany Jan Kowalski, zam. ul. Słoneczna 5 w Krakowie, PESEL 02070803628.", "Powódka Anna Nowak-Kowalska, e-mail a.nowak@example.pl, tel. 501 234 567.", ] if __name__ == "__main__": for text in SAMPLES: print("\n" + text) for ent in ner(text): print(f" {ent['entity_group']:8} [{ent['start']:>3}:{ent['end']:<3}] " f"{ent['word']!r} ({ent['score']:.2f})") # NOTE: for anonymisation, prefer a recall-first PER threshold (see # examples/inference_onnx.py) over the arg-max grouping the pipeline uses.