Token Classification
Transformers
Safetensors
modernbert
semantic-router
vela
hallucination-detection
Xunzhuo commited on
Commit
469c730
·
verified ·
1 Parent(s): 521cd05

Publish Vela Halu model and evaluation scores

Browse files
Files changed (3) hide show
  1. README.md +7 -5
  2. model.safetensors +1 -1
  3. scores.json +64 -64
README.md CHANGED
@@ -32,16 +32,18 @@ Vela Halu finds answer spans unsupported by the supplied evidence, for grounded
32
 
33
  Labels are `supported` (0) and `hallucinated` (1). Spans use Unicode character offsets in the answer. Evidence support is distinct from real-world factual truth; an empty result does not guarantee correctness.
34
 
 
 
35
  ## Evaluation
36
 
37
- Scores (0-100) on the 10,698-example held-out test set. Higher is better.
38
 
39
  | Metric | Vela Halu |
40
  |---|---:|
41
- | Overall span F1 | 63.85 |
42
- | Overall example F1 | 87.12 |
43
- | Code span F1 | 51.07 |
44
- | Tool output span F1 | 59.86 |
45
 
46
  Span F1 measures character overlap; example F1 measures whether an answer contains any hallucinated span. Evaluation uses bfloat16, an 8,192-token limit, `only_first` pair truncation, and token scores strictly above 0.5. [Full scores](./scores.json) include all source groups and truncation statistics.
47
 
 
32
 
33
  Labels are `supported` (0) and `hallucinated` (1). Spans use Unicode character offsets in the answer. Evidence support is distinct from real-world factual truth; an empty result does not guarantee correctness.
34
 
35
+ Answers requiring arithmetic or multi-step reasoning beyond explicit context may be incorrectly flagged.
36
+
37
  ## Evaluation
38
 
39
+ Scores (0-100) on the 10,698-example fixed evaluation set. Higher is better.
40
 
41
  | Metric | Vela Halu |
42
  |---|---:|
43
+ | Overall span F1 | 64.68 |
44
+ | Overall example F1 | 87.49 |
45
+ | Code span F1 | 53.04 |
46
+ | Tool output span F1 | 64.28 |
47
 
48
  Span F1 measures character overlap; example F1 measures whether an answer contains any hallucinated span. Evaluation uses bfloat16, an 8,192-token limit, `only_first` pair truncation, and token scores strictly above 0.5. [Full scores](./scores.json) include all source groups and truncation statistics.
49
 
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d303e4ef50590c103e30e4ab9e16a4f015d14fdb52d1c1935fba55191df22441
3
  size 1230141424
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0f51bf4a6e1462e88733c36e1e59fa96c1188c8c470de7829b51c95298247be6
3
  size 1230141424
scores.json CHANGED
@@ -9,92 +9,92 @@
9
  "threshold": 0.5,
10
  "overall": {
11
  "n": 10698,
12
- "span_f1": 0.6385179438029605,
13
- "span_p": 0.7104274748432198,
14
- "span_r": 0.5798277303088283,
15
- "ex_f1": 0.8712474983322215,
16
- "ex_p": 0.8894942959305295,
17
- "ex_r": 0.8537342703056054,
18
- "iou": 0.6702612740887652,
19
- "clean_fpr": 0.1417340030574361
20
  },
21
  "by_source": {
22
  "lettucedetect-acl": {
23
  "n": 440,
24
- "span_f1": 0.5702109294923666,
25
- "span_p": 0.7242702480471427,
26
- "span_r": 0.47019572953736655,
27
- "ex_f1": 0.8521739130434782,
28
- "ex_p": 0.8166666666666667,
29
- "ex_r": 0.8909090909090909,
30
- "iou": 0.6280894829592452,
31
- "clean_fpr": 0.2
32
  },
33
  "lettucedetect-code-agent": {
34
  "n": 2015,
35
- "span_f1": 0.5107270439913081,
36
- "span_p": 0.654087400055142,
37
- "span_r": 0.4189116111730364,
38
- "ex_f1": 0.7675568743818002,
39
- "ex_p": 0.7698412698412699,
40
- "ex_r": 0.7652859960552268,
41
- "iou": 0.5639770662416372,
42
- "clean_fpr": 0.23176823176823177
43
  },
44
  "lettucedetect-readme": {
45
  "n": 641,
46
- "span_f1": 0.7605563835072032,
47
- "span_p": 0.8224865694551036,
48
- "span_r": 0.7072993664202746,
49
- "ex_f1": 0.9209302325581395,
50
- "ex_p": 0.9138461538461539,
51
- "ex_r": 0.928125,
52
- "iou": 0.7809105019493161,
53
- "clean_fpr": 0.08722741433021806
54
  },
55
  "lettucedetect-tool-output": {
56
  "n": 617,
57
- "span_f1": 0.598570881714942,
58
- "span_p": 0.7726898369296656,
59
- "span_r": 0.4884931792148287,
60
- "ex_f1": 0.8082901554404145,
61
- "ex_p": 0.8634686346863468,
62
- "ex_r": 0.7597402597402597,
63
- "iou": 0.6758541479351077,
64
- "clean_fpr": 0.11974110032362459
65
  },
66
  "lettucedetect-wikipedia": {
67
  "n": 1388,
68
- "span_f1": 0.7253597853484023,
69
- "span_p": 0.7776770459221033,
70
- "span_r": 0.6796379814724525,
71
- "ex_f1": 0.9248067463106114,
72
- "ex_p": 0.9026063100137174,
73
- "ex_r": 0.9481268011527377,
74
- "iou": 0.7810017784449338,
75
- "clean_fpr": 0.10230547550432277
76
  },
77
  "psiloqa": {
78
  "n": 2897,
79
- "span_f1": 0.7077433515248708,
80
- "span_p": 0.7179186831478684,
81
- "span_r": 0.6978524265074418,
82
- "ex_f1": 0.94276875483372,
83
- "ex_p": 0.9553291536050157,
84
- "ex_r": 0.9305343511450381,
85
- "iou": 0.6217888817095617,
86
- "clean_fpr": 0.41155234657039713
87
  },
88
  "ragtruth": {
89
  "n": 2700,
90
- "span_f1": 0.5119871112687714,
91
- "span_p": 0.7006978861130778,
92
- "span_r": 0.40335596259766876,
93
- "ex_f1": 0.7392075694855116,
94
- "ex_p": 0.8355614973262032,
95
- "ex_r": 0.662778366914104,
96
- "iou": 0.7239864627418552,
97
- "clean_fpr": 0.07000569151963575
98
  }
99
  },
100
  "truncation": {
 
9
  "threshold": 0.5,
10
  "overall": {
11
  "n": 10698,
12
+ "span_f1": 0.6468063721645629,
13
+ "span_p": 0.7001953973777478,
14
+ "span_r": 0.6009822273490223,
15
+ "ex_f1": 0.8749271864858118,
16
+ "ex_p": 0.8913190912173619,
17
+ "ex_r": 0.8591273083837229,
18
+ "iou": 0.6791407403861177,
19
+ "clean_fpr": 0.13998689670233674
20
  },
21
  "by_source": {
22
  "lettucedetect-acl": {
23
  "n": 440,
24
+ "span_f1": 0.6033354880413704,
25
+ "span_p": 0.720407533189256,
26
+ "span_r": 0.5189946619217082,
27
+ "ex_f1": 0.8590021691973969,
28
+ "ex_p": 0.8215767634854771,
29
+ "ex_r": 0.9,
30
+ "iou": 0.6506637226712354,
31
+ "clean_fpr": 0.19545454545454546
32
  },
33
  "lettucedetect-code-agent": {
34
  "n": 2015,
35
+ "span_f1": 0.5304085445334227,
36
+ "span_p": 0.6548436524601328,
37
+ "span_r": 0.44571299290373134,
38
+ "ex_f1": 0.7826510721247564,
39
+ "ex_p": 0.7736030828516378,
40
+ "ex_r": 0.7919132149901381,
41
+ "iou": 0.5799284591741568,
42
+ "clean_fpr": 0.23476523476523475
43
  },
44
  "lettucedetect-readme": {
45
  "n": 641,
46
+ "span_f1": 0.7668571829306778,
47
+ "span_p": 0.8180594000149622,
48
+ "span_r": 0.7216869060190074,
49
+ "ex_f1": 0.9153605015673981,
50
+ "ex_p": 0.9182389937106918,
51
+ "ex_r": 0.9125,
52
+ "iou": 0.7844094959106822,
53
+ "clean_fpr": 0.08099688473520249
54
  },
55
  "lettucedetect-tool-output": {
56
  "n": 617,
57
+ "span_f1": 0.6427959358550618,
58
+ "span_p": 0.7796585003711952,
59
+ "span_r": 0.5468082890763303,
60
+ "ex_f1": 0.8273504273504273,
61
+ "ex_p": 0.8736462093862816,
62
+ "ex_r": 0.7857142857142857,
63
+ "iou": 0.7027343010780659,
64
+ "clean_fpr": 0.11326860841423948
65
  },
66
  "lettucedetect-wikipedia": {
67
  "n": 1388,
68
+ "span_f1": 0.7343291203881016,
69
+ "span_p": 0.7708389171803806,
70
+ "span_r": 0.7011214041930766,
71
+ "ex_f1": 0.9298369950389794,
72
+ "ex_p": 0.9149232914923291,
73
+ "ex_r": 0.9452449567723343,
74
+ "iou": 0.7905999840786986,
75
+ "clean_fpr": 0.08789625360230548
76
  },
77
  "psiloqa": {
78
  "n": 2897,
79
+ "span_f1": 0.7084256755107189,
80
+ "span_p": 0.7054881112533835,
81
+ "span_r": 0.7113878053564034,
82
+ "ex_f1": 0.9406055900621118,
83
+ "ex_p": 0.9569510268562401,
84
+ "ex_r": 0.9248091603053435,
85
+ "iou": 0.6254885367916755,
86
+ "clean_fpr": 0.3935018050541516
87
  },
88
  "ragtruth": {
89
  "n": 2700,
90
+ "span_f1": 0.5254962135743007,
91
+ "span_p": 0.6738326440634735,
92
+ "span_r": 0.43068574822129324,
93
+ "ex_f1": 0.7485448195576252,
94
+ "ex_p": 0.8296774193548387,
95
+ "ex_r": 0.6818663838812301,
96
+ "iou": 0.727708569558819,
97
+ "clean_fpr": 0.07512805919180421
98
  }
99
  },
100
  "truncation": {