alexwengg commited on
Commit
40e4036
·
verified ·
1 Parent(s): a4436fe

Add five-task Jev comparison reports

Browse files
reports/benchmark-jev-coreml.json ADDED
@@ -0,0 +1,103 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "chip" : "Apple M5 Pro",
3
+ "elapsed_s" : 2.7073659896850586,
4
+ "latency_ms" : {
5
+ "p50" : 4.4866669999999997,
6
+ "p95" : 9.6365829999999999
7
+ },
8
+ "lengths" : [
9
+ 128,
10
+ 256,
11
+ 512,
12
+ 1024
13
+ ],
14
+ "load_s" : 2.0176939964294434,
15
+ "questions" : 500,
16
+ "suites" : {
17
+ "emotion" : {
18
+ "accuracy" : 0.57999999999999996,
19
+ "buckets" : {
20
+ "L128" : 100
21
+ },
22
+ "dropped" : 0,
23
+ "latency_ms" : {
24
+ "p50" : 4.0849580000000003,
25
+ "p95" : 4.5919999999999996
26
+ },
27
+ "max_probability_delta_vs_reference" : 0.014793992042541504,
28
+ "n" : 100,
29
+ "reference_argmax_agreement" : 1,
30
+ "reference_compared" : 100,
31
+ "state_truncated" : 0
32
+ },
33
+ "news_topic" : {
34
+ "accuracy" : 0.96999999999999997,
35
+ "buckets" : {
36
+ "L128" : 86,
37
+ "L256" : 14
38
+ },
39
+ "dropped" : 0,
40
+ "latency_ms" : {
41
+ "p50" : 5.5224159999999998,
42
+ "p95" : 9.2874169999999996
43
+ },
44
+ "max_probability_delta_vs_reference" : 0.010490596294403076,
45
+ "n" : 100,
46
+ "reference_argmax_agreement" : 1,
47
+ "reference_compared" : 100,
48
+ "state_truncated" : 0
49
+ },
50
+ "prompt_injection" : {
51
+ "accuracy" : 0.65000000000000002,
52
+ "buckets" : {
53
+ "L128" : 96,
54
+ "L256" : 4
55
+ },
56
+ "dropped" : 0,
57
+ "latency_ms" : {
58
+ "p50" : 4.3983749999999997,
59
+ "p95" : 6.5922919999999996
60
+ },
61
+ "max_probability_delta_vs_reference" : 0.02795100212097168,
62
+ "n" : 100,
63
+ "reference_argmax_agreement" : 0.98999999999999999,
64
+ "reference_compared" : 100,
65
+ "state_truncated" : 0
66
+ },
67
+ "review_stars" : {
68
+ "accuracy" : 0.34999999999999998,
69
+ "buckets" : {
70
+ "L128" : 28,
71
+ "L256" : 35,
72
+ "L512" : 26,
73
+ "L1024" : 11
74
+ },
75
+ "dropped" : 0,
76
+ "latency_ms" : {
77
+ "p50" : 5.9012919999999998,
78
+ "p95" : 18.642083
79
+ },
80
+ "max_probability_delta_vs_reference" : 0.0017490386962890625,
81
+ "n" : 100,
82
+ "reference_argmax_agreement" : 1,
83
+ "reference_compared" : 100,
84
+ "state_truncated" : 0
85
+ },
86
+ "sms_spam" : {
87
+ "accuracy" : 0.57999999999999996,
88
+ "buckets" : {
89
+ "L128" : 100
90
+ },
91
+ "dropped" : 0,
92
+ "latency_ms" : {
93
+ "p50" : 4.1685420000000004,
94
+ "p95" : 5.0857080000000003
95
+ },
96
+ "max_probability_delta_vs_reference" : 0.020002603530883789,
97
+ "n" : 100,
98
+ "reference_argmax_agreement" : 1,
99
+ "reference_compared" : 100,
100
+ "state_truncated" : 0
101
+ }
102
+ }
103
+ }
reports/benchmark-jev-reference.json ADDED
@@ -0,0 +1,100 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "protocol": "Reproduce the five-task \"Laya vs Jev, measured\" table (brainfunctioncollapse.com/laya) on device.\n\nThe post gives task names, 100 labelled examples per task, laya 0.3.4 English checkpoint on an M1 Max\nGPU, and Jev's accuracy, but not its datasets, sampling or question wording. This script uses the\nobvious public dataset for each task, the first 100 rows of its test split (train for SMS spam, which\nhas no test split), and laya's own question wording where one exists. Jev is a closed API and is not\nmeasured here; its column is copied from the post. Both laya checkpoints are scored with the\nunmodified PyTorch runtime so the English numbers can be compared with the post and the\nmultilingual rows give the reference for the Core ML buckets (`FluidUseLaya benchmark`).\n\nOutputs:\n benchmark/jev-suites.jsonl the 500 questions with serialized states and gold labels\n benchmark/jev-reference-rows.jsonl multilingual PyTorch answers per row\n reports/benchmark-jev-reference.json accuracy per task for both checkpoints",
3
+ "n_per_task": 100,
4
+ "post": {
5
+ "news_topic": {
6
+ "laya_english_m1max": 0.93,
7
+ "jev": 0.92
8
+ },
9
+ "sms_spam": {
10
+ "laya_english_m1max": 0.96,
11
+ "jev": 0.96
12
+ },
13
+ "emotion": {
14
+ "laya_english_m1max": 0.45,
15
+ "jev": 0.53
16
+ },
17
+ "review_stars": {
18
+ "laya_english_m1max": 0.35,
19
+ "jev": 0.7
20
+ },
21
+ "prompt_injection": {
22
+ "laya_english_m1max": 0.65,
23
+ "jev": 0.71
24
+ },
25
+ "all": {
26
+ "laya_english_m1max": 0.668,
27
+ "jev": 0.764
28
+ }
29
+ },
30
+ "checkpoints": {
31
+ "multilingual": {
32
+ "max_len": 1024,
33
+ "seconds": 21.3,
34
+ "ms_per_question": 42.55,
35
+ "emotion": {
36
+ "n": 100,
37
+ "accuracy": 0.58,
38
+ "max_tokens": 82
39
+ },
40
+ "news_topic": {
41
+ "n": 100,
42
+ "accuracy": 0.97,
43
+ "max_tokens": 226
44
+ },
45
+ "prompt_injection": {
46
+ "n": 100,
47
+ "accuracy": 0.64,
48
+ "max_tokens": 177
49
+ },
50
+ "review_stars": {
51
+ "n": 100,
52
+ "accuracy": 0.35,
53
+ "max_tokens": 782
54
+ },
55
+ "sms_spam": {
56
+ "n": 100,
57
+ "accuracy": 0.58,
58
+ "max_tokens": 118
59
+ },
60
+ "all": {
61
+ "n": 500,
62
+ "accuracy": 0.624
63
+ }
64
+ },
65
+ "english": {
66
+ "max_len": 512,
67
+ "seconds": 48.9,
68
+ "ms_per_question": 97.87,
69
+ "emotion": {
70
+ "n": 100,
71
+ "accuracy": 0.65,
72
+ "max_tokens": 83
73
+ },
74
+ "news_topic": {
75
+ "n": 100,
76
+ "accuracy": 0.95,
77
+ "max_tokens": 232
78
+ },
79
+ "prompt_injection": {
80
+ "n": 100,
81
+ "accuracy": 0.79,
82
+ "max_tokens": 252
83
+ },
84
+ "review_stars": {
85
+ "n": 100,
86
+ "accuracy": 0.39,
87
+ "max_tokens": 512
88
+ },
89
+ "sms_spam": {
90
+ "n": 100,
91
+ "accuracy": 0.88,
92
+ "max_tokens": 116
93
+ },
94
+ "all": {
95
+ "n": 500,
96
+ "accuracy": 0.732
97
+ }
98
+ }
99
+ }
100
+ }