File size: 4,357 Bytes
656ca59
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
{
  "chart": "beyond_laya",
  "unit": "percent (value x 100)",
  "source": "runs/macjev/report_r2/scoreboard/scoreboard.json via hf_staging/v3_card/chart_data.json",
  "delta_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
  "rows": [
    {
      "task": "Banking77",
      "metric": "accuracy, 77 intents",
      "v3": 0.6825,
      "best_laya": 0.4925,
      "best_laya_checkpoint": "typed",
      "delta_pts": 19.0,
      "delta_ci95_pts": [
        13.99,
        24.0
      ],
      "n": 400,
      "v3_entry": "banking77.v3",
      "laya_entry": "banking77.laya_best",
      "scoreboard_field": "metrics['banking77']",
      "protocol_label": "v3: held-out (banking77 never trained; CLINC150/HWU64 intents are) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
    },
    {
      "task": "Jailbreak",
      "metric": "balanced accuracy",
      "v3": 0.9036238842064085,
      "best_laya": 0.8311462000782389,
      "best_laya_checkpoint": "multilingual",
      "delta_pts": 7.25,
      "delta_ci95_pts": [
        2.99,
        11.43
      ],
      "n": 400,
      "v3_entry": "theme.guardrails_jailbreak.balanced_accuracy.v3",
      "laya_entry": "theme.guardrails_jailbreak.balanced_accuracy.laya_best",
      "scoreboard_field": "metrics['theme.guardrails_jailbreak.balanced_accuracy']",
      "protocol_label": "v3: held-out source (other permissive jailbreak sets + teacher) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
    },
    {
      "task": "Toxicity",
      "metric": "macro-F1",
      "v3": 0.7089254055198327,
      "best_laya": 0.41360068097985436,
      "best_laya_checkpoint": "multilingual",
      "delta_pts": 29.53,
      "delta_ci95_pts": [
        24.65,
        34.37
      ],
      "n": 400,
      "v3_entry": "theme.moderation_toxicity.macro_f1.v3",
      "laya_entry": "theme.moderation_toxicity.macro_f1.laya_best",
      "scoreboard_field": "metrics['theme.moderation_toxicity.macro_f1']",
      "protocol_label": "v3: held-out source (civil_comments + teacher; toxic-chat eval-only) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
    },
    {
      "task": "Model routing",
      "metric": "accuracy",
      "v3": 0.9624060150375939,
      "best_laya": 0.6591478696741855,
      "best_laya_checkpoint": "typed",
      "delta_pts": 30.33,
      "delta_ci95_pts": [
        26.06,
        35.09
      ],
      "n": 399,
      "v3_entry": "theme.model_routing_domain.v3",
      "laya_entry": "theme.model_routing_domain.laya_best",
      "scoreboard_field": "metrics['theme.model_routing_domain']",
      "protocol_label": "v3: near-domain (teacher-written; gsm8k/mbpp/AG rows eval-only) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
    },
    {
      "task": "MASSIVE intent",
      "metric": "37 held-out locales",
      "v3": 0.6551351351351351,
      "best_laya": 0.36108108108108106,
      "best_laya_checkpoint": "multilingual",
      "delta_pts": 29.41,
      "delta_ci95_pts": [
        27.54,
        31.3
      ],
      "n": 3700,
      "v3_entry": "massive51.unseen37.macro_accuracy.v3",
      "laya_entry": "massive51.unseen37.macro_accuracy.laya_best",
      "scoreboard_field": "metrics['massive51.unseen37.macro_accuracy']",
      "protocol_label": "v3: held-out (37 languages never trained) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
    }
  ],
  "footnote": "Laya numbers: official checkpoints (English, typed-decisions, multilingual) re-run by us on identical rows with their shipped temperatures and\ndefault token budgets; the best of the three is shown per task. v3 trained on same-kind tasks from other datasets, never on these eval rows:\nintent (CLINC150/HWU64; Banking77 never trained), jailbreak (other permissive sets + teacher), toxicity (civil_comments + teacher; toxic-chat\neval-only), routing (teacher-written; gsm8k/mbpp/AG rows eval-only), MASSIVE in 14 other locales (no MASSIVE rows in these 37).\nn = 400 / 400 / 400 / 399 / 3,700 (37 x 100). Every gap's paired 95% bootstrap CI excludes zero. Plotted values: figures/beyond_laya.json."
}