Jev-Style-0.8B-Decision-v3 / figures /beyond_laya.json
chaoliangUNSW's picture
Release Jev-Style-0.8B-Decision-v3
656ca59 verified
Raw History Blame
4.36 kB
{
"chart": "beyond_laya",
"unit": "percent (value x 100)",
"source": "runs/macjev/report_r2/scoreboard/scoreboard.json via hf_staging/v3_card/chart_data.json",
"delta_ci_method": "case-cluster bootstrap within suites, 2000 resamples (scoreboard n_boot)",
"rows": [
{
"task": "Banking77",
"metric": "accuracy, 77 intents",
"v3": 0.6825,
"best_laya": 0.4925,
"best_laya_checkpoint": "typed",
"delta_pts": 19.0,
"delta_ci95_pts": [
13.99,
24.0
],
"n": 400,
"v3_entry": "banking77.v3",
"laya_entry": "banking77.laya_best",
"scoreboard_field": "metrics['banking77']",
"protocol_label": "v3: held-out (banking77 never trained; CLINC150/HWU64 intents are) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
},
{
"task": "Jailbreak",
"metric": "balanced accuracy",
"v3": 0.9036238842064085,
"best_laya": 0.8311462000782389,
"best_laya_checkpoint": "multilingual",
"delta_pts": 7.25,
"delta_ci95_pts": [
2.99,
11.43
],
"n": 400,
"v3_entry": "theme.guardrails_jailbreak.balanced_accuracy.v3",
"laya_entry": "theme.guardrails_jailbreak.balanced_accuracy.laya_best",
"scoreboard_field": "metrics['theme.guardrails_jailbreak.balanced_accuracy']",
"protocol_label": "v3: held-out source (other permissive jailbreak sets + teacher) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
},
{
"task": "Toxicity",
"metric": "macro-F1",
"v3": 0.7089254055198327,
"best_laya": 0.41360068097985436,
"best_laya_checkpoint": "multilingual",
"delta_pts": 29.53,
"delta_ci95_pts": [
24.65,
34.37
],
"n": 400,
"v3_entry": "theme.moderation_toxicity.macro_f1.v3",
"laya_entry": "theme.moderation_toxicity.macro_f1.laya_best",
"scoreboard_field": "metrics['theme.moderation_toxicity.macro_f1']",
"protocol_label": "v3: held-out source (civil_comments + teacher; toxic-chat eval-only) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
},
{
"task": "Model routing",
"metric": "accuracy",
"v3": 0.9624060150375939,
"best_laya": 0.6591478696741855,
"best_laya_checkpoint": "typed",
"delta_pts": 30.33,
"delta_ci95_pts": [
26.06,
35.09
],
"n": 399,
"v3_entry": "theme.model_routing_domain.v3",
"laya_entry": "theme.model_routing_domain.laya_best",
"scoreboard_field": "metrics['theme.model_routing_domain']",
"protocol_label": "v3: near-domain (teacher-written; gsm8k/mbpp/AG rows eval-only) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
},
{
"task": "MASSIVE intent",
"metric": "37 held-out locales",
"v3": 0.6551351351351351,
"best_laya": 0.36108108108108106,
"best_laya_checkpoint": "multilingual",
"delta_pts": 29.41,
"delta_ci95_pts": [
27.54,
31.3
],
"n": 3700,
"v3_entry": "massive51.unseen37.macro_accuracy.v3",
"laya_entry": "massive51.unseen37.macro_accuracy.laya_best",
"scoreboard_field": "metrics['massive51.unseen37.macro_accuracy']",
"protocol_label": "v3: held-out (37 languages never trained) / Laya: held-out. Laya numbers: official checkpoints re-run by us on identical rows, shipped temperatures."
}
],
"footnote": "Laya numbers: official checkpoints (English, typed-decisions, multilingual) re-run by us on identical rows with their shipped temperatures and\ndefault token budgets; the best of the three is shown per task. v3 trained on same-kind tasks from other datasets, never on these eval rows:\nintent (CLINC150/HWU64; Banking77 never trained), jailbreak (other permissive sets + teacher), toxicity (civil_comments + teacher; toxic-chat\neval-only), routing (teacher-written; gsm8k/mbpp/AG rows eval-only), MASSIVE in 14 other locales (no MASSIVE rows in these 37).\nn = 400 / 400 / 400 / 399 / 3,700 (37 x 100). Every gap's paired 95% bootstrap CI excludes zero. Plotted values: figures/beyond_laya.json."
}