jevify-smollm3-3b-apo-readout / jevify_config.json
Praveenrajus's picture
jevify-smollm3-3b-apo-readout@main: weights, recipe, results, model card (verified by reloading)
553e496 verified
Raw History Blame Contribute Delete
1.44 kB
{
"jevify_version": 1,
"tier": 2,
"method": "readout-lora",
"backbone": "HuggingFaceTB/SmolLM3-3B-checkpoints",
"backbone_revision": "cfb32d505f5025ec9be4e704f70cfbf5bdf8da94",
"base_model": "HuggingFaceTB/SmolLM3-3B-checkpoints",
"chat": true,
"recipe": {
"mode": "index",
"permutations": {
"choice": 2,
"score": 1,
"noul": 1
},
"prior_weight": {
"choice": 0.0,
"score": 0.0,
"noul": 0.0
},
"temperature": {
"choice": 1.2192,
"score": 1.4924,
"noul": 1.8579
},
"bias": {
"noul": -0.2
},
"state_last": false,
"prompt": "jevify"
},
"training": {
"objective": "proper scoring rule on the readout",
"arm": "sup",
"coherence_beta": 0.0,
"seed": 0,
"lr": 3e-05,
"epochs": 2,
"best_epoch": 1,
"best_val_loss": 0.6561664592338493,
"records_per_source_cap": 400,
"n_train_families": 5885,
"full_fine_tune": false,
"precision": "bf16",
"trained_on": "Praveenrajus/jev-bench train splits of the 16 sources that are not held out",
"heldout_sources": [
"clinc150",
"arc_challenge",
"yelp5",
"measuring_hate_speech",
"fever_evidence",
"strategyqa_grounded"
]
},
"evaluated_on": "Praveenrajus/jev-bench (test, 22,773 records)",
"lora": {
"r": 16,
"alpha": 32,
"target_modules": [
"gate_proj",
"q_proj",
"up_proj",
"v_proj",
"down_proj",
"k_proj",
"o_proj"
],
"trainable": 30228480,
"merged_at_load": true
}
}