File size: 4,365 Bytes
eaea151 5f2a11f eaea151 5f2a11f eaea151 5f2a11f eaea151 5f2a11f eaea151 5f2a11f eaea151 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 | {
"experiment": "E0j five-seed confirmation",
"seeds": [
0,
1,
2,
3,
4
],
"created_at": "2026-09-27T15:29:06.749690+00:00",
"data": {
"train": {
"format_version": 1,
"sha256": "5b273d74d4eed80f75763af64cfaf3fabae1d2a31a0b67551a5a753b6972e22e",
"cases": 1080,
"decisions": 5400,
"profile": "hybrid-train",
"role": "train",
"source": "LocalLLaMA/typed-decisions",
"revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8",
"source_split": "train",
"split_seed": 42
},
"validation": {
"format_version": 1,
"sha256": "95fe23fc9e82f9992c5056e7ab7ff76888aa614f49aa6d48929aba5d4605812a",
"cases": 120,
"decisions": 600,
"profile": "hybrid-validation",
"role": "validation",
"source": "LocalLLaMA/typed-decisions",
"revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8",
"source_split": "train",
"split_seed": 42
},
"test": {
"format_version": 1,
"sha256": "dd34a1df70013b7835b57aa474e36bf432f6e586a98e960e2b43b80ab7b75d06",
"cases": 400,
"decisions": 2000,
"profile": "typed-decisions-test",
"role": "test",
"source_split": "test",
"parent_sha256": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435"
}
},
"heldout_manifest": {
"format_version": 1,
"sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854",
"cases": 3516,
"decisions": 5116,
"profile": "heldout-eight",
"role": "test",
"parent_bundles": {
"core": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435",
"language_en": "a3d41f5060826770f0f28b16a0aff26b4294691423624349db6b008789ea9096"
},
"purpose": "Base vs official typed-decisions specialist vs trained Laya-only input interface; no new training."
},
"max_len": 1024,
"head_max_len": 256,
"epochs": null,
"min_epochs": 10,
"early_stopping_patience": 3,
"early_stopping_metric": "validation.soft_cross_entropy",
"early_stopping_min_delta": 0.0,
"microbatch": 8,
"accumulation": 4,
"effective_batch": 32,
"interface_lr": 0.0001,
"weight_decay": 0.01,
"gradient_clip": 1,
"training_scope": "bridge",
"training_dropout": false,
"selection": "minimum validation soft cross-entropy among trained epochs",
"release_candidate_selection": "lowest selected validation soft cross-entropy across all five seeds; tie goes to the lowest seed; no test-based selection",
"device": "cuda:1",
"deterministic": true,
"native_sdk_calibration": true,
"publish_policy": "provide user-requested local bundle; preserve existing card; no assistant upload",
"baseline_predictions_sha256": "37dff666a00a340cd6e37f775dc60df33a3c6eb78f76bfee7f094443c225b6e8",
"run_order": [
0,
2,
3,
1,
4
],
"seed_0_reference": "completed controlled pilot, unchanged",
"comparison": "LR is the only training change from E0i; min10/patience3 and all other settings fixed",
"scope": "adaptive experimental follow-up selected after the seed-0 pilot; not a fresh blind test",
"replay_evidence": "Two fresh seed-0 LR 1e-4 processes matched losses, metrics and tensors for two epochs; reported seed-0 metric prefix matched. Verified after the sweep.",
"driver_sha256": {
"/mnt/storage/ariadne/scripts/run_frozen_linear_lr_confirmation.py": "f7c3841277682cbae43f4eb51fa3cb5fc87fd53068607206240ce137b7e94a9b"
},
"release_replay_sha256": "dd3479114e8af11d06708febc1e86b887cc6e3cc5ce1523a249e54b9432412ec",
"base_model": "convaiinnovations/laya",
"base_revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851",
"training_sources_sha256": {
"ariadne_bench/reproducibility.py": "5b0c161c596f25277b3326c645df53d5d1f2dc0272da9e77bc4a4c0d56188b6e",
"ariadne_bench/frozen_input_interface.py": "253f0186dac6e4a7f4b0fcf3c33449c3b3882cf8be538dca6eeca8d9e4ce9b83",
"ariadne_bench/full_input_interface.py": "a8f440a2185085c64896300a243ac52a9619c4d05a1b4e43c50c2c32aa1da996",
"ariadne_bench/full_finetune.py": "6067be8cf837d254b919c841c9a9851e1546a413dbf6324ff242e5d2f56c9734",
"ariadne_bench/interfaces.py": "19700fb8170b2460401900cb96ebdff22bd7bf778ca5132e98d56ba68f344187",
"ariadne_bench/experiments/align.py": "31e2733dcc223a1e1a6250d41c4158f84091f1cb4067d61388e06b5b01ae34c4"
}
}
|