{ "experiment": "E0j five-seed confirmation", "seeds": [ 0, 1, 2, 3, 4 ], "created_at": "2026-09-27T15:29:06.749690+00:00", "data": { "train": { "format_version": 1, "sha256": "5b273d74d4eed80f75763af64cfaf3fabae1d2a31a0b67551a5a753b6972e22e", "cases": 1080, "decisions": 5400, "profile": "hybrid-train", "role": "train", "source": "LocalLLaMA/typed-decisions", "revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8", "source_split": "train", "split_seed": 42 }, "validation": { "format_version": 1, "sha256": "95fe23fc9e82f9992c5056e7ab7ff76888aa614f49aa6d48929aba5d4605812a", "cases": 120, "decisions": 600, "profile": "hybrid-validation", "role": "validation", "source": "LocalLLaMA/typed-decisions", "revision": "f7a2487edd7a043a5441a5e9ccc7fe5ddbd9ebe8", "source_split": "train", "split_seed": 42 }, "test": { "format_version": 1, "sha256": "dd34a1df70013b7835b57aa474e36bf432f6e586a98e960e2b43b80ab7b75d06", "cases": 400, "decisions": 2000, "profile": "typed-decisions-test", "role": "test", "source_split": "test", "parent_sha256": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435" } }, "heldout_manifest": { "format_version": 1, "sha256": "10fb671245cdac1ee09e9c4d2b9b09829fbf1abd06e0954fddee0b9ec8519854", "cases": 3516, "decisions": 5116, "profile": "heldout-eight", "role": "test", "parent_bundles": { "core": "477c9cbd4f9eaf8844fa4d6870560e751960735e7441b803306d6e011e689435", "language_en": "a3d41f5060826770f0f28b16a0aff26b4294691423624349db6b008789ea9096" }, "purpose": "Base vs official typed-decisions specialist vs trained Laya-only input interface; no new training." }, "max_len": 1024, "head_max_len": 256, "epochs": null, "min_epochs": 10, "early_stopping_patience": 3, "early_stopping_metric": "validation.soft_cross_entropy", "early_stopping_min_delta": 0.0, "microbatch": 8, "accumulation": 4, "effective_batch": 32, "interface_lr": 0.0001, "weight_decay": 0.01, "gradient_clip": 1, "training_scope": "bridge", "training_dropout": false, "selection": "minimum validation soft cross-entropy among trained epochs", "release_candidate_selection": "lowest selected validation soft cross-entropy across all five seeds; tie goes to the lowest seed; no test-based selection", "device": "cuda:1", "deterministic": true, "native_sdk_calibration": true, "publish_policy": "provide user-requested local bundle; preserve existing card; no assistant upload", "baseline_predictions_sha256": "37dff666a00a340cd6e37f775dc60df33a3c6eb78f76bfee7f094443c225b6e8", "run_order": [ 0, 2, 3, 1, 4 ], "seed_0_reference": "completed controlled pilot, unchanged", "comparison": "LR is the only training change from E0i; min10/patience3 and all other settings fixed", "scope": "adaptive experimental follow-up selected after the seed-0 pilot; not a fresh blind test", "replay_evidence": "Two fresh seed-0 LR 1e-4 processes matched losses, metrics and tensors for two epochs; reported seed-0 metric prefix matched. Verified after the sweep.", "driver_sha256": { "/mnt/storage/ariadne/scripts/run_frozen_linear_lr_confirmation.py": "f7c3841277682cbae43f4eb51fa3cb5fc87fd53068607206240ce137b7e94a9b" }, "release_replay_sha256": "dd3479114e8af11d06708febc1e86b887cc6e3cc5ce1523a249e54b9432412ec", "base_model": "convaiinnovations/laya", "base_revision": "55cf4c4ebb4ebe31b2550e8bdf3bd21b99753851", "training_sources_sha256": { "ariadne_bench/reproducibility.py": "5b0c161c596f25277b3326c645df53d5d1f2dc0272da9e77bc4a4c0d56188b6e", "ariadne_bench/frozen_input_interface.py": "253f0186dac6e4a7f4b0fcf3c33449c3b3882cf8be538dca6eeca8d9e4ce9b83", "ariadne_bench/full_input_interface.py": "a8f440a2185085c64896300a243ac52a9619c4d05a1b4e43c50c2c32aa1da996", "ariadne_bench/full_finetune.py": "6067be8cf837d254b919c841c9a9851e1546a413dbf6324ff242e5d2f56c9734", "ariadne_bench/interfaces.py": "19700fb8170b2460401900cb96ebdff22bd7bf778ca5132e98d56ba68f344187", "ariadne_bench/experiments/align.py": "31e2733dcc223a1e1a6250d41c4158f84091f1cb4067d61388e06b5b01ae34c4" } }