{ "target_model": "Accio-Lab/occamy-1.0", "target_revision": "c1ce84770260c4137712cf22115574c4d06b993a", "initialization_model": "Qwen/Qwen3.6-35B-A3B", "initialization_revision": "995ad96eacd98c81ed38be0c5b274b04031597b0", "base_frozen": true, "trainable_parameters": 8392704, "head_parameters": 844640768, "steps": 512, "train_samples": 256, "heldout_samples": 32, "sequence_length": 512, "batch_size": 1, "optimizer": "AdamW", "learning_rate": 1e-05, "weight_decay": 0, "gradient_clip_norm": 1.0, "data": { "repo": "Accio-Lab/occamy-data-1.0", "revision": "7dd8c12a4382edce172da01b88bd3257ead336cf", "file_sha256": "310c0ebcd90d92a4cab3f2f8ba96ee42b8dbaf0261278de1ae1c73b521656202", "selection": "Seed 42; one last substantial assistant continuation per text-only session; exact-answer dedup; 256 training and 32 heldout sessions. Up to 128 local-context tokens and 382 assistant tokens. Unnormalized tool_call roles excluded; images excluded. No identical answer or token-window overlap. Heldout excluded from MTP training, not guaranteed unseen by the base." }, "objective": "Cross entropy against frozen Occamy greedy next-token predictions at assistant positions", "external_teacher": false, "heldout_before_nll": 2.2849533979471612, "heldout_after_nll": 2.603421883325693, "heldout_target_top1_before": 0.7357322577928339, "heldout_target_top1_after": 0.7465660697511188, "nll_measurement_note": "Teacher-forced data-answer NLL and target top-1 agreement evaluated after BF16 export. Hard target training is specific to greedy decoding, not a sampling validation.", "export_note": "19 BF16 tensors; only fusion and pre-fusion normalization parameters were trainable.", "checkpoint_save_restore_verified": true, "epochs": 2 }