opensysone / source /results /20260917-fleet-progress /spark-b-2b-completed-summary.json
andyshu's picture
Back up verified OpenSysOne training snapshot and pinned source
2d5c26a verified
Raw History Blame Contribute Delete
8.4 kB
{
"checked_utc": "2026-09-17T02:11:24.308357+00:00",
"host": "spark-3e2a",
"campaign": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h",
"source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1",
"source_status": "",
"status": "validation_early_stop",
"finished_utc": "2026-09-17T01:59:29.971800+00:00",
"training_exit_code": 0,
"campaign_exit_code": 0,
"live_original_processes": " PID PPID STAT ELAPSED COMMAND\n",
"gpu_processes": "",
"memory": " total used free shared buff/cache available\nMem: 130661138432 4221775872 89653522432 4759552 37928640512 126439362560\nSwap: 0 0 0\n",
"best": {
"step": 2000,
"crossfit_nll": 0.2552943737691716,
"raw_nll": 0.25676974726429014,
"accuracy": 0.8984375,
"improved": true,
"family_accuracy": {
"arc": 0.890625,
"banking": 0.9609375,
"boolq": 0.84375,
"snli": 0.8984375
}
},
"validation_history": [
{
"step": 500,
"crossfit_nll": 0.36986992229694793,
"raw_nll": 0.45433305375755806,
"accuracy": 0.876953125,
"improved": true,
"family_accuracy": {
"arc": 0.859375,
"banking": 0.9453125,
"boolq": 0.8515625,
"snli": 0.8515625
}
},
{
"step": 1000,
"crossfit_nll": 0.3281878376433132,
"raw_nll": 0.4657101983683963,
"accuracy": 0.880859375,
"improved": true,
"family_accuracy": {
"arc": 0.890625,
"banking": 0.953125,
"boolq": 0.8359375,
"snli": 0.84375
}
},
{
"step": 1500,
"crossfit_nll": 0.3300765088337009,
"raw_nll": 0.3541424148692931,
"accuracy": 0.8828125,
"improved": false,
"family_accuracy": {
"arc": 0.8828125,
"banking": 0.9375,
"boolq": 0.8203125,
"snli": 0.890625
}
},
{
"step": 2000,
"crossfit_nll": 0.2552943737691716,
"raw_nll": 0.25676974726429014,
"accuracy": 0.8984375,
"improved": true,
"family_accuracy": {
"arc": 0.890625,
"banking": 0.9609375,
"boolq": 0.84375,
"snli": 0.8984375
}
},
{
"step": 2500,
"crossfit_nll": 0.2873950432295314,
"raw_nll": 0.34164803258559817,
"accuracy": 0.908203125,
"improved": false,
"family_accuracy": {
"arc": 0.890625,
"banking": 0.9765625,
"boolq": 0.8359375,
"snli": 0.9296875
}
},
{
"step": 3000,
"crossfit_nll": 0.26419258441127697,
"raw_nll": 0.343850982809214,
"accuracy": 0.91015625,
"improved": false,
"family_accuracy": {
"arc": 0.9140625,
"banking": 0.9921875,
"boolq": 0.8515625,
"snli": 0.8828125
}
},
{
"step": 3500,
"crossfit_nll": 0.2805558053510232,
"raw_nll": 0.32187738830487217,
"accuracy": 0.9140625,
"improved": false,
"family_accuracy": {
"arc": 0.90625,
"banking": 0.9765625,
"boolq": 0.859375,
"snli": 0.9140625
}
},
{
"step": 4000,
"crossfit_nll": 0.31337868570193705,
"raw_nll": 0.6221038123596042,
"accuracy": 0.896484375,
"improved": false,
"family_accuracy": {
"arc": 0.875,
"banking": 0.9765625,
"boolq": 0.875,
"snli": 0.859375
}
},
{
"step": 4500,
"crossfit_nll": 0.2841199300221572,
"raw_nll": 0.4348410999061469,
"accuracy": 0.8984375,
"improved": false,
"family_accuracy": {
"arc": 0.8515625,
"banking": 0.9765625,
"boolq": 0.8515625,
"snli": 0.9140625
}
},
{
"step": 5000,
"crossfit_nll": 0.279832908363464,
"raw_nll": 0.2879291938128496,
"accuracy": 0.90625,
"improved": false,
"family_accuracy": {
"arc": 0.859375,
"banking": 0.9921875,
"boolq": 0.8828125,
"snli": 0.890625
}
},
{
"step": 5500,
"crossfit_nll": 0.30398422164076616,
"raw_nll": 0.388565277749868,
"accuracy": 0.896484375,
"improved": false,
"family_accuracy": {
"arc": 0.8828125,
"banking": 0.9921875,
"boolq": 0.84375,
"snli": 0.8671875
}
},
{
"step": 6000,
"crossfit_nll": 0.35159254105589344,
"raw_nll": 0.40940693625302227,
"accuracy": 0.869140625,
"improved": false,
"family_accuracy": {
"arc": 0.8125,
"banking": 0.984375,
"boolq": 0.8359375,
"snli": 0.84375
}
}
],
"training_audit": {
"rows": 5960,
"first_step": 41,
"last_step": 6000,
"steps_contiguous": true,
"finite_losses": true,
"finite_gradients": true,
"max_gradient_norm": 267.5868225097656,
"max_loss": 7.096554353829106,
"max_cuda_allocated_bytes": 8786432512,
"median_step_seconds": 3.5843077870013076,
"last_100_median_step_seconds": 3.610054773500451,
"last_update": {
"step": 6000,
"loss": 0.0019449404207989573,
"gradient_norm": 0.06868523359298706,
"seconds": 3.64171387499664,
"decisions": 4,
"branches": 14,
"actual_branch_tokens": 1412,
"padded_branch_tokens": 1423,
"peak_cuda_allocated_bytes": 8786432512,
"lr_factor": 0.911059698995469
}
},
"correctness": {
"branch_chunks_1_probability_max_abs": 4.76837158203125e-07,
"branch_chunks_2_probability_max_abs": 0.0,
"branch_chunks_4_probability_max_abs": 6.556510925292969e-07,
"question_isolation_probability_max_abs": 2.0489096641540527e-08,
"candidate_permutation_probability_max_abs": 1.1920928955078125e-07,
"repeat_probability_max_abs": 0.0,
"tolerance_probability_abs": 0.0001
},
"checkpoint": {
"path": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h/training/checkpoint.pt",
"modified_utc": "2026-09-17T01:59:13.373975+00:00",
"sha256": "ab70dcb62432039d93ec71e200157250dc59c563404237e4dfad3662fd30f44f",
"has_optimizer": true,
"has_rng": true
},
"checkpoint_step": 6000,
"best_checkpoint": {
"path": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h/training/best.pt",
"modified_utc": "2026-09-16T21:45:31.514410+00:00",
"sha256": "be08efdc9121a64547b5c9331d5908eeb0978c4fbc962885fd34f9758ec89d11"
},
"stale_evaluations": 8,
"errors": {
"supervisor.log": [],
"training.log": []
},
"training_deadline": "2026-09-17T16:00:00+00:00",
"final_deadline": "2026-09-17T18:16:10+00:00",
"inference": "Eight evaluations since step 2000 failed to improve the fixed crossfit criterion. Resuming unchanged is unsupported by this trend; retain the best artifact for fleet selection. Further experimentation requires a separately recorded branch.",
"reserved_data_access": "No reserved predictions read; only train/validation logs, metadata and checkpoint state.",
"best_prediction_evidence": {
"rows": 512,
"best_validation_predictions_sha256": "980c9df18461f5de79e54701bfee1b5fbff19dd3b85125e64f619a88c60f80e4",
"validation_step_002000_predictions_sha256": "980c9df18461f5de79e54701bfee1b5fbff19dd3b85125e64f619a88c60f80e4",
"equal": true
},
"model_availability": {
"directory": "/home/andy/ai/models/opensysone",
"present": [
"Qwen3.5-2B-15852e8c"
],
"required_transfer": "Qwen3-4B-Instruct-2507-cdbee75f",
"available_disk_bytes": 3573074882560
},
"suggested_followup": {
"status": "recommendation_only_not_launched",
"parent": "freeze current best fixed-CV 4B after comparing GX10 and Spark A",
"preserve_original_2b_candidate": true,
"initialization": "weights_only_with_fresh_Adam_and_RNG",
"lr": 1e-05,
"head_lr": 1e-05,
"seed": 432,
"rank": 8,
"alpha": 16,
"effective_batch": 4,
"branch_batch_size": 1,
"two_pass": true,
"max_tokens": 512,
"schedule_steps": 5000,
"selection_metric": "crossfit_temperature_nll_v1",
"training_deadline": "2026-09-17T16:00:00+00:00",
"final_deadline": "2026-09-17T18:16:10+00:00",
"expected_new_steps": "Approximately 5000 given 13h40 and measured ~8-9s 4B updates plus verification/validation; stop by existing deadline."
}
}