{ "checked_utc": "2026-09-17T02:11:24.308357+00:00", "host": "spark-3e2a", "campaign": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h", "source_commit": "4a60423c39d70f8d50472ce4f4f7fa4a4bd9fce1", "source_status": "", "status": "validation_early_stop", "finished_utc": "2026-09-17T01:59:29.971800+00:00", "training_exit_code": 0, "campaign_exit_code": 0, "live_original_processes": " PID PPID STAT ELAPSED COMMAND\n", "gpu_processes": "", "memory": " total used free shared buff/cache available\nMem: 130661138432 4221775872 89653522432 4759552 37928640512 126439362560\nSwap: 0 0 0\n", "best": { "step": 2000, "crossfit_nll": 0.2552943737691716, "raw_nll": 0.25676974726429014, "accuracy": 0.8984375, "improved": true, "family_accuracy": { "arc": 0.890625, "banking": 0.9609375, "boolq": 0.84375, "snli": 0.8984375 } }, "validation_history": [ { "step": 500, "crossfit_nll": 0.36986992229694793, "raw_nll": 0.45433305375755806, "accuracy": 0.876953125, "improved": true, "family_accuracy": { "arc": 0.859375, "banking": 0.9453125, "boolq": 0.8515625, "snli": 0.8515625 } }, { "step": 1000, "crossfit_nll": 0.3281878376433132, "raw_nll": 0.4657101983683963, "accuracy": 0.880859375, "improved": true, "family_accuracy": { "arc": 0.890625, "banking": 0.953125, "boolq": 0.8359375, "snli": 0.84375 } }, { "step": 1500, "crossfit_nll": 0.3300765088337009, "raw_nll": 0.3541424148692931, "accuracy": 0.8828125, "improved": false, "family_accuracy": { "arc": 0.8828125, "banking": 0.9375, "boolq": 0.8203125, "snli": 0.890625 } }, { "step": 2000, "crossfit_nll": 0.2552943737691716, "raw_nll": 0.25676974726429014, "accuracy": 0.8984375, "improved": true, "family_accuracy": { "arc": 0.890625, "banking": 0.9609375, "boolq": 0.84375, "snli": 0.8984375 } }, { "step": 2500, "crossfit_nll": 0.2873950432295314, "raw_nll": 0.34164803258559817, "accuracy": 0.908203125, "improved": false, "family_accuracy": { "arc": 0.890625, "banking": 0.9765625, "boolq": 0.8359375, "snli": 0.9296875 } }, { "step": 3000, "crossfit_nll": 0.26419258441127697, "raw_nll": 0.343850982809214, "accuracy": 0.91015625, "improved": false, "family_accuracy": { "arc": 0.9140625, "banking": 0.9921875, "boolq": 0.8515625, "snli": 0.8828125 } }, { "step": 3500, "crossfit_nll": 0.2805558053510232, "raw_nll": 0.32187738830487217, "accuracy": 0.9140625, "improved": false, "family_accuracy": { "arc": 0.90625, "banking": 0.9765625, "boolq": 0.859375, "snli": 0.9140625 } }, { "step": 4000, "crossfit_nll": 0.31337868570193705, "raw_nll": 0.6221038123596042, "accuracy": 0.896484375, "improved": false, "family_accuracy": { "arc": 0.875, "banking": 0.9765625, "boolq": 0.875, "snli": 0.859375 } }, { "step": 4500, "crossfit_nll": 0.2841199300221572, "raw_nll": 0.4348410999061469, "accuracy": 0.8984375, "improved": false, "family_accuracy": { "arc": 0.8515625, "banking": 0.9765625, "boolq": 0.8515625, "snli": 0.9140625 } }, { "step": 5000, "crossfit_nll": 0.279832908363464, "raw_nll": 0.2879291938128496, "accuracy": 0.90625, "improved": false, "family_accuracy": { "arc": 0.859375, "banking": 0.9921875, "boolq": 0.8828125, "snli": 0.890625 } }, { "step": 5500, "crossfit_nll": 0.30398422164076616, "raw_nll": 0.388565277749868, "accuracy": 0.896484375, "improved": false, "family_accuracy": { "arc": 0.8828125, "banking": 0.9921875, "boolq": 0.84375, "snli": 0.8671875 } }, { "step": 6000, "crossfit_nll": 0.35159254105589344, "raw_nll": 0.40940693625302227, "accuracy": 0.869140625, "improved": false, "family_accuracy": { "arc": 0.8125, "banking": 0.984375, "boolq": 0.8359375, "snli": 0.84375 } } ], "training_audit": { "rows": 5960, "first_step": 41, "last_step": 6000, "steps_contiguous": true, "finite_losses": true, "finite_gradients": true, "max_gradient_norm": 267.5868225097656, "max_loss": 7.096554353829106, "max_cuda_allocated_bytes": 8786432512, "median_step_seconds": 3.5843077870013076, "last_100_median_step_seconds": 3.610054773500451, "last_update": { "step": 6000, "loss": 0.0019449404207989573, "gradient_norm": 0.06868523359298706, "seconds": 3.64171387499664, "decisions": 4, "branches": 14, "actual_branch_tokens": 1412, "padded_branch_tokens": 1423, "peak_cuda_allocated_bytes": 8786432512, "lr_factor": 0.911059698995469 } }, "correctness": { "branch_chunks_1_probability_max_abs": 4.76837158203125e-07, "branch_chunks_2_probability_max_abs": 0.0, "branch_chunks_4_probability_max_abs": 6.556510925292969e-07, "question_isolation_probability_max_abs": 2.0489096641540527e-08, "candidate_permutation_probability_max_abs": 1.1920928955078125e-07, "repeat_probability_max_abs": 0.0, "tolerance_probability_abs": 0.0001 }, "checkpoint": { "path": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h/training/checkpoint.pt", "modified_utc": "2026-09-17T01:59:13.373975+00:00", "sha256": "ab70dcb62432039d93ec71e200157250dc59c563404237e4dfad3662fd30f44f", "has_optimizer": true, "has_rng": true }, "checkpoint_step": 6000, "best_checkpoint": { "path": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h/training/best.pt", "modified_utc": "2026-09-16T21:45:31.514410+00:00", "sha256": "be08efdc9121a64547b5c9331d5908eeb0978c4fbc962885fd34f9758ec89d11" }, "stale_evaluations": 8, "errors": { "supervisor.log": [], "training.log": [] }, "training_deadline": "2026-09-17T16:00:00+00:00", "final_deadline": "2026-09-17T18:16:10+00:00", "inference": "Eight evaluations since step 2000 failed to improve the fixed crossfit criterion. Resuming unchanged is unsupported by this trend; retain the best artifact for fleet selection. Further experimentation requires a separately recorded branch.", "reserved_data_access": "No reserved predictions read; only train/validation logs, metadata and checkpoint state.", "best_prediction_evidence": { "rows": 512, "best_validation_predictions_sha256": "980c9df18461f5de79e54701bfee1b5fbff19dd3b85125e64f619a88c60f80e4", "validation_step_002000_predictions_sha256": "980c9df18461f5de79e54701bfee1b5fbff19dd3b85125e64f619a88c60f80e4", "equal": true }, "model_availability": { "directory": "/home/andy/ai/models/opensysone", "present": [ "Qwen3.5-2B-15852e8c" ], "required_transfer": "Qwen3-4B-Instruct-2507-cdbee75f", "available_disk_bytes": 3573074882560 }, "suggested_followup": { "status": "recommendation_only_not_launched", "parent": "freeze current best fixed-CV 4B after comparing GX10 and Spark A", "preserve_original_2b_candidate": true, "initialization": "weights_only_with_fresh_Adam_and_RNG", "lr": 1e-05, "head_lr": 1e-05, "seed": 432, "rank": 8, "alpha": 16, "effective_batch": 4, "branch_batch_size": 1, "two_pass": true, "max_tokens": 512, "schedule_steps": 5000, "selection_metric": "crossfit_temperature_nll_v1", "training_deadline": "2026-09-17T16:00:00+00:00", "final_deadline": "2026-09-17T18:16:10+00:00", "expected_new_steps": "Approximately 5000 given 13h40 and measured ~8-9s 4B updates plus verification/validation; stop by existing deadline." } }