{ "created_utc": "2026-09-17T02:19:29.647198+00:00", "source_commit": "b1ce82b1396cd6ef050c9904a4aa8453befc0ced", "selection_source_sha256": "be0a7a8496b5b830aa572ceba93606320f442063fd38180503fd6980dc1c578f", "scope": "Validation-only fixed-ensemble diagnostic; no weight tuning; no reserved calibration/test/holdout data or predictions accessed", "protocol_status": "Active fleet selects individual checkpoints only. Ensembles are not registered, deployed, or promised. Any future adoption must freeze its rule before reserved-data evaluation.", "validation_identity": "Exact same512 IDs, source groups, families, targets and ordered choices verified with scripts.fleet_campaign.identity", "inputs": { "gx10_4b_step2500": { "host": "local", "source_path": "/home/andy/ai/opensysone/runs/20260916T193741Z-24h/training/validation_step_002500_predictions.json", "sha256": "16ff7745cdbfa53b87ccf2445796cf6a0c28d148a00e6c3926c72c36e75d058e", "copied_path": "/home/andy/ai/opensysone/fleet-20260917/gx10_4b_step2500-validation-predictions.json", "rows": 512 }, "spark_a_4b_step2500": { "host": "andy@192.168.8.111", "source_path": "/home/andy/ai/opensysone/runs/20260916T194258Z-24h/training/validation_step_002500_predictions.json", "sha256": "9d0676e4f318f0e419d4e73ce6b73afdc21a6252f751d8d88abcdd644b6ca979", "copied_path": "/home/andy/ai/opensysone/fleet-20260917/spark_a_4b_step2500-validation-predictions.json", "rows": 512 }, "spark_b_2b_step2000": { "host": "andy@192.168.8.204", "source_path": "/home/andy/ai/opensysone/runs/20260916T193803Z-24h/training/validation_step_002000_predictions.json", "sha256": "980c9df18461f5de79e54701bfee1b5fbff19dd3b85125e64f619a88c60f80e4", "copied_path": "/home/andy/ai/opensysone/fleet-20260917/spark_b_2b_step2000-validation-predictions.json", "rows": 512 } }, "individuals": { "gx10_4b_step2500": { "metric": "crossfit_temperature_nll_v1", "score": 0.18864030037724333, "raw_macro_nll": 0.30729043330681044, "accuracy": 0.935546875, "per_family": { "arc": { "count": 128, "score": 0.14837152412562352, "raw_macro_nll": 0.10747530600586384, "accuracy": 0.953125 }, "banking": { "count": 128, "score": 0.06337688481028368, "raw_macro_nll": 0.09334830709159572, "accuracy": 0.9765625 }, "boolq": { "count": 128, "score": 0.3119620718470344, "raw_macro_nll": 0.6294207016878283, "accuracy": 0.890625 }, "snli": { "count": 128, "score": 0.23085072072603174, "raw_macro_nll": 0.3989174184419539, "accuracy": 0.921875 } }, "fold_temperatures": [ 2.39883291901949, 2.5292979964461435, 2.666858664521479, 2.666858664521479 ] }, "spark_a_4b_step2500": { "metric": "crossfit_temperature_nll_v1", "score": 0.21801211708868656, "raw_macro_nll": 0.3004267213630067, "accuracy": 0.92578125, "per_family": { "arc": { "count": 128, "score": 0.19928542525398854, "raw_macro_nll": 0.18728708915519038, "accuracy": 0.9453125 }, "banking": { "count": 128, "score": 0.08454689397045242, "raw_macro_nll": 0.09826600992870173, "accuracy": 0.96875 }, "boolq": { "count": 128, "score": 0.3652053358471964, "raw_macro_nll": 0.6254968111183671, "accuracy": 0.875 }, "snli": { "count": 128, "score": 0.22301081328310884, "raw_macro_nll": 0.29065697524976764, "accuracy": 0.9140625 } }, "fold_temperatures": [ 2.046444636724674, 1.9408858775927773, 2.1577444091526656, 2.2750974307720706 ] }, "spark_b_2b_step2000": { "metric": "crossfit_temperature_nll_v1", "score": 0.2552943737691716, "raw_macro_nll": 0.25676974955497256, "accuracy": 0.8984375, "per_family": { "arc": { "count": 128, "score": 0.3379051798328397, "raw_macro_nll": 0.33955257651665527, "accuracy": 0.890625 }, "banking": { "count": 128, "score": 0.06793271948328157, "raw_macro_nll": 0.06537141992655102, "accuracy": 0.9609375 }, "boolq": { "count": 128, "score": 0.33084980282335186, "raw_macro_nll": 0.34265219288051024, "accuracy": 0.84375 }, "snli": { "count": 128, "score": 0.28448979293721316, "raw_macro_nll": 0.2795028088961736, "accuracy": 0.8984375 } }, "fold_temperatures": [ 1.2705741052085413, 1.2050359403717974, 1.0839269140212033, 1.2050359403717974 ] } }, "ensembles": { "gx10_plus_spark_a_equal_logits": { "members": [ "gx10_4b_step2500", "spark_a_4b_step2500" ], "fixed_weights": [ 0.5, 0.5 ], "combination": "arithmetic mean of raw logits, no per-member temperature or normalization", "metrics": { "metric": "crossfit_temperature_nll_v1", "score": 0.19398305226507087, "raw_macro_nll": 0.2877537539614745, "accuracy": 0.931640625, "per_family": { "arc": { "count": 128, "score": 0.16499940668045984, "raw_macro_nll": 0.1397725827944651, "accuracy": 0.953125 }, "banking": { "count": 128, "score": 0.06593766264787593, "raw_macro_nll": 0.08660166528079918, "accuracy": 0.96875 }, "boolq": { "count": 128, "score": 0.3215770523678777, "raw_macro_nll": 0.5933697343172437, "accuracy": 0.890625 }, "snli": { "count": 128, "score": 0.22341808736407, "raw_macro_nll": 0.33127103345339015, "accuracy": 0.9140625 } }, "fold_temperatures": [ 2.2750974307720706, 2.1577444091526656, 2.39883291901949, 2.5292979964461435 ] }, "member_error_overlap": { "both_correct": 470, "only_first_correct": 9, "only_second_correct": 4, "both_wrong": 29, "argmax_agreement": 0.970703125 }, "paired_vs_members": { "gx10_4b_step2500": { "gained_correct": 1, "lost_correct": 3, "accuracy_delta": -0.00390625, "mcnemar_exact_two_sided_p": 0.625 }, "spark_a_4b_step2500": { "gained_correct": 6, "lost_correct": 3, "accuracy_delta": 0.005859375, "mcnemar_exact_two_sided_p": 0.5078125 } } }, "spark_a_plus_spark_b_equal_logits": { "members": [ "spark_a_4b_step2500", "spark_b_2b_step2000" ], "fixed_weights": [ 0.5, 0.5 ], "combination": "arithmetic mean of raw logits, no per-member temperature or normalization", "metrics": { "metric": "crossfit_temperature_nll_v1", "score": 0.1805601057147578, "raw_macro_nll": 0.19435914397726234, "accuracy": 0.935546875, "per_family": { "arc": { "count": 128, "score": 0.16581505416388923, "raw_macro_nll": 0.1533871126809031, "accuracy": 0.9609375 }, "banking": { "count": 128, "score": 0.05441042849904982, "raw_macro_nll": 0.0512741354347431, "accuracy": 0.984375 }, "boolq": { "count": 128, "score": 0.31417451798452084, "raw_macro_nll": 0.37796216743844757, "accuracy": 0.8828125 }, "snli": { "count": 128, "score": 0.18784042221157138, "raw_macro_nll": 0.19481316035495563, "accuracy": 0.9140625 } }, "fold_temperatures": [ 1.412537544622754, 1.3396766874259345, 1.412537544622754, 1.4893610777109154 ] }, "member_error_overlap": { "both_correct": 441, "only_first_correct": 33, "only_second_correct": 19, "both_wrong": 19, "argmax_agreement": 0.892578125 }, "paired_vs_members": { "spark_a_4b_step2500": { "gained_correct": 9, "lost_correct": 4, "accuracy_delta": 0.009765625, "mcnemar_exact_two_sided_p": 0.266845703125 }, "spark_b_2b_step2000": { "gained_correct": 29, "lost_correct": 10, "accuracy_delta": 0.037109375, "mcnemar_exact_two_sided_p": 0.0033778479119064286 } }, "paired_vs_best_individual": { "comparison": "Fixed equal raw-logit mean of Spark A4Bstep2500 and Spark B2Bstep2000 minus GX104Bstep2500", "method": "1000 paired source-group bootstrap resamples stratified by family; groups remain in their original fixed four folds; independently refit each model/ensemble fold temperature on resampled other-fold decisions using macro-family NLL; same resample used for both predictors", "replicates": 1000, "seed": 431, "groups": 512, "source_groups_each_have_one_decision": true, "crossfit_nll_delta": -0.008080194662485524, "crossfit_nll_delta_ci95": [ -0.039063572735885795, 0.019451132921767013 ], "accuracy_delta": 0.0, "accuracy_delta_ci95": [ -0.015625, 0.015625 ], "gained_correct": 8, "lost_correct": 8, "mcnemar_exact_two_sided_p": 1, "elapsed_cpu_analysis_seconds": 0.16473380899697077, "interpretation": "The NLL interval includes zero; this diagnostic does not establish improvement over the best individual checkpoint. Accuracy ties overall. All evidence is on previously used validation rows and remains conditional on prior checkpoint selection; no multiplicity adjustment." } } }, "limitations": [ "These are previously selected checkpoints evaluated on their selection validation rows, not independent test evidence.", "Averaging raw logits weights models equally in logit units; differing confidence scales still affect their influence. No scaling or ensemble weight was tuned.", "No inference latency or distributed-serving throughput was measured; two full model replicas and response aggregation would require implementation and verification." ], "method_clarification": "For every decision/candidate, form z_ensemble=(z_A+z_B)/2 from saved RAW logits. Fit one ensemble temperature separately in each fixed validation fold using only the other three folds, then compute held-out crossfit NLL. This is not averaging probabilities and does not fit member-specific scales or ensemble weights.", "uncertainty_updated_utc": "2026-09-17T02:20:51.275883+00:00" }