{ "current_repair_accuracy_evaluated": false, "current_repair_note": "Historical counts are preserved exactly; native reliability qualification does not establish a new score.", "current_repaired_image_config_digest": "sha256:94d0791cb96f3ac9248e9ec7c918e3bbdc5960646b48d201cb478d488fcda576", "evaluated_image_config_digest": "sha256:8d3bbd92c7ef8f73ff058382bef1b8523ec55961b831422c1fb04226b31b35d5", "evaluation_gold_or_rows_included": false, "fresh_native_protocol": { "all_primary_outcomes_retained": 2174, "external_process_pool": "120 + 30 * B seconds", "failure_causes": { "pfull_best_platform": { "original_process_pool_timeout_rows": 124 }, "selected615": { "stream_cap_rows": 28, "validation_demotion_group_rows": 39 } }, "profiles": { "pfull_best_platform": { "five_bucket_mean_exact": { "denominator": 3272160, "float": 0.46532229475331277, "numerator": 1522609 }, "n": 1087, "n_correct": 444, "outcomes": { "answered": 963, "wrapper_failed": 124 }, "per_bucket_correct": { "aggregation": 39, "complex_reasoning": 13, "event_understanding": 10, "object_recognition": 267, "temporal_grounding": 115 } }, "selected615": { "five_bucket_mean_exact": { "denominator": 654432, "float": 0.5593201432692778, "numerator": 366037 }, "n": 1087, "n_correct": 579, "outcomes": { "answered": 1020, "wrapper_failed": 67 }, "per_bucket_correct": { "aggregation": 53, "complex_reasoning": 16, "event_understanding": 10, "object_recognition": 399, "temporal_grounding": 101 } } }, "resources": { "cpus": 16, "gpu": "H100", "host_memory_mib": 196608 } }, "historical_local_protocol": { "five_bucket_mean_exact": { "denominator": 204510, "float": 0.6191042002836047, "numerator": 126613 }, "n": 1087, "n_correct": 615, "outcomes": { "answered": 1087 }, "per_bucket_correct": { "aggregation": 56, "complex_reasoning": 18, "event_understanding": 12, "object_recognition": 416, "temporal_grounding": 113 } }, "independent_native_scoring_review": { "bytes": 13528, "sha256": "fb18091701d6dcbfc55abd1d646cd68ff29c9fad3b32a8273d2b49722e216515" }, "joint_qualified_sensitivity": { "comparator_correct": 421, "n": 910, "replaces_primary_population": false, "selected_correct": 511 }, "limits": [ "Local selected development population; no untouched or challenge OOD generalization claim.", "One clinical-flagged row cannot establish clinical quality.", "Correlated question/video observations and repeated model selection limit rowwise comparisons.", "Historical615 and fresh native579 are different protocols." ], "official_challenge_result": false, "population": { "challenge_ood_rows": 0, "clinical_flagged_rows": 1, "questions": 1087, "used_during_development_and_selection": true, "videos": 27 }, "schema": "procedure-final-local-evaluation-summary/1" }