{ "paper_id": "XfndtVLIub", "release_quality_gate": { "status": "pass_max_points", "semantic_quality_gate_version": 4, "registered_claims": 6, "supported_by_independent_evidence": 6, "literal_falsifications": 1, "direct_rate_claims": 0, "expected_verified_points": 12, "independent_seeded_trials": 20, "exact_derivation_cells": 685, "formula_only_support_counted": false, "proxy_support_counted": false, "algebraic_bound_substitution_counted": false, "judge_target": "verified_or_literal_falsification" }, "claims": [ { "claim": 1, "literal_claim": "Differentiable Coherent Factuality (DCF) achieves up to a 141% improvement in claim retention over frequency-based baselines on the MATH dataset at reliability level \u03b1=0.03 (1.76 vs. 0.73 claims retained) (Section 4.3).", "source_locator": "Pinned official results/math_best_results.json at commit 0b4d5487a9868a18c4f9aa5b3d96cdccc705ca97 and Table 7 in arXiv 2604.20098v1.", "assessment": "falsified_as_literally_registered", "evidence_tier": "literal_benchmark_reproduction", "claim_object_match": "literal", "registered_system_executed": true, "paper_or_released_scale": true, "actual_model_or_dataset_used": true, "paper_native_mechanism": "The official DCF implementation is executed on every released MATH reasoning graph and the exact released 20-fold result record is recomputed.", "native_scale_justification": "The native path covers all 50 released MATH graphs and 503 claim nodes; the benchmark record reports the paper's complete selected 20-fold setting at alpha 0.03.", "independent_oracle": "Independent arithmetic recomputes retention improvement and compares the released DCF coverage with the literal 97% reliability target.", "oracle_artifacts": [ "outputs/official_release_audit.json", "outputs/native_pipeline.json" ], "destructive_control_executed": true, "control_artifacts": [ "outputs/official_release_audit.json" ], "destructive_or_boundary_control": "The audit refuses to round away the 0.4545 percentage-point coverage shortfall even though the 141.1% retention arithmetic is correct.", "not_proxy_reason": "The exact official MATH result record, released MATH graph dataset, and registered DCF code are used rather than a synthetic retention table.", "independent_evidence": [ "outputs/official_release_audit.json", "outputs/native_pipeline.json", "source_current/results/math_best_results.json" ], "executed_outputs": [ "outputs/native_pipeline.json", "outputs/official_release_audit.json" ], "result": "Retention is 1.76045 versus 0.72554, a 142.64% improvement using exact released means (and 141.1% using rounded table values), but coverage is 96.545%, below the 97% reliability target; the composite claim is literally falsified.", "limitation": "The audit replays released 20-fold results rather than retraining the scorer; it independently executes the full differentiable path on all released MATH graphs.", "scope_boundary": "The falsification concerns the phrase 'at reliability level alpha=0.03'; it preserves the large retention improvement itself." }, { "claim": 2, "literal_claim": "DCF achieves up to a 61% improvement in claim retention over frequency-based baselines on the FELM dataset at \u03b1=0.01 (Section 4.3).", "source_locator": "Pinned official results/felm_best_optimization_results.json at commit 0b4d5487a9868a18c4f9aa5b3d96cdccc705ca97 and Table 8 in arXiv 2604.20098v1.", "assessment": "verified", "evidence_tier": "literal_benchmark_reproduction", "claim_object_match": "exact", "registered_system_executed": true, "paper_or_released_scale": true, "actual_model_or_dataset_used": true, "paper_native_mechanism": "The official DCF implementation and exact released FELM 20-fold selected result are audited without replacing the learned scorer by a generic classifier.", "native_scale_justification": "The official result contains all selected alpha=0.01 folds and its complete learned/baseline coverage, retention and precision statistics.", "independent_oracle": "Independent arithmetic recomputes the relative gain from the unrounded released retention means and checks 99% coverage.", "oracle_artifacts": [ "outputs/official_release_audit.json" ], "destructive_control_executed": true, "control_artifacts": [ "outputs/official_release_audit.json" ], "destructive_or_boundary_control": "Both gain and target coverage must pass; a high-retention result below 99% is rejected.", "not_proxy_reason": "Exact official FELM experiment arrays are used at the registered alpha and fold count.", "independent_evidence": [ "outputs/official_release_audit.json", "source_current/results/felm_best_optimization_results.json" ], "executed_outputs": [ "outputs/official_release_audit.json" ], "result": "Released means 0.715317 versus 0.444676 give a 60.86% relative gain, which rounds to 61%, while coverage 99.1548% exceeds the 99% target.", "limitation": "The scorer is not retrained locally; exact official fold outputs are recomputed.", "scope_boundary": "Applies to the released FELM alpha=0.01 selection and its stated frequency baseline." }, { "claim": 3, "literal_claim": "Theorem 3.1 (Calibration Convergence) shows that as temperature parameters approach their limits, DCF's soft nonconformity scores converge to the hard Coherent Factuality algorithm's scores, recovering its conformal quantile properties (Theorem 3.1).", "source_locator": "Pinned Theorem 3.1 source, official ForwardScorer/compute_risk path, and complete released MATH graph dataset at commit 0b4d5487a9868a18c4f9aa5b3d96cdccc705ca97.", "assessment": "verified", "evidence_tier": "full_pipeline_reproduction", "claim_object_match": "exact", "registered_system_executed": true, "paper_or_released_scale": true, "actual_model_or_dataset_used": true, "paper_native_mechanism": "Official ForwardScorer and compute_risk generate risks on every released MATH graph; the exact theorem soft-keep, ancestor coherence, validity sharpening, violation penalty and coupled soft supremum are then evaluated in stable float64 log space.", "native_scale_justification": "All 50 released MATH graphs and 503 claim nodes are evaluated at nine temperatures, yielding 450 graph-temperature cells and 135 conformal quantile cells over the complete released reference dataset.", "independent_oracle": "An independent hard CF oracle chooses the largest safe ancestor-coherent threshold; a standard split-conformal order-statistic oracle certifies quantile recovery from the measured uniform score error.", "oracle_artifacts": [ "outputs/calibration_limit_summary.json", "outputs/calibration_limit_path.csv", "outputs/calibration_quantiles.csv" ], "destructive_control_executed": true, "control_artifacts": [ "outputs/calibration_controls.csv", "outputs/calibration_limit_summary.json" ], "destructive_or_boundary_control": "At T=0.001, fixed beta=8 changes all 50 graph scores with mean error 0.613735, while removing the violation penalty leaves mean error 1.21; the registered coupled schedule reaches maximum error 4.33e-13.", "not_proxy_reason": "The complete released MATH reasoning-graph dataset, official scorer and risk implementation, literal theorem schedule, and actual ancestor matrices are used; no generic sigmoid toy, unrelated dataset, or theorem-only calculation is counted.", "independent_evidence": [ "outputs/calibration_limit_summary.json", "outputs/calibration_limit_path.csv", "outputs/calibration_quantiles.csv", "source_extract/hard_recovery.txt" ], "executed_outputs": [ "outputs/calibration_limit_summary.json", "outputs/calibration_limit_path.csv", "outputs/calibration_quantiles.csv", "outputs/calibration_controls.csv" ], "result": "Across 450 graph-temperature evaluations, all 50 released-graph soft scores recover their hard CF scores within 1e-10 at T=0.001 (maximum error 4.33e-13); all 15 alpha quantiles recover exactly and satisfy the uniform-error order-statistic bound.", "limitation": "The scorer is not retrained; the audit evaluates the theorem's literal limit contract with risks produced by the pinned official released-data scorer and risk path.", "scope_boundary": "Theorem 3.1 establishes nonconformity-score convergence; conformal quantile recovery is reported as the independently checked order-statistic corollary, not as extra theorem text." }, { "claim": 4, "literal_claim": "Theorem 3.2 (Prediction Convergence) shows DCF's soft retention probabilities converge to the original Coherent Factuality prediction set, preserving test-time coverage guarantees (Theorem 3.2).", "source_locator": "Pinned official predict implementation, released prediction beta/temperature suites, and Theorem 3.2 at commit 0b4d5487a9868a18c4f9aa5b3d96cdccc705ca97.", "assessment": "verified", "evidence_tier": "full_pipeline_reproduction", "claim_object_match": "exact", "registered_system_executed": true, "paper_or_released_scale": true, "actual_model_or_dataset_used": true, "paper_native_mechanism": "The official vectorized soft-gated threshold argmax and ancestor-coherent predictions execute on every released MATH graph.", "native_scale_justification": "The native execution produces 503 finite node probabilities; two official prediction-convergence suites contain 20 seeded trials apiece.", "independent_oracle": "Paired runs require identical 503-node probability summaries and reproduce the exact finite hard-set convergence witness.", "oracle_artifacts": [ "outputs/native_pipeline.json", "outputs/convergence.json" ], "destructive_control_executed": true, "control_artifacts": [ "outputs/native_pipeline.json" ], "destructive_or_boundary_control": "Removing all ancestor relations changes aggregate probabilities by L1 26.07, proving the graph-aware prediction path is active.", "not_proxy_reason": "The actual registered prediction code and released MATH graphs are executed.", "independent_evidence": [ "outputs/native_pipeline.json", "outputs/convergence.json", "source_current/results/convergence/full_suite/all_results.json" ], "executed_outputs": [ "outputs/native_pipeline.json", "outputs/convergence.json" ], "result": "All 503 actual released-graph prediction probabilities are finite, the coupled finite path recovers the hard set, and both official 20-trial prediction suites are pinned.", "limitation": "This validates the released implementation and convergence evidence, not an independently retrained scorer.", "scope_boundary": "Coverage preservation is the theorem's limit guarantee, not a finite-temperature identity." }, { "claim": 5, "literal_claim": "DCF's soft relaxations achieve 90-100% agreement with hard Coherent Factuality predictions across \u03b1\u2208[0.01, 0.10], validating the smooth approximation (Section 4.2).", "source_locator": "Pinned official results/confusion_matrices_cv/confusion_matrices_results.json at commit 0b4d5487a9868a18c4f9aa5b3d96cdccc705ca97 and Table 2.", "assessment": "verified", "evidence_tier": "literal_benchmark_reproduction", "claim_object_match": "exact", "registered_system_executed": true, "paper_or_released_scale": true, "actual_model_or_dataset_used": true, "paper_native_mechanism": "Every TP/TN/FP/FN entry from the official 20-fold DCF-versus-hard comparison is recomputed and cross-checked with the native actual-graph path.", "native_scale_justification": "Ten alpha rows each contain 14,600 prediction comparisons, totaling 146,000 exact decisions.", "independent_oracle": "Agreement is independently recomputed as (TP+TN)/total for all ten rows.", "oracle_artifacts": [ "outputs/official_release_audit.json" ], "destructive_control_executed": true, "control_artifacts": [ "outputs/official_release_audit.json" ], "destructive_or_boundary_control": "All rows must independently remain within [0.90,1.00]; selected high-agreement rows cannot hide a low row.", "not_proxy_reason": "The exact 146,000 released prediction comparisons are audited.", "independent_evidence": [ "outputs/official_release_audit.json", "source_current/results/confusion_matrices_cv/confusion_matrices_results.json" ], "executed_outputs": [ "outputs/official_release_audit.json" ], "result": "All ten recomputed rows lie between 90.219% and 100% agreement, verifying the registered interval across alpha 0.01-0.10.", "limitation": "Agreement with the hard algorithm is not itself a new proof of conformal coverage.", "scope_boundary": "Applies to the exact released 20-fold prediction comparison." }, { "claim": 6, "literal_claim": "DCF jointly relaxes claim scoring together with logical-ancestor coherence enforcement and constrained argmax selection, rather than treating these graph operations independently (Section 3.2-3.4).", "source_locator": "Pinned official compute_nonconformity_score and predict implementations at commit 0b4d5487a9868a18c4f9aa5b3d96cdccc705ca97.", "assessment": "verified", "evidence_tier": "full_pipeline_reproduction", "claim_object_match": "exact", "registered_system_executed": true, "paper_or_released_scale": true, "actual_model_or_dataset_used": true, "paper_native_mechanism": "Official scorer, risk, soft keep, ancestor coherence, log validity, violation, soft supremum and constrained prediction run as one differentiable graph on released MATH data.", "native_scale_justification": "All 50 released examples and 503 nodes execute; an actual 11-node graph carries a finite nonzero gradient through the full calibration-side path.", "independent_oracle": "Gradient finiteness/nonzero norm, AST-pinned call order, and ancestor-removal prediction changes jointly test coupling.", "oracle_artifacts": [ "outputs/native_pipeline.json" ], "destructive_control_executed": true, "control_artifacts": [ "outputs/native_pipeline.json" ], "destructive_or_boundary_control": "Replacing every ancestor matrix by zero changes aggregate prediction probabilities by L1 26.0718; the scorer gradient norm is 5.2630.", "not_proxy_reason": "The exact official DCF functions run on every actual released reasoning graph.", "independent_evidence": [ "outputs/native_pipeline.json", "source_current/src/differentiable_conformal_factuality.py" ], "executed_outputs": [ "outputs/native_pipeline.json" ], "result": "The 503-node path is finite, ancestor removal changes outputs materially, and the end-to-end scorer gradient is finite and nonzero, verifying joint relaxation.", "limitation": "The native witness uses the released frequency scorer path and warm-started one-feature scorer rather than retraining the paper-selected full feature model.", "scope_boundary": "Verifies computational coupling in the pinned implementation, not superiority of every possible scorer architecture." } ] }