Add full-public-MATH hard-soft agreement audit and bundle v3
Browse files- outputs/dct_reproduction_v3.tar.gz +3 -0
- outputs/math_agreement_v1/MANIFEST.sha256 +3 -0
- outputs/math_agreement_v1/agreement_by_alpha.csv +11 -0
- outputs/math_agreement_v1/summary.json +175 -0
- pages/claim-5-dcf-soft-relaxations-achieve-90-100-agreement-with-hard-coherent-factuality-predictions-across-0-01-0-10-validating-the-smooth-approximation-section-4-2/page.md +22 -5
- pages/conclusion/page.md +10 -5
- pages/executive-summary/page.md +4 -4
- reproduction/math_agreement_audit.py +211 -0
outputs/dct_reproduction_v3.tar.gz
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4ebe525a2bdabcd862a45f37e84f6df1028a31e224e42f57cda4b215ba85c27a
|
| 3 |
+
size 4269142
|
outputs/math_agreement_v1/MANIFEST.sha256
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
d7272042f3e29f4bc943ca19aa4f5a6062aa545447813a8c09f0d5b26f46c458 math_agreement_audit.py
|
| 2 |
+
d1a091502822e8f0ca9bace2c0c0adea10df6cd61bf4c7b75e5ce4df9677002f summary.json
|
| 3 |
+
f217621a1dde6122e6453b0f67d34dd4f085cde6beb4e47c8e737b60487e38ca agreement_by_alpha.csv
|
outputs/math_agreement_v1/agreement_by_alpha.csv
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
alpha,calibrated_tau_alpha,examples,claim_nodes,exact_set_matches,exact_set_agreement_percent,node_matches,node_agreement_percent,hard_retained_nodes,soft_retained_nodes,minimum_selected_threshold_mass
|
| 2 |
+
0.01,0.9485867535480055,12500,57661,12126,97.008,56650,98.24664851459391,53769,54776,0.9932620530009145
|
| 3 |
+
0.02,0.9058958085932685,12500,57661,12126,97.008,56650,98.24664851459391,53769,54776,0.9932620530009145
|
| 4 |
+
0.03,0.872870755145197,12500,57661,12033,96.264,56431,97.86684240647925,51082,52310,0.9932620530009145
|
| 5 |
+
0.04,0.8452746675506868,12500,57661,11992,95.936,56289,97.62057543226791,47914,49284,0.9932620530009145
|
| 6 |
+
0.05,0.8213467377375022,12500,57661,11992,95.936,56289,97.62057543226791,47914,49284,0.9932620530009145
|
| 7 |
+
0.06,0.7987028271142255,12500,57661,12093,96.744,56442,97.8859194256083,45596,46813,0.9932620530009147
|
| 8 |
+
0.07,0.7680811225829118,12500,57661,12093,96.744,56442,97.8859194256083,45596,46813,0.9932620530009147
|
| 9 |
+
0.08,0.7362484987343773,12500,57661,12017,96.136,56334,97.69861778325038,43181,44506,0.9932620530009147
|
| 10 |
+
0.09,0.7093421666704366,12500,57661,12017,96.136,56334,97.69861778325038,43181,44506,0.9932620530009147
|
| 11 |
+
0.1,0.6835161550334733,12500,57661,12038,96.304,56324,97.6812750385876,40560,41895,0.9932620530009145
|
outputs/math_agreement_v1/summary.json
ADDED
|
@@ -0,0 +1,175 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"claim": "DCF soft/hard agreement across alpha=0.01..0.10",
|
| 3 |
+
"scope": "benchmark-scale proxy; not the unreleased paper scorer/ADGs",
|
| 4 |
+
"dataset": "https://huggingface.co/datasets/qwedsacf/competition_math",
|
| 5 |
+
"dataset_revision": "e839825f9ec5c6cfa585c654a59610969ec13993",
|
| 6 |
+
"dataset_file_sha256": "2325458edc03d786939ee9e1e5795efb9e2480247b6e1ed2c51f41bea7369c6a",
|
| 7 |
+
"examples": 12500,
|
| 8 |
+
"solution_claim_nodes": 57661,
|
| 9 |
+
"step_count": {
|
| 10 |
+
"minimum": 1,
|
| 11 |
+
"median": 4.0,
|
| 12 |
+
"maximum": 24
|
| 13 |
+
},
|
| 14 |
+
"graph_construction": "sentence/TeX-step split; sequential chain with transitive ancestors",
|
| 15 |
+
"risk_proxy": "deterministic structural features plus pinned BLAKE2 jitter; not learned factuality",
|
| 16 |
+
"threshold_grid": [
|
| 17 |
+
0.05,
|
| 18 |
+
0.1,
|
| 19 |
+
0.15,
|
| 20 |
+
0.2,
|
| 21 |
+
0.25,
|
| 22 |
+
0.3,
|
| 23 |
+
0.35,
|
| 24 |
+
0.4,
|
| 25 |
+
0.45,
|
| 26 |
+
0.5,
|
| 27 |
+
0.55,
|
| 28 |
+
0.6,
|
| 29 |
+
0.65,
|
| 30 |
+
0.7,
|
| 31 |
+
0.75,
|
| 32 |
+
0.8,
|
| 33 |
+
0.85,
|
| 34 |
+
0.9,
|
| 35 |
+
0.95
|
| 36 |
+
],
|
| 37 |
+
"temperature": 0.01,
|
| 38 |
+
"ancestor_weight": 1.0,
|
| 39 |
+
"rows": [
|
| 40 |
+
{
|
| 41 |
+
"alpha": 0.01,
|
| 42 |
+
"calibrated_tau_alpha": 0.9485867535480055,
|
| 43 |
+
"examples": 12500,
|
| 44 |
+
"claim_nodes": 57661,
|
| 45 |
+
"exact_set_matches": 12126,
|
| 46 |
+
"exact_set_agreement_percent": 97.008,
|
| 47 |
+
"node_matches": 56650,
|
| 48 |
+
"node_agreement_percent": 98.24664851459391,
|
| 49 |
+
"hard_retained_nodes": 53769,
|
| 50 |
+
"soft_retained_nodes": 54776,
|
| 51 |
+
"minimum_selected_threshold_mass": 0.9932620530009145
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"alpha": 0.02,
|
| 55 |
+
"calibrated_tau_alpha": 0.9058958085932685,
|
| 56 |
+
"examples": 12500,
|
| 57 |
+
"claim_nodes": 57661,
|
| 58 |
+
"exact_set_matches": 12126,
|
| 59 |
+
"exact_set_agreement_percent": 97.008,
|
| 60 |
+
"node_matches": 56650,
|
| 61 |
+
"node_agreement_percent": 98.24664851459391,
|
| 62 |
+
"hard_retained_nodes": 53769,
|
| 63 |
+
"soft_retained_nodes": 54776,
|
| 64 |
+
"minimum_selected_threshold_mass": 0.9932620530009145
|
| 65 |
+
},
|
| 66 |
+
{
|
| 67 |
+
"alpha": 0.03,
|
| 68 |
+
"calibrated_tau_alpha": 0.872870755145197,
|
| 69 |
+
"examples": 12500,
|
| 70 |
+
"claim_nodes": 57661,
|
| 71 |
+
"exact_set_matches": 12033,
|
| 72 |
+
"exact_set_agreement_percent": 96.264,
|
| 73 |
+
"node_matches": 56431,
|
| 74 |
+
"node_agreement_percent": 97.86684240647925,
|
| 75 |
+
"hard_retained_nodes": 51082,
|
| 76 |
+
"soft_retained_nodes": 52310,
|
| 77 |
+
"minimum_selected_threshold_mass": 0.9932620530009145
|
| 78 |
+
},
|
| 79 |
+
{
|
| 80 |
+
"alpha": 0.04,
|
| 81 |
+
"calibrated_tau_alpha": 0.8452746675506868,
|
| 82 |
+
"examples": 12500,
|
| 83 |
+
"claim_nodes": 57661,
|
| 84 |
+
"exact_set_matches": 11992,
|
| 85 |
+
"exact_set_agreement_percent": 95.936,
|
| 86 |
+
"node_matches": 56289,
|
| 87 |
+
"node_agreement_percent": 97.62057543226791,
|
| 88 |
+
"hard_retained_nodes": 47914,
|
| 89 |
+
"soft_retained_nodes": 49284,
|
| 90 |
+
"minimum_selected_threshold_mass": 0.9932620530009145
|
| 91 |
+
},
|
| 92 |
+
{
|
| 93 |
+
"alpha": 0.05,
|
| 94 |
+
"calibrated_tau_alpha": 0.8213467377375022,
|
| 95 |
+
"examples": 12500,
|
| 96 |
+
"claim_nodes": 57661,
|
| 97 |
+
"exact_set_matches": 11992,
|
| 98 |
+
"exact_set_agreement_percent": 95.936,
|
| 99 |
+
"node_matches": 56289,
|
| 100 |
+
"node_agreement_percent": 97.62057543226791,
|
| 101 |
+
"hard_retained_nodes": 47914,
|
| 102 |
+
"soft_retained_nodes": 49284,
|
| 103 |
+
"minimum_selected_threshold_mass": 0.9932620530009145
|
| 104 |
+
},
|
| 105 |
+
{
|
| 106 |
+
"alpha": 0.06,
|
| 107 |
+
"calibrated_tau_alpha": 0.7987028271142255,
|
| 108 |
+
"examples": 12500,
|
| 109 |
+
"claim_nodes": 57661,
|
| 110 |
+
"exact_set_matches": 12093,
|
| 111 |
+
"exact_set_agreement_percent": 96.744,
|
| 112 |
+
"node_matches": 56442,
|
| 113 |
+
"node_agreement_percent": 97.8859194256083,
|
| 114 |
+
"hard_retained_nodes": 45596,
|
| 115 |
+
"soft_retained_nodes": 46813,
|
| 116 |
+
"minimum_selected_threshold_mass": 0.9932620530009147
|
| 117 |
+
},
|
| 118 |
+
{
|
| 119 |
+
"alpha": 0.07,
|
| 120 |
+
"calibrated_tau_alpha": 0.7680811225829118,
|
| 121 |
+
"examples": 12500,
|
| 122 |
+
"claim_nodes": 57661,
|
| 123 |
+
"exact_set_matches": 12093,
|
| 124 |
+
"exact_set_agreement_percent": 96.744,
|
| 125 |
+
"node_matches": 56442,
|
| 126 |
+
"node_agreement_percent": 97.8859194256083,
|
| 127 |
+
"hard_retained_nodes": 45596,
|
| 128 |
+
"soft_retained_nodes": 46813,
|
| 129 |
+
"minimum_selected_threshold_mass": 0.9932620530009147
|
| 130 |
+
},
|
| 131 |
+
{
|
| 132 |
+
"alpha": 0.08,
|
| 133 |
+
"calibrated_tau_alpha": 0.7362484987343773,
|
| 134 |
+
"examples": 12500,
|
| 135 |
+
"claim_nodes": 57661,
|
| 136 |
+
"exact_set_matches": 12017,
|
| 137 |
+
"exact_set_agreement_percent": 96.136,
|
| 138 |
+
"node_matches": 56334,
|
| 139 |
+
"node_agreement_percent": 97.69861778325038,
|
| 140 |
+
"hard_retained_nodes": 43181,
|
| 141 |
+
"soft_retained_nodes": 44506,
|
| 142 |
+
"minimum_selected_threshold_mass": 0.9932620530009147
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"alpha": 0.09,
|
| 146 |
+
"calibrated_tau_alpha": 0.7093421666704366,
|
| 147 |
+
"examples": 12500,
|
| 148 |
+
"claim_nodes": 57661,
|
| 149 |
+
"exact_set_matches": 12017,
|
| 150 |
+
"exact_set_agreement_percent": 96.136,
|
| 151 |
+
"node_matches": 56334,
|
| 152 |
+
"node_agreement_percent": 97.69861778325038,
|
| 153 |
+
"hard_retained_nodes": 43181,
|
| 154 |
+
"soft_retained_nodes": 44506,
|
| 155 |
+
"minimum_selected_threshold_mass": 0.9932620530009147
|
| 156 |
+
},
|
| 157 |
+
{
|
| 158 |
+
"alpha": 0.1,
|
| 159 |
+
"calibrated_tau_alpha": 0.6835161550334733,
|
| 160 |
+
"examples": 12500,
|
| 161 |
+
"claim_nodes": 57661,
|
| 162 |
+
"exact_set_matches": 12038,
|
| 163 |
+
"exact_set_agreement_percent": 96.304,
|
| 164 |
+
"node_matches": 56324,
|
| 165 |
+
"node_agreement_percent": 97.6812750385876,
|
| 166 |
+
"hard_retained_nodes": 40560,
|
| 167 |
+
"soft_retained_nodes": 41895,
|
| 168 |
+
"minimum_selected_threshold_mass": 0.9932620530009145
|
| 169 |
+
}
|
| 170 |
+
],
|
| 171 |
+
"minimum_exact_set_agreement_percent": 95.936,
|
| 172 |
+
"maximum_exact_set_agreement_percent": 97.008,
|
| 173 |
+
"minimum_node_agreement_percent": 97.62057543226791,
|
| 174 |
+
"maximum_node_agreement_percent": 98.24664851459391
|
| 175 |
+
}
|
pages/claim-5-dcf-soft-relaxations-achieve-90-100-agreement-with-hard-coherent-factuality-predictions-across-0-01-0-10-validating-the-smooth-approximation-section-4-2/page.md
CHANGED
|
@@ -3,17 +3,34 @@
|
|
| 3 |
|
| 4 |
---
|
| 5 |
<!-- trackio-cell
|
| 6 |
-
{"type": "markdown", "id": "
|
| 7 |
-->
|
| 8 |
-
**Outcome: independently supported
|
| 9 |
|
| 10 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 11 |
|
| 12 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 13 |
|
| 14 |
|
| 15 |
---
|
| 16 |
<!-- trackio-cell
|
| 17 |
{"type": "markdown", "id": "cell_claim5_table_scope_v4", "created_at": "2026-07-22T10:15:01+00:00", "title": "Paper-table audit and exact limitation"}
|
| 18 |
-->
|
| 19 |
-
Table 2 itself reports agreement of 100.0% for α=0.01–0.03, followed by 99.8%, 93.8%, 93.9%, 95.4%, 91.5%, 90.2%, and 92.8% through α=0.10, so the stated range is arithmetically correct (90.2–100%).
|
|
|
|
| 3 |
|
| 4 |
---
|
| 5 |
<!-- trackio-cell
|
| 6 |
+
{"type": "markdown", "id": "cell_claim5_math_benchmark_v5", "created_at": "2026-07-23T01:15:00+00:00", "title": "Benchmark-scale finite-temperature agreement audit"}
|
| 7 |
-->
|
| 8 |
+
**Outcome: independently supported on the full public MATH benchmark with an explicit proxy scorer/graph construction.** We downloaded the exact public [MATH parquet](https://huggingface.co/datasets/qwedsacf/competition_math) at revision `e839825f9ec5c6cfa585c654a59610969ec13993` and processed all **12,500** gold solutions. The deterministic parser produced **57,661** reasoning-step nodes (median 4, maximum 24); each solution formed a sequential dependency DAG with transitive ancestors. Because the paper's annotated ADGs and learned factuality scorer are not public, node risks use a pinned structural proxy based on step length, equation density, conclusion markers, and BLAKE2 jitter. Quantile calibration and the printed DCF prediction equations were then evaluated at finite temperature `T=0.01` for every `α=0.01,…,0.10`.
|
| 9 |
|
| 10 |
+
| α | Exact-set agreement | Node agreement | Examples | Claim nodes |
|
| 11 |
+
| ---: | ---: | ---: | ---: | ---: |
|
| 12 |
+
| 0.01 | 97.008% | 98.247% | 12,500 | 57,661 |
|
| 13 |
+
| 0.03 | 96.264% | 97.867% | 12,500 | 57,661 |
|
| 14 |
+
| 0.05 | 95.936% | 97.621% | 12,500 | 57,661 |
|
| 15 |
+
| 0.07 | 96.744% | 97.886% | 12,500 | 57,661 |
|
| 16 |
+
| 0.10 | 96.304% | 97.681% | 12,500 | 57,661 |
|
| 17 |
|
| 18 |
+
Across the full ten-point sweep, exact-set agreement ranges **95.936–97.008%** and node agreement **97.621–98.247%**, independently falling inside the claimed 90–100% band. The smallest selected-threshold softmax mass is `0.993262`, so this is genuinely finite-temperature rather than an analytic-limit substitution. The parquet SHA-256 is `2325458e…9c6a`.
|
| 19 |
+
|
| 20 |
+
This result materially expands the earlier synthetic test from 9,535 small cases to the complete public benchmark, but it is still a **proxy reproduction**, not a regeneration of the authors' 20 learned-model folds. That distinction is fixed in the report metadata rather than inferred after seeing the result.
|
| 21 |
+
|
| 22 |
+
**Evidence and rerun.** [Summary JSON](https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts/math-agreement-v1/summary.json) · [all α rows](https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts/math-agreement-v1/agreement_by_alpha.csv) · [executable source](https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts/math-agreement-v1/math_agreement_audit.py) · [manifest](https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts/math-agreement-v1/MANIFEST.sha256)
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
---
|
| 26 |
+
<!-- trackio-cell
|
| 27 |
+
{"type": "markdown", "id": "cell_claim5_independent_finite_v4", "created_at": "2026-07-22T10:15:00+00:00", "title": "Exhaustive synthetic companion test"}
|
| 28 |
+
-->
|
| 29 |
+
The independent finite-DAG companion covers every ordered DAG with 1–5 nodes (1,099 graphs), risk values `{0.2,0.5,0.8}` including exact ties, five calibrated thresholds, and ancestor weights `{0.25,1,4}`. At finite temperature `T=0.002`, 9,535 stratified cases produced **0** hard/soft disagreements. [Executable audit and verification JSON](https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts#formal-v2)
|
| 30 |
|
| 31 |
|
| 32 |
---
|
| 33 |
<!-- trackio-cell
|
| 34 |
{"type": "markdown", "id": "cell_claim5_table_scope_v4", "created_at": "2026-07-22T10:15:01+00:00", "title": "Paper-table audit and exact limitation"}
|
| 35 |
-->
|
| 36 |
+
Table 2 itself reports agreement of 100.0% for α=0.01–0.03, followed by 99.8%, 93.8%, 93.9%, 95.4%, 91.5%, 90.2%, and 92.8% through α=0.10, so the stated range is arithmetically correct (90.2–100%). Our full-public-benchmark proxy reproduces the range but not those exact values; exact regeneration remains impossible because the submission omits annotated MATH graphs, fold predictions, trained weights, and implementation.
|
pages/conclusion/page.md
CHANGED
|
@@ -5,15 +5,20 @@
|
|
| 5 |
<!-- trackio-cell
|
| 6 |
{"type": "markdown", "id": "cell_conclusion_formal_v2", "created_at": "2026-07-21T21:05:04+00:00", "title": "Formal v2 bundle and rerun"}
|
| 7 |
-->
|
| 8 |
-
The
|
| 9 |
|
| 10 |
-
Download
|
| 11 |
|
| 12 |
```bash
|
|
|
|
|
|
|
| 13 |
tar -xzf dct_formal_reproduction_v2.tar.gz
|
| 14 |
python -m pip install -r requirements_formal_v2.txt
|
| 15 |
python exhaustive_prediction_gradient_audit.py \
|
| 16 |
--finite-stride 400 --gradient-repetitions 500
|
|
|
|
|
|
|
|
|
|
| 17 |
```
|
| 18 |
|
| 19 |
Expected invariants are zero failures across 3,813,795 exhaustive theorem cases and nonzero calibration, prediction, and ancestor gradients in all 500 composed graphs.
|
|
@@ -29,8 +34,8 @@ It must satisfy all four checks and converge to soft score `0.5` while the hard
|
|
| 29 |
|
| 30 |
---
|
| 31 |
<!-- trackio-cell
|
| 32 |
-
{"type": "artifact", "id": "cell_bundle_formal_v2", "created_at": "2026-07-21T21:05:05+00:00", "title": "
|
| 33 |
-->
|
| 34 |
-
**📦 Artifact** `outputs/
|
| 35 |
|
| 36 |
-
https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts
|
|
|
|
| 5 |
<!-- trackio-cell
|
| 6 |
{"type": "markdown", "id": "cell_conclusion_formal_v2", "created_at": "2026-07-21T21:05:04+00:00", "title": "Formal v2 bundle and rerun"}
|
| 7 |
-->
|
| 8 |
+
The v3 bundle combines an exact Theorem 3.1 boundary counterexample, a general Theorem 3.2 proof certificate, the exhaustive Theorem 3.2 checker, the 500-DAG composed-gradient test, and a new full-public-MATH agreement audit. It falsifies Claim 3 as written, strengthens Claims 4 and 6, and tests Claim 5 on 12,500 real MATH solutions / 57,661 proxy graph nodes across all ten α values. Exact-set agreement is 95.936–97.008%, inside the claimed range. Claims 1 and 2—and the exact paper-specific percentages in Claim 5—still require the unreleased annotated MATH/FELM ADGs, feature pipeline, fold assignments, checkpoints, and predictions.
|
| 9 |
|
| 10 |
+
Download the [v3 reproduction bundle](https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts/release-v3/dct_reproduction_v3.tar.gz), verify SHA-256 `4ebe525a2bdabcd862a45f37e84f6df1028a31e224e42f57cda4b215ba85c27a` with its [sidecar](https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts/release-v3/dct_reproduction_v3.tar.gz.sha256), and rerun:
|
| 11 |
|
| 12 |
```bash
|
| 13 |
+
tar -xzf dct_reproduction_v3.tar.gz
|
| 14 |
+
cd dct_reproduction_v3
|
| 15 |
tar -xzf dct_formal_reproduction_v2.tar.gz
|
| 16 |
python -m pip install -r requirements_formal_v2.txt
|
| 17 |
python exhaustive_prediction_gradient_audit.py \
|
| 18 |
--finite-stride 400 --gradient-repetitions 500
|
| 19 |
+
python math-agreement-v1/math_agreement_audit.py \
|
| 20 |
+
--data math-agreement-v1/competition_math.parquet \
|
| 21 |
+
--output math-agreement-v1/recomputed
|
| 22 |
```
|
| 23 |
|
| 24 |
Expected invariants are zero failures across 3,813,795 exhaustive theorem cases and nonzero calibration, prediction, and ancestor gradients in all 500 composed graphs.
|
|
|
|
| 34 |
|
| 35 |
---
|
| 36 |
<!-- trackio-cell
|
| 37 |
+
{"type": "artifact", "id": "cell_bundle_formal_v2", "created_at": "2026-07-21T21:05:05+00:00", "title": "DCF reproduction bundle v3", "path": "outputs/dct_reproduction_v3.tar.gz", "size": 4269142, "artifact_type": "reproduction bundle"}
|
| 38 |
-->
|
| 39 |
+
**📦 Artifact** `outputs/dct_reproduction_v3.tar.gz` · reproduction bundle · 4.1 MB · SHA-256 `4ebe525a2bdabcd862a45f37e84f6df1028a31e224e42f57cda4b215ba85c27a`
|
| 40 |
|
| 41 |
+
https://huggingface.co/buckets/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality-artifacts/release-v3/dct_reproduction_v3.tar.gz
|
pages/executive-summary/page.md
CHANGED
|
@@ -5,17 +5,17 @@
|
|
| 5 |
<!-- trackio-cell
|
| 6 |
{"type": "markdown", "id": "cell_564ab5e96189", "created_at": "2026-07-19T14:46:57+00:00", "title": "Executive summary", "pinned": true, "pinned_at": "2026-07-19T14:46:57+00:00"}
|
| 7 |
-->
|
| 8 |
-
Claim 3 is **falsified as written** by an exact boundary counterexample: the theorem permits a utility span equal to one, which ties a valid and invalid threshold and makes the soft score converge to 0.5 instead of the hard score 0. Claim 4
|
| 9 |
|
| 10 |
## Scope & cost
|
| 11 |
|
| 12 |
| | This reproduction | Full replication |
|
| 13 |
| --- | --- | --- |
|
| 14 |
-
| Scope | Theorem 3.1 analytic counterexample; general Theorem 3.2 proof + exhaustive audit; 500-DAG end-to-end gradient test | Rebuild annotated MATH/FELM ADGs,
|
| 15 |
| Hardware | Local CPU; existing HF Job retained as provenance | LLM/API annotation plus CPU/GPU training, unspecified by paper |
|
| 16 |
-
| Compute time | 3.81M theorem cases + 500 autograd graphs; no GPU | Not reported; likely hours plus annotation/API time |
|
| 17 |
| Cost | Negligible (<$0.01) | Unknown; unreleased pipeline prevents a reliable estimate |
|
| 18 |
-
| Outcome | Claim 3 falsified; Claim 4 proved; Claim 5
|
| 19 |
|
| 20 |
|
| 21 |
|
|
|
|
| 5 |
<!-- trackio-cell
|
| 6 |
{"type": "markdown", "id": "cell_564ab5e96189", "created_at": "2026-07-19T14:46:57+00:00", "title": "Executive summary", "pinned": true, "pinned_at": "2026-07-19T14:46:57+00:00"}
|
| 7 |
-->
|
| 8 |
+
Claim 3 is **falsified as written** by an exact boundary counterexample: the theorem permits a utility span equal to one, which ties a valid and invalid threshold and makes the soft score converge to 0.5 instead of the hard score 0. Claim 4 has an arbitrary-finite-DAG proof with the decisive boundary inequality `b>2`, corroborated by 18,983,745 exhaustive node decisions and zero failures. The full differentiable composition is exercised with autograd on 500 additional DAGs; calibration-to-loss, prediction-to-loss, and ancestor-to-descendant gradients are finite and nonzero in all 500. For Claim 5, a new benchmark-scale proxy processes all 12,500 public MATH solutions into 57,661 reasoning-step nodes and obtains 95.936–97.008% exact-set hard/soft agreement across α=0.01–0.10. The exact MATH/FELM gains and authors' 20-fold values remain unreproduced because no annotated graphs, folds, checkpoints, predictions, or training code are released.
|
| 9 |
|
| 10 |
## Scope & cost
|
| 11 |
|
| 12 |
| | This reproduction | Full replication |
|
| 13 |
| --- | --- | --- |
|
| 14 |
+
| Scope | Theorem 3.1 analytic counterexample; general Theorem 3.2 proof + exhaustive audit; 500-DAG end-to-end gradient test; full public MATH benchmark proxy (12,500 solutions, 57,661 nodes, 10 α values) | Rebuild authors' annotated MATH/FELM ADGs, learned scorer, 20 folds, training and evaluation |
|
| 15 |
| Hardware | Local CPU; existing HF Job retained as provenance | LLM/API annotation plus CPU/GPU training, unspecified by paper |
|
| 16 |
+
| Compute time | 3.81M theorem cases + 500 autograd graphs; full MATH agreement audit completes in ~23 s; no GPU | Not reported; likely hours plus annotation/API time |
|
| 17 |
| Cost | Negligible (<$0.01) | Unknown; unreleased pipeline prevents a reliable estimate |
|
| 18 |
+
| Outcome | Claim 3 falsified; Claim 4 proved; Claim 5 benchmark proxy falls in the claimed 90–100% band; Claim 6 end-to-end; Claims 1/2 and paper-specific trained-scorer runs unavailable | Not attempted |
|
| 19 |
|
| 20 |
|
| 21 |
|
reproduction/math_agreement_audit.py
ADDED
|
@@ -0,0 +1,211 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Benchmark-scale hard/soft agreement audit on all public MATH solutions.
|
| 3 |
+
|
| 4 |
+
The paper-specific Atomic Dependency Graph annotations and learned factuality
|
| 5 |
+
scorer are not public. This audit therefore makes a deliberately scoped proxy:
|
| 6 |
+
each gold solution is split into reasoning steps, the steps form a chain DAG,
|
| 7 |
+
and a deterministic structural risk score is assigned to each step. The DCF
|
| 8 |
+
prediction relaxation itself follows the equations audited for Theorem 3.2.
|
| 9 |
+
"""
|
| 10 |
+
|
| 11 |
+
from __future__ import annotations
|
| 12 |
+
|
| 13 |
+
import argparse
|
| 14 |
+
import csv
|
| 15 |
+
import hashlib
|
| 16 |
+
import json
|
| 17 |
+
import math
|
| 18 |
+
import re
|
| 19 |
+
from pathlib import Path
|
| 20 |
+
|
| 21 |
+
import numpy as np
|
| 22 |
+
import pyarrow.parquet as pq
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
DATASET = "https://huggingface.co/datasets/qwedsacf/competition_math"
|
| 26 |
+
DATASET_REVISION = "e839825f9ec5c6cfa585c654a59610969ec13993"
|
| 27 |
+
ALPHAS = tuple(round(i / 100, 2) for i in range(1, 11))
|
| 28 |
+
THRESHOLDS = tuple(round(i / 20, 2) for i in range(1, 20))
|
| 29 |
+
TEMPERATURE = 0.01
|
| 30 |
+
ANCESTOR_WEIGHT = 1.0
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
def sha256(path: Path) -> str:
|
| 34 |
+
digest = hashlib.sha256()
|
| 35 |
+
with path.open("rb") as handle:
|
| 36 |
+
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
| 37 |
+
digest.update(chunk)
|
| 38 |
+
return digest.hexdigest()
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def split_steps(solution: str) -> list[str]:
|
| 42 |
+
text = re.sub(r"\s+", " ", solution).strip()
|
| 43 |
+
parts = [part.strip() for part in re.split(r"(?<=[.!?])\s+|(?=\\boxed)|(?=\\Rightarrow)", text)]
|
| 44 |
+
parts = [part for part in parts if len(part) >= 8]
|
| 45 |
+
if not parts:
|
| 46 |
+
parts = [text or "empty"]
|
| 47 |
+
# Bound pathological TeX fragments while retaining full benchmark coverage.
|
| 48 |
+
if len(parts) > 24:
|
| 49 |
+
parts = parts[:23] + [" ".join(parts[23:])]
|
| 50 |
+
return parts
|
| 51 |
+
|
| 52 |
+
|
| 53 |
+
def structural_risk(step: str, row_index: int, step_index: int) -> float:
|
| 54 |
+
"""Deterministic, non-learned proxy risk in [0.05, 0.95]."""
|
| 55 |
+
tokens = re.findall(r"[A-Za-z0-9]+|\\[A-Za-z]+", step)
|
| 56 |
+
equation_density = min(1.0, (step.count("=") + step.count("\\")) / 12.0)
|
| 57 |
+
length_term = min(1.0, len(tokens) / 80.0)
|
| 58 |
+
conclusion_bonus = 1.0 if "boxed" in step or "therefore" in step.lower() else 0.0
|
| 59 |
+
digest = hashlib.blake2b(
|
| 60 |
+
f"{row_index}:{step_index}:{step}".encode(), digest_size=8
|
| 61 |
+
).digest()
|
| 62 |
+
jitter = int.from_bytes(digest, "big") / (2**64 - 1)
|
| 63 |
+
score = 0.10 + 0.42 * length_term + 0.28 * equation_density - 0.10 * conclusion_bonus + 0.20 * jitter
|
| 64 |
+
return min(0.95, max(0.05, score))
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def ancestors_for_chain(nodes: int) -> list[list[int]]:
|
| 68 |
+
return [list(range(node)) for node in range(nodes)]
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def log_sigmoid(value: float) -> float:
|
| 72 |
+
if value >= 0:
|
| 73 |
+
return -math.log1p(math.exp(-value))
|
| 74 |
+
return value - math.log1p(math.exp(value))
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def hard_prediction(risks: list[float], ancestors: list[list[int]], tau_alpha: float) -> list[int]:
|
| 78 |
+
tau_star = max(tau for tau in THRESHOLDS if tau < tau_alpha)
|
| 79 |
+
return [
|
| 80 |
+
int(risks[node] <= tau_star and all(risks[parent] <= tau_star for parent in ancestors[node]))
|
| 81 |
+
for node in range(len(risks))
|
| 82 |
+
]
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def soft_prediction(risks: list[float], ancestors: list[list[int]], tau_alpha: float) -> tuple[list[int], float]:
|
| 86 |
+
# Theorem 3.2 schedule: a=1, b=3, beta=T^-a, tau_z=T^(ab).
|
| 87 |
+
beta = 1.0 / TEMPERATURE
|
| 88 |
+
tau_z = TEMPERATURE**3
|
| 89 |
+
log_weights = [
|
| 90 |
+
beta * tau + log_sigmoid((tau_alpha - tau - math.sqrt(tau_z)) / tau_z)
|
| 91 |
+
for tau in THRESHOLDS
|
| 92 |
+
]
|
| 93 |
+
maximum = max(log_weights)
|
| 94 |
+
weights = np.exp(np.asarray(log_weights) - maximum)
|
| 95 |
+
weights /= weights.sum()
|
| 96 |
+
memberships = []
|
| 97 |
+
for tau in THRESHOLDS:
|
| 98 |
+
probabilities = [
|
| 99 |
+
1.0 / (1.0 + math.exp(max(-700.0, min(700.0, (risk - tau) / TEMPERATURE))))
|
| 100 |
+
for risk in risks
|
| 101 |
+
]
|
| 102 |
+
coherent = []
|
| 103 |
+
for node, parent_nodes in enumerate(ancestors):
|
| 104 |
+
indices = [node] + parent_nodes
|
| 105 |
+
node_weights = [1.0] + [ANCESTOR_WEIGHT] * len(parent_nodes)
|
| 106 |
+
denominator = sum(node_weights)
|
| 107 |
+
coherent.append(
|
| 108 |
+
math.exp(
|
| 109 |
+
sum(
|
| 110 |
+
weight * math.log(max(probabilities[index], 1e-300))
|
| 111 |
+
for index, weight in zip(indices, node_weights)
|
| 112 |
+
)
|
| 113 |
+
/ denominator
|
| 114 |
+
)
|
| 115 |
+
)
|
| 116 |
+
memberships.append(coherent)
|
| 117 |
+
mixed = weights @ np.asarray(memberships)
|
| 118 |
+
return [int(value >= 0.5) for value in mixed], float(weights.max())
|
| 119 |
+
|
| 120 |
+
|
| 121 |
+
def main() -> None:
|
| 122 |
+
parser = argparse.ArgumentParser()
|
| 123 |
+
parser.add_argument("--data", type=Path, default=Path("competition_math.parquet"))
|
| 124 |
+
parser.add_argument("--output", type=Path, default=Path("outputs/math_agreement_v1"))
|
| 125 |
+
args = parser.parse_args()
|
| 126 |
+
args.output.mkdir(parents=True, exist_ok=True)
|
| 127 |
+
|
| 128 |
+
table = pq.read_table(args.data, columns=["solution", "level", "type"])
|
| 129 |
+
records = table.to_pylist()
|
| 130 |
+
examples = []
|
| 131 |
+
all_risks = []
|
| 132 |
+
step_counts = []
|
| 133 |
+
for row_index, record in enumerate(records):
|
| 134 |
+
steps = split_steps(record["solution"])
|
| 135 |
+
risks = [structural_risk(step, row_index, step_index) for step_index, step in enumerate(steps)]
|
| 136 |
+
examples.append((risks, ancestors_for_chain(len(risks)), record["level"], record["type"]))
|
| 137 |
+
all_risks.extend(risks)
|
| 138 |
+
step_counts.append(len(steps))
|
| 139 |
+
|
| 140 |
+
output_rows = []
|
| 141 |
+
for alpha in ALPHAS:
|
| 142 |
+
# Quantile calibration is performed once over all public solution-step risks.
|
| 143 |
+
tau_alpha = float(np.quantile(np.asarray(all_risks), 1.0 - alpha, method="higher"))
|
| 144 |
+
# Ensure the prediction gate has at least one feasible threshold and is
|
| 145 |
+
# not exactly on a grid point, as in the theorem's separated regime.
|
| 146 |
+
tau_alpha = min(0.999, max(0.051, tau_alpha + 1e-7))
|
| 147 |
+
exact_matches = 0
|
| 148 |
+
node_matches = 0
|
| 149 |
+
node_total = 0
|
| 150 |
+
hard_retained = 0
|
| 151 |
+
soft_retained = 0
|
| 152 |
+
minimum_selected_mass = 1.0
|
| 153 |
+
for risks, ancestors, _, _ in examples:
|
| 154 |
+
hard = hard_prediction(risks, ancestors, tau_alpha)
|
| 155 |
+
soft, selected_mass = soft_prediction(risks, ancestors, tau_alpha)
|
| 156 |
+
exact_matches += int(hard == soft)
|
| 157 |
+
node_matches += sum(int(left == right) for left, right in zip(hard, soft))
|
| 158 |
+
node_total += len(hard)
|
| 159 |
+
hard_retained += sum(hard)
|
| 160 |
+
soft_retained += sum(soft)
|
| 161 |
+
minimum_selected_mass = min(minimum_selected_mass, selected_mass)
|
| 162 |
+
output_rows.append(
|
| 163 |
+
{
|
| 164 |
+
"alpha": alpha,
|
| 165 |
+
"calibrated_tau_alpha": tau_alpha,
|
| 166 |
+
"examples": len(examples),
|
| 167 |
+
"claim_nodes": node_total,
|
| 168 |
+
"exact_set_matches": exact_matches,
|
| 169 |
+
"exact_set_agreement_percent": 100.0 * exact_matches / len(examples),
|
| 170 |
+
"node_matches": node_matches,
|
| 171 |
+
"node_agreement_percent": 100.0 * node_matches / node_total,
|
| 172 |
+
"hard_retained_nodes": hard_retained,
|
| 173 |
+
"soft_retained_nodes": soft_retained,
|
| 174 |
+
"minimum_selected_threshold_mass": minimum_selected_mass,
|
| 175 |
+
}
|
| 176 |
+
)
|
| 177 |
+
|
| 178 |
+
with (args.output / "agreement_by_alpha.csv").open("w", newline="") as handle:
|
| 179 |
+
writer = csv.DictWriter(handle, fieldnames=list(output_rows[0]))
|
| 180 |
+
writer.writeheader()
|
| 181 |
+
writer.writerows(output_rows)
|
| 182 |
+
report = {
|
| 183 |
+
"claim": "DCF soft/hard agreement across alpha=0.01..0.10",
|
| 184 |
+
"scope": "benchmark-scale proxy; not the unreleased paper scorer/ADGs",
|
| 185 |
+
"dataset": DATASET,
|
| 186 |
+
"dataset_revision": DATASET_REVISION,
|
| 187 |
+
"dataset_file_sha256": sha256(args.data),
|
| 188 |
+
"examples": len(examples),
|
| 189 |
+
"solution_claim_nodes": sum(step_counts),
|
| 190 |
+
"step_count": {
|
| 191 |
+
"minimum": min(step_counts),
|
| 192 |
+
"median": float(np.median(step_counts)),
|
| 193 |
+
"maximum": max(step_counts),
|
| 194 |
+
},
|
| 195 |
+
"graph_construction": "sentence/TeX-step split; sequential chain with transitive ancestors",
|
| 196 |
+
"risk_proxy": "deterministic structural features plus pinned BLAKE2 jitter; not learned factuality",
|
| 197 |
+
"threshold_grid": list(THRESHOLDS),
|
| 198 |
+
"temperature": TEMPERATURE,
|
| 199 |
+
"ancestor_weight": ANCESTOR_WEIGHT,
|
| 200 |
+
"rows": output_rows,
|
| 201 |
+
"minimum_exact_set_agreement_percent": min(row["exact_set_agreement_percent"] for row in output_rows),
|
| 202 |
+
"maximum_exact_set_agreement_percent": max(row["exact_set_agreement_percent"] for row in output_rows),
|
| 203 |
+
"minimum_node_agreement_percent": min(row["node_agreement_percent"] for row in output_rows),
|
| 204 |
+
"maximum_node_agreement_percent": max(row["node_agreement_percent"] for row in output_rows),
|
| 205 |
+
}
|
| 206 |
+
(args.output / "summary.json").write_text(json.dumps(report, indent=2) + "\n")
|
| 207 |
+
print(json.dumps(report, indent=2))
|
| 208 |
+
|
| 209 |
+
|
| 210 |
+
if __name__ == "__main__":
|
| 211 |
+
main()
|