repro-differentiable-conformal-training-for-llm-reasoning-factuality / outputs /calibration_limit_summary.json
Download outputs/calibration_limit_summary.json from SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality: direct link, hf CLI and curl.
- Browser
- Download file 5.32 kB
-
https://huggingface.co/spaces/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality/resolve/main/outputs/calibration_limit_summary.json
- Command line
-
hf download hf://spaces/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality/outputs/calibration_limit_summary.json
-
curl -L -o calibration_limit_summary.json https://huggingface.co/spaces/SabaPivot/repro-differentiable-conformal-training-for-llm-reasoning-factuality/resolve/main/outputs/calibration_limit_summary.json
5.32 kB
| { | |
| "destructive_controls": { | |
| "fixed_beta_8": { | |
| "all_finite": true, | |
| "graphs_changed_beyond_1e-6": 50, | |
| "maximum_absolute_error": 1.8368553314178566, | |
| "mean_absolute_error": 0.613734691169597 | |
| }, | |
| "removed_violation_penalty": { | |
| "all_finite": true, | |
| "graphs_changed_beyond_1e-6": 13, | |
| "maximum_absolute_error": 10.0, | |
| "mean_absolute_error": 1.21 | |
| } | |
| }, | |
| "execution_scope": { | |
| "alpha_grid": [ | |
| 0.01, | |
| 0.02, | |
| 0.03, | |
| 0.04, | |
| 0.05, | |
| 0.06, | |
| 0.07, | |
| 0.08, | |
| 0.09, | |
| 0.1, | |
| 0.11, | |
| 0.12, | |
| 0.13, | |
| 0.14, | |
| 0.15 | |
| ], | |
| "claim_nodes": 503, | |
| "dataset": "released MATH_open_subclaims_with_scores_and_semantic_eval.json", | |
| "dependency_edges": 496, | |
| "graph_temperature_evaluations": 450, | |
| "graphs": 50, | |
| "graphs_with_false_nodes": 13, | |
| "quantile_evaluations": 135, | |
| "temperature_schedule": [ | |
| 0.5, | |
| 0.2, | |
| 0.1, | |
| 0.05, | |
| 0.02, | |
| 0.01, | |
| 0.005, | |
| 0.002, | |
| 0.001 | |
| ] | |
| }, | |
| "final_temperature_result": { | |
| "beta": 1000.0, | |
| "graphs_within_1e-10": 50, | |
| "graphs_within_1e-6": 50, | |
| "maximum_absolute_error": 4.32542890393961e-13, | |
| "mean_absolute_error": 8.92619311798626e-15, | |
| "tau_s": 0.03162277660168379, | |
| "temperature": 0.001 | |
| }, | |
| "gates": { | |
| "all_15_final_quantiles_within_1e-10": true, | |
| "all_reported_values_finite": true, | |
| "final_all_50_within_1e-10": true, | |
| "final_max_error_below_1e-10": true, | |
| "fixed_beta_destructive_control_fails": true, | |
| "lambda_grid_span_exactly_one_half": true, | |
| "mean_error_contracts_by_1e8": true, | |
| "official_dataset_sha256_exact": true, | |
| "official_release_scale_50_graphs_503_nodes": true, | |
| "quantile_order_statistic_stability_bound": true, | |
| "removed_violation_destructive_control_fails": true, | |
| "theorem_source_markers_exact": true | |
| }, | |
| "no_paper_scale_rerun_invented": true, | |
| "official_commit": "0b4d5487a9868a18c4f9aa5b3d96cdccc705ca97", | |
| "paper_id": "XfndtVLIub", | |
| "quantile_recovery": { | |
| "alphas": 15, | |
| "final_maximum_absolute_error": 0.0, | |
| "order_statistic_stability_bound_holds_all_cells": true, | |
| "scope_note": "Quantile recovery is certified as an order-statistic corollary of uniform score convergence; it is not attributed to the theorem text alone." | |
| }, | |
| "registered_claim": "Theorem 3.1 calibration convergence and conformal quantile recovery", | |
| "source_checks": { | |
| "lambda_grid_span": true, | |
| "score_conclusion": true, | |
| "single_limit_schedule": true, | |
| "sqrt_margin": true | |
| }, | |
| "status": "pass", | |
| "temperature_results": [ | |
| { | |
| "beta": 2.0, | |
| "graphs_within_1e-10": 0, | |
| "graphs_within_1e-6": 0, | |
| "maximum_absolute_error": 5.023173588989156, | |
| "mean_absolute_error": 1.5931000670957345, | |
| "tau_s": 0.7071067811865476, | |
| "temperature": 0.5 | |
| }, | |
| { | |
| "beta": 5.0, | |
| "graphs_within_1e-10": 0, | |
| "graphs_within_1e-6": 0, | |
| "maximum_absolute_error": 5.358487725428276, | |
| "mean_absolute_error": 1.1243537537855317, | |
| "tau_s": 0.4472135954999579, | |
| "temperature": 0.2 | |
| }, | |
| { | |
| "beta": 10.0, | |
| "graphs_within_1e-10": 0, | |
| "graphs_within_1e-6": 0, | |
| "maximum_absolute_error": 4.13099223342072, | |
| "mean_absolute_error": 0.6374636678771154, | |
| "tau_s": 0.31622776601683794, | |
| "temperature": 0.1 | |
| }, | |
| { | |
| "beta": 20.0, | |
| "graphs_within_1e-10": 0, | |
| "graphs_within_1e-6": 0, | |
| "maximum_absolute_error": 1.9926415571154799, | |
| "mean_absolute_error": 0.1997130122709958, | |
| "tau_s": 0.22360679774997896, | |
| "temperature": 0.05 | |
| }, | |
| { | |
| "beta": 50.0, | |
| "graphs_within_1e-10": 1, | |
| "graphs_within_1e-6": 2, | |
| "maximum_absolute_error": 1.4585052946539498, | |
| "mean_absolute_error": 0.043990007453380324, | |
| "tau_s": 0.1414213562373095, | |
| "temperature": 0.02 | |
| }, | |
| { | |
| "beta": 100.0, | |
| "graphs_within_1e-10": 9, | |
| "graphs_within_1e-6": 23, | |
| "maximum_absolute_error": 0.02927062476515907, | |
| "mean_absolute_error": 0.0016116510764306513, | |
| "tau_s": 0.1, | |
| "temperature": 0.01 | |
| }, | |
| { | |
| "beta": 200.0, | |
| "graphs_within_1e-10": 29, | |
| "graphs_within_1e-6": 43, | |
| "maximum_absolute_error": 0.001925516180114606, | |
| "mean_absolute_error": 6.329592270834183e-05, | |
| "tau_s": 0.07071067811865475, | |
| "temperature": 0.005 | |
| }, | |
| { | |
| "beta": 500.0, | |
| "graphs_within_1e-10": 47, | |
| "graphs_within_1e-6": 50, | |
| "maximum_absolute_error": 4.6462416758430436e-07, | |
| "mean_absolute_error": 1.0947915756176485e-08, | |
| "tau_s": 0.044721359549995794, | |
| "temperature": 0.002 | |
| }, | |
| { | |
| "beta": 1000.0, | |
| "graphs_within_1e-10": 50, | |
| "graphs_within_1e-6": 50, | |
| "maximum_absolute_error": 4.32542890393961e-13, | |
| "mean_absolute_error": 8.92619311798626e-15, | |
| "tau_s": 0.03162277660168379, | |
| "temperature": 0.001 | |
| } | |
| ], | |
| "theorem_contract": { | |
| "beta": "T^-1", | |
| "hard_oracle": "largest threshold whose selected ancestor-coherent subgraph contains no false node", | |
| "lambda_grid_span": 0.5, | |
| "required_upper_bound": 1.0, | |
| "soft_keep": "sigmoid((tau-risk+sqrt(T))/T)", | |
| "soft_sort_shim_used_in_reported_path": false, | |
| "tau_s": "T^0.5" | |
| } | |
| } | |